gento 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/CHANGELOG.md +60 -0
- data/LICENSE.txt +21 -0
- data/README.md +143 -0
- data/exe/gento +7 -0
- data/lib/gento/adapters/base.rb +121 -0
- data/lib/gento/adapters/docswell.rb +58 -0
- data/lib/gento/adapters/google_slides.rb +90 -0
- data/lib/gento/adapters/slide_share.rb +82 -0
- data/lib/gento/adapters/speaker_deck.rb +79 -0
- data/lib/gento/charset.rb +45 -0
- data/lib/gento/cli.rb +121 -0
- data/lib/gento/client.rb +55 -0
- data/lib/gento/deck.rb +56 -0
- data/lib/gento/entities.rb +29 -0
- data/lib/gento/errors.rb +46 -0
- data/lib/gento/fetcher.rb +188 -0
- data/lib/gento/html.rb +225 -0
- data/lib/gento/registry.rb +49 -0
- data/lib/gento/response.rb +34 -0
- data/lib/gento/robots.rb +195 -0
- data/lib/gento/robots_fetcher.rb +126 -0
- data/lib/gento/slide.rb +40 -0
- data/lib/gento/version.rb +5 -0
- data/lib/gento.rb +44 -0
- metadata +75 -0
data/lib/gento/html.rb
ADDED
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "strscan"
|
|
4
|
+
require "cgi"
|
|
5
|
+
require "json"
|
|
6
|
+
|
|
7
|
+
module Gento
|
|
8
|
+
# A deliberately small HTML scanner.
|
|
9
|
+
#
|
|
10
|
+
# This is not a DOM parser and does not try to be one. It finds tags by name,
|
|
11
|
+
# reads their attributes, and grabs the raw text of non-void elements, which
|
|
12
|
+
# covers everything the adapters need (meta, link, script, img, title).
|
|
13
|
+
#
|
|
14
|
+
# Why not Nokogiri: it ships a native extension. Staying on pure Ruby keeps
|
|
15
|
+
# the gem usable from slim containers and from ruby.wasm, where compiling
|
|
16
|
+
# libxml is anywhere from painful to impossible.
|
|
17
|
+
class Html
|
|
18
|
+
VOID_ELEMENTS = %w[
|
|
19
|
+
area base br col embed hr img input link meta param source track wbr
|
|
20
|
+
].freeze
|
|
21
|
+
|
|
22
|
+
attr_reader :source
|
|
23
|
+
|
|
24
|
+
def initialize(source)
|
|
25
|
+
# Defensive: adapters are the only caller in practice and Response has
|
|
26
|
+
# already normalized, but a caller handing us a raw body should not blow
|
|
27
|
+
# up mid-scan with an encoding error.
|
|
28
|
+
@source = Charset.normalize(source)
|
|
29
|
+
@tags = {}
|
|
30
|
+
@elements = {}
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
# Attribute hashes for every occurrence of `name`, in document order.
|
|
34
|
+
#
|
|
35
|
+
# Unlike #elements this never consumes element content, so it still finds
|
|
36
|
+
# tags nested inside another tag of the same name — which ordinary page
|
|
37
|
+
# markup is full of.
|
|
38
|
+
def tags(name)
|
|
39
|
+
@tags[name.to_s.downcase] ||= scan_tags(name.to_s)
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
# [attributes, inner_html] pairs for every occurrence of `name`.
|
|
43
|
+
#
|
|
44
|
+
# Only useful for elements that do not nest, such as <script> and <title>;
|
|
45
|
+
# for anything else use #tags.
|
|
46
|
+
def elements(name)
|
|
47
|
+
@elements[name.to_s.downcase] ||= scan_elements(name.to_s)
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
# Content of the first <meta> whose name/property/itemprop matches any key.
|
|
51
|
+
def meta(*keys)
|
|
52
|
+
meta_all(*keys).first
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
# Content of every matching <meta>, for properties that legitimately repeat.
|
|
56
|
+
def meta_all(*keys)
|
|
57
|
+
wanted = keys.flatten.map { |key| key.to_s.downcase }
|
|
58
|
+
tags("meta").filter_map { |attrs| meta_content(attrs, wanted) }
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
# href of the first <link> carrying the given rel token.
|
|
62
|
+
def link(rel)
|
|
63
|
+
wanted = rel.to_s.downcase
|
|
64
|
+
tags("link").each do |attrs|
|
|
65
|
+
tokens = attrs["rel"].to_s.downcase.split(/\s+/)
|
|
66
|
+
return attrs["href"] if tokens.include?(wanted)
|
|
67
|
+
end
|
|
68
|
+
nil
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
def title
|
|
72
|
+
inner = elements("title").first&.last
|
|
73
|
+
inner && decode_entities(inner.strip)
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
# Every parsed application/ld+json block, with @graph arrays flattened.
|
|
77
|
+
def json_ld
|
|
78
|
+
documents = elements("script").filter_map do |attrs, inner|
|
|
79
|
+
next unless attrs["type"].to_s.downcase.include?("ld+json")
|
|
80
|
+
|
|
81
|
+
parse_json(inner)
|
|
82
|
+
end
|
|
83
|
+
documents.flat_map { |doc| flatten_json_ld(doc) }
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
# Raw bodies of inline <script> tags, for sites that stash a JSON blob there.
|
|
87
|
+
def inline_scripts
|
|
88
|
+
elements("script").filter_map do |attrs, inner|
|
|
89
|
+
next if attrs.key?("src")
|
|
90
|
+
|
|
91
|
+
inner
|
|
92
|
+
end
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
# Any absolute URL appearing anywhere in the document, in document order,
|
|
96
|
+
# entity-decoded and deduplicated.
|
|
97
|
+
#
|
|
98
|
+
# Slide hosts put page images wherever suits them: Speaker Deck uses
|
|
99
|
+
# <a href>, Docswell a lazy-loading data attribute, Google Slides a CSS
|
|
100
|
+
# background inside a style attribute. Matching the URL shape across the
|
|
101
|
+
# raw document handles all three, and keeps working when a site moves its
|
|
102
|
+
# images from one element to another.
|
|
103
|
+
ABSOLUTE_URL = %r{https?://[^\s"'<>\\)]+}
|
|
104
|
+
|
|
105
|
+
def urls(pattern = //)
|
|
106
|
+
source.scan(ABSOLUTE_URL)
|
|
107
|
+
.map { |url| decode_entities(url) }
|
|
108
|
+
.grep(pattern)
|
|
109
|
+
.uniq
|
|
110
|
+
end
|
|
111
|
+
|
|
112
|
+
def decode_entities(text)
|
|
113
|
+
Entities.decode(text)
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
private
|
|
117
|
+
|
|
118
|
+
# A <meta> carries its key under any of three attributes depending on the
|
|
119
|
+
# vocabulary in use; an empty content is treated as absent.
|
|
120
|
+
def meta_content(attrs, wanted)
|
|
121
|
+
key = attrs["property"] || attrs["name"] || attrs["itemprop"]
|
|
122
|
+
return nil unless key && wanted.include?(key.downcase)
|
|
123
|
+
|
|
124
|
+
content = attrs["content"]
|
|
125
|
+
content unless content.nil? || content.empty?
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
def scan_tags(name)
|
|
129
|
+
results = []
|
|
130
|
+
scanner = StringScanner.new(@source)
|
|
131
|
+
opening = opening_pattern(name)
|
|
132
|
+
|
|
133
|
+
while scanner.skip_until(opening)
|
|
134
|
+
attributes, = scan_attributes(scanner)
|
|
135
|
+
results << attributes
|
|
136
|
+
end
|
|
137
|
+
|
|
138
|
+
results
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
def scan_elements(name)
|
|
142
|
+
results = []
|
|
143
|
+
scanner = StringScanner.new(@source)
|
|
144
|
+
opening = opening_pattern(name)
|
|
145
|
+
void = VOID_ELEMENTS.include?(name.downcase)
|
|
146
|
+
|
|
147
|
+
while scanner.skip_until(opening)
|
|
148
|
+
attributes, self_closing = scan_attributes(scanner)
|
|
149
|
+
inner = void || self_closing ? nil : scan_inner(scanner, name)
|
|
150
|
+
results << [attributes, inner]
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
results
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
def opening_pattern(name)
|
|
157
|
+
%r{<#{Regexp.escape(name)}(?=[\s>/])}i
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
# Consumes the attribute list and the closing `>` of an open tag.
|
|
161
|
+
def scan_attributes(scanner)
|
|
162
|
+
attributes = {}
|
|
163
|
+
|
|
164
|
+
loop do
|
|
165
|
+
scanner.skip(/\s+/)
|
|
166
|
+
return [attributes, true] if scanner.scan(%r{/\s*>})
|
|
167
|
+
return [attributes, false] if scanner.scan(">")
|
|
168
|
+
return [attributes, false] if scanner.eos?
|
|
169
|
+
|
|
170
|
+
name = scanner.scan(%r{[^\s=/>]+})
|
|
171
|
+
# Never stall: an unexpected byte here would otherwise loop forever.
|
|
172
|
+
next scanner.getch if name.nil?
|
|
173
|
+
|
|
174
|
+
scanner.skip(/\s*/)
|
|
175
|
+
attributes[name.downcase] = decode_entities(scan_attribute_value(scanner))
|
|
176
|
+
end
|
|
177
|
+
end
|
|
178
|
+
|
|
179
|
+
def scan_attribute_value(scanner)
|
|
180
|
+
return "" unless scanner.scan(/=\s*/)
|
|
181
|
+
|
|
182
|
+
if scanner.scan(/"([^"]*)"/) || scanner.scan(/'([^']*)'/)
|
|
183
|
+
scanner[1]
|
|
184
|
+
else
|
|
185
|
+
scanner.scan(/[^\s>]*/).to_s
|
|
186
|
+
end
|
|
187
|
+
end
|
|
188
|
+
|
|
189
|
+
# Returns the raw text up to the matching close tag.
|
|
190
|
+
#
|
|
191
|
+
# StringScanner#pos counts bytes while String#[] counts characters, so
|
|
192
|
+
# slicing has to go through byteslice or every multibyte document comes
|
|
193
|
+
# back mangled. Tag boundaries are ASCII, so byte slicing is safe here.
|
|
194
|
+
def scan_inner(scanner, name)
|
|
195
|
+
start = scanner.pos
|
|
196
|
+
closing = %r{</#{Regexp.escape(name)}\s*>}i
|
|
197
|
+
|
|
198
|
+
unless scanner.skip_until(closing)
|
|
199
|
+
scanner.terminate
|
|
200
|
+
return byteslice_from(start, @source.bytesize - start)
|
|
201
|
+
end
|
|
202
|
+
|
|
203
|
+
byteslice_from(start, scanner.pos - scanner.matched_size - start)
|
|
204
|
+
end
|
|
205
|
+
|
|
206
|
+
def byteslice_from(start, length)
|
|
207
|
+
slice = @source.byteslice(start, length).to_s
|
|
208
|
+
slice.valid_encoding? ? slice : slice.scrub
|
|
209
|
+
end
|
|
210
|
+
|
|
211
|
+
def parse_json(text)
|
|
212
|
+
JSON.parse(text.to_s)
|
|
213
|
+
rescue JSON::ParserError
|
|
214
|
+
nil
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
def flatten_json_ld(doc)
|
|
218
|
+
case doc
|
|
219
|
+
when Array then doc.flat_map { |entry| flatten_json_ld(entry) }
|
|
220
|
+
when Hash then doc.key?("@graph") ? flatten_json_ld(doc["@graph"]) + [doc] : [doc]
|
|
221
|
+
else []
|
|
222
|
+
end
|
|
223
|
+
end
|
|
224
|
+
end
|
|
225
|
+
end
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Gento
|
|
4
|
+
# Maps a URL to the adapter that claims it.
|
|
5
|
+
#
|
|
6
|
+
# Adding support for a new slide host is one class plus one `register` call;
|
|
7
|
+
# nothing else in the library needs to know the site exists.
|
|
8
|
+
class Registry
|
|
9
|
+
class << self
|
|
10
|
+
def default
|
|
11
|
+
@default ||= new(
|
|
12
|
+
[
|
|
13
|
+
Adapters::SpeakerDeck,
|
|
14
|
+
Adapters::SlideShare,
|
|
15
|
+
Adapters::Docswell,
|
|
16
|
+
Adapters::GoogleSlides
|
|
17
|
+
]
|
|
18
|
+
)
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
attr_writer :default
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
def initialize(adapters = [])
|
|
25
|
+
@adapters = adapters.dup
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def adapters
|
|
29
|
+
@adapters.dup.freeze
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
def register(adapter)
|
|
33
|
+
@adapters.unshift(adapter)
|
|
34
|
+
self
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
def find(url)
|
|
38
|
+
@adapters.find { |adapter| adapter.handles?(url) }
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
def supports?(url)
|
|
42
|
+
!find(url).nil?
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
def providers
|
|
46
|
+
@adapters.map(&:provider)
|
|
47
|
+
end
|
|
48
|
+
end
|
|
49
|
+
end
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Gento
|
|
4
|
+
# A minimal HTTP response, so adapters never touch a fetcher-specific object.
|
|
5
|
+
class Response
|
|
6
|
+
attr_reader :status, :headers, :body, :url
|
|
7
|
+
|
|
8
|
+
def initialize(status:, body:, url:, headers: {})
|
|
9
|
+
@status = Integer(status)
|
|
10
|
+
@url = url.to_s
|
|
11
|
+
@headers = headers.transform_keys { |key| key.to_s.downcase }.freeze
|
|
12
|
+
@body = Charset.normalize(body, content_type: @headers["content-type"])
|
|
13
|
+
freeze
|
|
14
|
+
end
|
|
15
|
+
|
|
16
|
+
def success?
|
|
17
|
+
status.between?(200, 299)
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
def content_type
|
|
21
|
+
headers["content-type"]
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
def json
|
|
25
|
+
JSON.parse(body)
|
|
26
|
+
rescue JSON::ParserError => e
|
|
27
|
+
raise ExtractionError, "expected JSON from #{url}: #{e.message}"
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
def html
|
|
31
|
+
Html.new(body)
|
|
32
|
+
end
|
|
33
|
+
end
|
|
34
|
+
end
|
data/lib/gento/robots.rb
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Gento
|
|
4
|
+
# A parsed robots.txt, following RFC 9309.
|
|
5
|
+
#
|
|
6
|
+
# The one rule worth implementing carefully is precedence: **the longest
|
|
7
|
+
# matching pattern wins**, not the first one. Google's robots.txt ends with
|
|
8
|
+
#
|
|
9
|
+
# Allow: /presentation
|
|
10
|
+
# Disallow: /
|
|
11
|
+
#
|
|
12
|
+
# and a first-match reading would conclude that every Google Slides deck is
|
|
13
|
+
# off limits, which is the opposite of what that file says.
|
|
14
|
+
class Robots
|
|
15
|
+
# A single Allow or Disallow line.
|
|
16
|
+
class Rule
|
|
17
|
+
attr_reader :pattern
|
|
18
|
+
|
|
19
|
+
def initialize(pattern, allow:)
|
|
20
|
+
@pattern = pattern
|
|
21
|
+
@allow = allow
|
|
22
|
+
@regexp = self.class.compile(pattern)
|
|
23
|
+
freeze
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
def allow?
|
|
27
|
+
@allow
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
def covers?(path)
|
|
31
|
+
@regexp.match?(path)
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# RFC 9309 orders rules by the length of the pattern, wildcards included.
|
|
35
|
+
def specificity
|
|
36
|
+
pattern.length
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
# `*` stands for any run of characters and `$` anchors the end; every
|
|
40
|
+
# other character is literal.
|
|
41
|
+
def self.compile(pattern)
|
|
42
|
+
anchored = pattern.end_with?("$")
|
|
43
|
+
body = anchored ? pattern[0..-2] : pattern
|
|
44
|
+
source = body.split("*", -1).map { |part| Regexp.escape(part) }.join(".*")
|
|
45
|
+
source += "\\z" if anchored
|
|
46
|
+
|
|
47
|
+
Regexp.new("\\A#{source}")
|
|
48
|
+
end
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
Group = Struct.new(:agents, :rules, :crawl_delay)
|
|
52
|
+
|
|
53
|
+
DIRECTIVES = %w[user-agent allow disallow crawl-delay].freeze
|
|
54
|
+
# A product token runs to the first slash or space: the "gento" of
|
|
55
|
+
# "gento/0.1.0 (+https://...)".
|
|
56
|
+
PRODUCT_TOKEN = %r{\A[^\s/]+}
|
|
57
|
+
|
|
58
|
+
attr_reader :groups, :reason
|
|
59
|
+
|
|
60
|
+
def initialize(groups = [], reason: nil)
|
|
61
|
+
@groups = groups.freeze
|
|
62
|
+
@reason = reason
|
|
63
|
+
freeze
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
class << self
|
|
67
|
+
def parse(body)
|
|
68
|
+
new(Parser.new(body).groups)
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
# No robots.txt to obey — the host said so with a 404, or answered with
|
|
72
|
+
# something that is not a robots.txt at all.
|
|
73
|
+
def allow_all
|
|
74
|
+
new
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
# RFC 9309 §2.3.1.4: a robots.txt that cannot be read is not permission
|
|
78
|
+
# to crawl. `reason` explains which failure it was.
|
|
79
|
+
def disallow_all(reason: nil)
|
|
80
|
+
new([Group.new(["*"], [Rule.new("/", allow: false)], nil)], reason: reason)
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
def product_token(user_agent)
|
|
84
|
+
user_agent.to_s[PRODUCT_TOKEN].to_s.downcase
|
|
85
|
+
end
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
def allowed?(path, agent: nil)
|
|
89
|
+
group = group_for(agent)
|
|
90
|
+
return true unless group
|
|
91
|
+
|
|
92
|
+
matching = group.rules.select { |rule| rule.covers?(path) }
|
|
93
|
+
return true if matching.empty?
|
|
94
|
+
|
|
95
|
+
# Longest pattern wins; an Allow and a Disallow of equal length are a
|
|
96
|
+
# tie the site did not resolve, and RFC 9309 gives those to Allow.
|
|
97
|
+
matching.max_by { |rule| [rule.specificity, rule.allow? ? 1 : 0] }.allow?
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
def crawl_delay(agent: nil)
|
|
101
|
+
group_for(agent)&.crawl_delay
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
def empty?
|
|
105
|
+
groups.empty?
|
|
106
|
+
end
|
|
107
|
+
|
|
108
|
+
private
|
|
109
|
+
|
|
110
|
+
# A crawler obeys exactly one group: the one naming it most specifically,
|
|
111
|
+
# falling back to `*`.
|
|
112
|
+
def group_for(agent)
|
|
113
|
+
token = self.class.product_token(agent)
|
|
114
|
+
|
|
115
|
+
groups.filter_map { |group| [group, specificity(group, token)] if applies?(group, token) }
|
|
116
|
+
.max_by(&:last)
|
|
117
|
+
&.first
|
|
118
|
+
end
|
|
119
|
+
|
|
120
|
+
def applies?(group, token)
|
|
121
|
+
group.agents.any? { |name| name == "*" || (!token.empty? && token.start_with?(name)) }
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
def specificity(group, token)
|
|
125
|
+
group.agents.filter_map { |name| name.length if name != "*" && token.start_with?(name) }
|
|
126
|
+
.max || 0
|
|
127
|
+
end
|
|
128
|
+
|
|
129
|
+
# robots.txt is a line-oriented `key: value` format. Anything unrecognised
|
|
130
|
+
# is skipped rather than treated as an error, which is also what keeps a
|
|
131
|
+
# page served in place of a robots.txt from parsing into rules.
|
|
132
|
+
class Parser
|
|
133
|
+
attr_reader :groups
|
|
134
|
+
|
|
135
|
+
def initialize(body)
|
|
136
|
+
@groups = []
|
|
137
|
+
@current = nil
|
|
138
|
+
@reading_agents = false
|
|
139
|
+
|
|
140
|
+
each_directive(body) { |key, value| apply(key, value) }
|
|
141
|
+
end
|
|
142
|
+
|
|
143
|
+
private
|
|
144
|
+
|
|
145
|
+
def apply(key, value)
|
|
146
|
+
case key
|
|
147
|
+
when "user-agent" then add_agent(value)
|
|
148
|
+
when "allow", "disallow" then add_rule(value, allow: key == "allow")
|
|
149
|
+
when "crawl-delay" then record_crawl_delay(value)
|
|
150
|
+
end
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
# Consecutive User-agent lines share one group; a rule line ends the
|
|
154
|
+
# list, so the next User-agent starts a new group.
|
|
155
|
+
def add_agent(value)
|
|
156
|
+
unless @reading_agents && @current
|
|
157
|
+
@current = Group.new([], [], nil)
|
|
158
|
+
@groups << @current
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
@current.agents << value.downcase
|
|
162
|
+
@reading_agents = true
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
def add_rule(value, allow:)
|
|
166
|
+
@reading_agents = false
|
|
167
|
+
# "Disallow:" with nothing after it lifts the restriction rather than
|
|
168
|
+
# imposing one, so there is no rule to record.
|
|
169
|
+
return if @current.nil? || value.empty?
|
|
170
|
+
|
|
171
|
+
@current.rules << Rule.new(value, allow: allow)
|
|
172
|
+
end
|
|
173
|
+
|
|
174
|
+
def record_crawl_delay(value)
|
|
175
|
+
@reading_agents = false
|
|
176
|
+
return if @current.nil?
|
|
177
|
+
|
|
178
|
+
@current.crawl_delay = Float(value)
|
|
179
|
+
rescue ArgumentError, TypeError
|
|
180
|
+
nil
|
|
181
|
+
end
|
|
182
|
+
|
|
183
|
+
def each_directive(body)
|
|
184
|
+
body.to_s.each_line do |line|
|
|
185
|
+
line = line.sub(/#.*/, "").strip
|
|
186
|
+
key, separator, value = line.partition(":")
|
|
187
|
+
next if separator.empty?
|
|
188
|
+
|
|
189
|
+
key = key.strip.downcase
|
|
190
|
+
yield key, value.strip if DIRECTIVES.include?(key)
|
|
191
|
+
end
|
|
192
|
+
end
|
|
193
|
+
end
|
|
194
|
+
end
|
|
195
|
+
end
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "uri"
|
|
4
|
+
|
|
5
|
+
module Gento
|
|
6
|
+
# A Fetcher that consults robots.txt before letting a request through.
|
|
7
|
+
#
|
|
8
|
+
# It wraps another fetcher rather than living inside one, so it applies to
|
|
9
|
+
# every request an adapter makes — the deck page, the oEmbed endpoint, the
|
|
10
|
+
# embed view — and works just as well over an injected fetcher as over the
|
|
11
|
+
# net/http one.
|
|
12
|
+
#
|
|
13
|
+
# On by default. `Gento.fetch(url, robots: false)` turns it off; the
|
|
14
|
+
# caller then owns whatever that implies.
|
|
15
|
+
#
|
|
16
|
+
# One gap: the wrapped fetcher follows redirects itself, so a redirect into
|
|
17
|
+
# a disallowed path is not caught. Checking that needs a per-hop callback in
|
|
18
|
+
# the fetcher, which is not worth the interface it would add.
|
|
19
|
+
class RobotsFetcher < Fetcher
|
|
20
|
+
ROBOTS_PATH = "/robots.txt"
|
|
21
|
+
# Speaker Deck serves its robots.txt through the site's HTML layout unless
|
|
22
|
+
# this header asks for plain text, so it is load-bearing rather than
|
|
23
|
+
# politeness. The body is parsed whatever content type comes back: a page
|
|
24
|
+
# that is not a robots.txt yields no directives anyway, and refusing on
|
|
25
|
+
# content type alone would throw away real rules from a host that answers
|
|
26
|
+
# in HTML.
|
|
27
|
+
PLAIN_TEXT = { "accept" => "text/plain" }.freeze
|
|
28
|
+
DEFAULT_PORTS = { "http" => 80, "https" => 443 }.freeze
|
|
29
|
+
|
|
30
|
+
attr_reader :fetcher, :user_agent
|
|
31
|
+
|
|
32
|
+
def initialize(fetcher, user_agent: nil)
|
|
33
|
+
@fetcher = fetcher
|
|
34
|
+
@user_agent = user_agent || inherited_user_agent(fetcher)
|
|
35
|
+
@robots = {}
|
|
36
|
+
super()
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def get(url, headers: {})
|
|
40
|
+
uri = coerce_uri(url)
|
|
41
|
+
guard!(uri) if uri
|
|
42
|
+
|
|
43
|
+
fetcher.get(url, headers: headers)
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
# The Crawl-delay the host asks for, in seconds, or nil.
|
|
47
|
+
#
|
|
48
|
+
# Not enforced here: one fetch is a handful of requests, and sleeping
|
|
49
|
+
# inside a fetcher would surprise a caller who already paces its own work.
|
|
50
|
+
# Exposed so a caller doing more than one deck can honour it.
|
|
51
|
+
def crawl_delay(url)
|
|
52
|
+
uri = coerce_uri(url)
|
|
53
|
+
return nil unless uri
|
|
54
|
+
|
|
55
|
+
robots_for(uri).crawl_delay(agent: user_agent)
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
def robots_for(uri)
|
|
59
|
+
@robots[origin(uri)] ||= load_robots(origin(uri))
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
private
|
|
63
|
+
|
|
64
|
+
def guard!(uri)
|
|
65
|
+
robots = robots_for(uri)
|
|
66
|
+
path = request_path(uri)
|
|
67
|
+
return if robots.allowed?(path, agent: user_agent)
|
|
68
|
+
|
|
69
|
+
raise RobotsDisallowedError.new(refusal(uri, path, robots), url: uri.to_s)
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
def refusal(uri, path, robots)
|
|
73
|
+
cause = robots.reason || "#{origin(uri)}#{ROBOTS_PATH} disallows #{path}"
|
|
74
|
+
|
|
75
|
+
"#{cause} for #{user_agent_label}. " \
|
|
76
|
+
"Pass robots: false to fetch it anyway, and take responsibility for doing so."
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
def user_agent_label
|
|
80
|
+
user_agent.to_s.empty? ? "this client" : user_agent.inspect
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
def load_robots(origin)
|
|
84
|
+
Robots.parse(fetcher.get("#{origin}#{ROBOTS_PATH}", headers: PLAIN_TEXT).body)
|
|
85
|
+
rescue ResponseError => e
|
|
86
|
+
absent?(e.status) ? Robots.allow_all : unreachable(origin, "responded #{e.status}")
|
|
87
|
+
rescue FetchError => e
|
|
88
|
+
unreachable(origin, e.message)
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
# 4xx is the host saying there is no robots.txt, which RFC 9309 reads as
|
|
92
|
+
# "help yourself". 429 is it saying to come back later, which is not the
|
|
93
|
+
# same thing.
|
|
94
|
+
def absent?(status)
|
|
95
|
+
status.to_i.between?(400, 499) && status.to_i != 429
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
def unreachable(origin, detail)
|
|
99
|
+
Robots.disallow_all(reason: "could not read #{origin}#{ROBOTS_PATH} (#{detail}), " \
|
|
100
|
+
"so there is no permission to rely on")
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
def origin(uri)
|
|
104
|
+
port = ":#{uri.port}" unless uri.port == DEFAULT_PORTS[uri.scheme]
|
|
105
|
+
|
|
106
|
+
"#{uri.scheme}://#{uri.host}#{port}"
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
# Rules match against the path and query together.
|
|
110
|
+
def request_path(uri)
|
|
111
|
+
path = uri.path.to_s.empty? ? "/" : uri.path
|
|
112
|
+
uri.query ? "#{path}?#{uri.query}" : path
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
def coerce_uri(url)
|
|
116
|
+
uri = url.is_a?(URI) ? url : URI.parse(url.to_s)
|
|
117
|
+
uri.is_a?(URI::HTTP) ? uri : nil
|
|
118
|
+
rescue URI::InvalidURIError
|
|
119
|
+
nil
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
def inherited_user_agent(fetcher)
|
|
123
|
+
fetcher.user_agent if fetcher.respond_to?(:user_agent)
|
|
124
|
+
end
|
|
125
|
+
end
|
|
126
|
+
end
|
data/lib/gento/slide.rb
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Gento
|
|
4
|
+
# A single page of a deck.
|
|
5
|
+
#
|
|
6
|
+
# `url` always points at the original host. We deliberately never re-host or
|
|
7
|
+
# proxy slide images: this library reports where the images are, and leaves
|
|
8
|
+
# serving them to the site that owns them.
|
|
9
|
+
class Slide
|
|
10
|
+
attr_reader :number, :url, :width, :height, :thumbnail_url
|
|
11
|
+
|
|
12
|
+
def initialize(number:, url:, width: nil, height: nil, thumbnail_url: nil)
|
|
13
|
+
@number = Integer(number)
|
|
14
|
+
@url = url.to_s
|
|
15
|
+
@width = width && Integer(width)
|
|
16
|
+
@height = height && Integer(height)
|
|
17
|
+
@thumbnail_url = thumbnail_url&.to_s
|
|
18
|
+
freeze
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
def to_h
|
|
22
|
+
{
|
|
23
|
+
number: number,
|
|
24
|
+
url: url,
|
|
25
|
+
width: width,
|
|
26
|
+
height: height,
|
|
27
|
+
thumbnail_url: thumbnail_url
|
|
28
|
+
}.compact
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
def ==(other)
|
|
32
|
+
other.is_a?(Slide) && other.to_h == to_h
|
|
33
|
+
end
|
|
34
|
+
alias eql? ==
|
|
35
|
+
|
|
36
|
+
def hash
|
|
37
|
+
to_h.hash
|
|
38
|
+
end
|
|
39
|
+
end
|
|
40
|
+
end
|