gento 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,45 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Gento
4
+ # Turns a raw response body into a UTF-8 string we can safely scan.
5
+ #
6
+ # net/http hands back ASCII-8BIT, and Japanese slide decks are the whole
7
+ # point of this library, so guessing the charset properly is not optional.
8
+ module Charset
9
+ DEFAULT = "UTF-8"
10
+ IN_CONTENT_TYPE = /charset\s*=\s*"?([\w-]+)"?/i
11
+ IN_META = /<meta[^>]+charset\s*=\s*["']?([\w-]+)/i
12
+ # A charset declaration must appear early; do not scan a whole document.
13
+ META_SCAN_BYTES = 4096
14
+
15
+ module_function
16
+
17
+ def normalize(body, content_type: nil)
18
+ body = body.to_s
19
+ name = from_content_type(content_type) || from_meta(body) || DEFAULT
20
+ transcode(body, name)
21
+ end
22
+
23
+ def from_content_type(content_type)
24
+ content_type && content_type[IN_CONTENT_TYPE, 1]
25
+ end
26
+
27
+ def from_meta(body)
28
+ head = body.byteslice(0, META_SCAN_BYTES).to_s.dup.force_encoding(::Encoding::ASCII_8BIT)
29
+ head[IN_META, 1]
30
+ end
31
+
32
+ def transcode(body, name)
33
+ encoding = ::Encoding.find(name)
34
+ decoded = body.dup.force_encoding(encoding)
35
+ decoded = decoded.encode(::Encoding::UTF_8, invalid: :replace, undef: :replace) unless utf8?(encoding)
36
+ decoded.valid_encoding? ? decoded : decoded.scrub
37
+ rescue ArgumentError, ::Encoding::ConverterNotFoundError, ::Encoding::UndefinedConversionError
38
+ body.dup.force_encoding(::Encoding::UTF_8).scrub
39
+ end
40
+
41
+ def utf8?(encoding)
42
+ encoding == ::Encoding::UTF_8
43
+ end
44
+ end
45
+ end
data/lib/gento/cli.rb ADDED
@@ -0,0 +1,121 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "optparse"
4
+ require "json"
5
+
6
+ module Gento
7
+ # `gento <URL>` — prints the deck as JSON, or one image URL per line.
8
+ class CLI
9
+ EXIT_SUCCESS = 0
10
+ EXIT_FAILURE = 1
11
+ EXIT_USAGE = 2
12
+
13
+ def self.start(argv, stdout: $stdout, stderr: $stderr)
14
+ new(stdout: stdout, stderr: stderr).run(argv)
15
+ end
16
+
17
+ def initialize(stdout: $stdout, stderr: $stderr)
18
+ @stdout = stdout
19
+ @stderr = stderr
20
+ @options = { format: :json, pretty: true, robots: true }
21
+ end
22
+
23
+ def run(argv)
24
+ urls = parser.parse(argv.dup)
25
+
26
+ if urls.empty?
27
+ @stderr.puts parser.help
28
+ return EXIT_USAGE
29
+ end
30
+
31
+ fetch_all(urls)
32
+ rescue OptionParser::ParseError => e
33
+ @stderr.puts "gento: #{e.message}"
34
+ EXIT_USAGE
35
+ end
36
+
37
+ private
38
+
39
+ def fetch_all(urls)
40
+ client = Client.new(fetcher: build_fetcher, robots: @options[:robots])
41
+ status = EXIT_SUCCESS
42
+
43
+ urls.each do |url|
44
+ emit(client.fetch(url))
45
+ rescue Error => e
46
+ @stderr.puts "gento: #{url}: #{e.message}"
47
+ status = EXIT_FAILURE
48
+ end
49
+
50
+ status
51
+ end
52
+
53
+ def emit(deck)
54
+ case @options[:format]
55
+ when :urls then deck.slides.each { |slide| @stdout.puts slide.url }
56
+ else @stdout.puts serialize(deck)
57
+ end
58
+ end
59
+
60
+ def serialize(deck)
61
+ @options[:pretty] ? JSON.pretty_generate(deck.to_h) : deck.to_h.to_json
62
+ end
63
+
64
+ def build_fetcher
65
+ NetHttpFetcher.new(**{
66
+ user_agent: @options[:user_agent],
67
+ read_timeout: @options[:timeout]
68
+ }.compact)
69
+ end
70
+
71
+ def parser
72
+ @parser ||= OptionParser.new do |opts|
73
+ opts.banner = "Usage: gento [options] URL [URL...]"
74
+ opts.separator ""
75
+ opts.separator "Supported providers: #{Gento.providers.join(', ')}"
76
+ opts.separator ""
77
+
78
+ define_output_options(opts)
79
+ define_request_options(opts)
80
+ define_meta_options(opts)
81
+ end
82
+ end
83
+
84
+ def define_output_options(opts)
85
+ opts.on("-f", "--format FORMAT", %i[json urls],
86
+ "Output format: json (default) or urls") do |format|
87
+ @options[:format] = format
88
+ end
89
+
90
+ opts.on("--[no-]pretty", "Pretty-print JSON output (default: on)") do |pretty|
91
+ @options[:pretty] = pretty
92
+ end
93
+ end
94
+
95
+ def define_request_options(opts)
96
+ opts.on("-t", "--timeout SECONDS", Integer, "Read timeout per request") do |timeout|
97
+ @options[:timeout] = timeout
98
+ end
99
+
100
+ opts.on("-A", "--user-agent STRING", "User-Agent to send") do |agent|
101
+ @options[:user_agent] = agent
102
+ end
103
+
104
+ opts.on("--[no-]robots", "Obey the host's robots.txt (default: on)") do |robots|
105
+ @options[:robots] = robots
106
+ end
107
+ end
108
+
109
+ def define_meta_options(opts)
110
+ opts.on("-h", "--help", "Show this message") do
111
+ @stdout.puts opts
112
+ exit EXIT_SUCCESS
113
+ end
114
+
115
+ opts.on("-v", "--version", "Show the version") do
116
+ @stdout.puts Gento::VERSION
117
+ exit EXIT_SUCCESS
118
+ end
119
+ end
120
+ end
121
+ end
@@ -0,0 +1,55 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Gento
4
+ # The entry point: give it a slide URL, get a Deck back.
5
+ #
6
+ # client = Gento::Client.new
7
+ # deck = client.fetch("https://speakerdeck.com/user/talk")
8
+ # deck.slides.map(&:url)
9
+ #
10
+ # Both collaborators are injectable: swap the fetcher to run on a host with
11
+ # its own HTTP stack, swap the registry to limit or extend the supported
12
+ # sites.
13
+ #
14
+ # Every request goes through the host's robots.txt unless robots: false is
15
+ # passed. This holds for an injected fetcher too — the point of the check is
16
+ # the request, not who makes it.
17
+ class Client
18
+ attr_reader :fetcher, :registry
19
+
20
+ def initialize(fetcher: nil, registry: nil, robots: true)
21
+ @registry = registry || Registry.default
22
+ @fetcher = wrap(fetcher || NetHttpFetcher.new, robots: robots)
23
+ end
24
+
25
+ def robots?
26
+ fetcher.is_a?(RobotsFetcher)
27
+ end
28
+
29
+ # @raise [UnsupportedURLError] when no adapter claims the URL
30
+ # @raise [RobotsDisallowedError] when robots.txt does not allow the fetch
31
+ # @raise [FetchError] when the remote host cannot be reached
32
+ # @raise [ExtractionError] when the page held no slides
33
+ # @return [Deck]
34
+ def fetch(url)
35
+ adapter = registry.find(url)
36
+ raise UnsupportedURLError, url unless adapter
37
+
38
+ adapter.new(fetcher: fetcher).fetch(url.to_s)
39
+ end
40
+
41
+ def supports?(url)
42
+ registry.supports?(url)
43
+ end
44
+
45
+ private
46
+
47
+ # A fetcher that already checks robots.txt is left alone, so passing one
48
+ # in explicitly does not stack two checks.
49
+ def wrap(fetcher, robots:)
50
+ return fetcher if !robots || fetcher.is_a?(RobotsFetcher)
51
+
52
+ RobotsFetcher.new(fetcher)
53
+ end
54
+ end
55
+ end
data/lib/gento/deck.rb ADDED
@@ -0,0 +1,56 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "json"
4
+
5
+ module Gento
6
+ # The result of fetching one slide URL: metadata plus the ordered pages.
7
+ class Deck
8
+ attr_reader :provider, :source_url, :title, :author, :description,
9
+ :published_at, :slides
10
+
11
+ def initialize(provider:, source_url:, slides:, title: nil, author: nil,
12
+ description: nil, published_at: nil)
13
+ @provider = provider.to_s
14
+ @source_url = source_url.to_s
15
+ @title = normalize(title)
16
+ @author = normalize(author)
17
+ @description = normalize(description)
18
+ @published_at = normalize(published_at)
19
+ @slides = slides.sort_by(&:number).freeze
20
+ freeze
21
+ end
22
+
23
+ def page_count
24
+ slides.size
25
+ end
26
+
27
+ def to_h
28
+ {
29
+ provider: provider,
30
+ source_url: source_url,
31
+ title: title,
32
+ author: author,
33
+ description: description,
34
+ published_at: published_at,
35
+ page_count: page_count,
36
+ slides: slides.map(&:to_h)
37
+ }.compact
38
+ end
39
+
40
+ def to_json(*args)
41
+ to_h.to_json(*args)
42
+ end
43
+
44
+ private
45
+
46
+ # Slide titles are typed into a design tool, so they arrive carrying the
47
+ # author's own line breaks — vertical tabs, in Docswell's case. Collapsing
48
+ # them here means every adapter gets it right without remembering to.
49
+ def normalize(text)
50
+ return nil if text.nil?
51
+
52
+ collapsed = text.to_s.gsub(/[[:space:][:cntrl:]]+/, " ").strip
53
+ collapsed.empty? ? nil : collapsed
54
+ end
55
+ end
56
+ end
@@ -0,0 +1,29 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "cgi"
4
+
5
+ module Gento
6
+ # HTML entity decoding.
7
+ #
8
+ # CGI.unescapeHTML handles the five predefined entities and numeric
9
+ # references, which leaves the typographic ones that real slide titles and
10
+ # descriptions are full of. Rather than carry the whole HTML5 entity table,
11
+ # translate the handful that actually show up.
12
+ module Entities
13
+ NAMED = {
14
+ "hellip" => "\u2026", "mdash" => "\u2014", "ndash" => "\u2013", "nbsp" => "\u00A0",
15
+ "lsquo" => "\u2018", "rsquo" => "\u2019", "ldquo" => "\u201C", "rdquo" => "\u201D",
16
+ "laquo" => "\u00AB", "raquo" => "\u00BB", "bull" => "\u2022", "middot" => "\u00B7",
17
+ "times" => "\u00D7", "deg" => "\u00B0", "copy" => "\u00A9", "reg" => "\u00AE",
18
+ "trade" => "\u2122"
19
+ }.freeze
20
+
21
+ PATTERN = /&(#{Regexp.union(NAMED.keys)});/
22
+
23
+ module_function
24
+
25
+ def decode(text)
26
+ CGI.unescapeHTML(text.to_s).gsub(PATTERN) { NAMED.fetch(Regexp.last_match(1)) }
27
+ end
28
+ end
29
+ end
@@ -0,0 +1,46 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Gento
4
+ # Base class for every error raised by this library. Callers that just want
5
+ # "did it work?" can rescue this one class.
6
+ class Error < StandardError; end
7
+
8
+ # The given URL is not a URL we know how to handle.
9
+ class UnsupportedURLError < Error
10
+ attr_reader :url
11
+
12
+ def initialize(url)
13
+ @url = url
14
+ super("no adapter registered for #{url.inspect}")
15
+ end
16
+ end
17
+
18
+ # The URL looked supported, but the remote document did not contain the
19
+ # slide data we expected. Usually means the site changed its markup.
20
+ class ExtractionError < Error; end
21
+
22
+ # Something went wrong while talking to the remote host.
23
+ class FetchError < Error
24
+ attr_reader :url, :status
25
+
26
+ def initialize(message, url: nil, status: nil)
27
+ @url = url
28
+ @status = status
29
+ super(message)
30
+ end
31
+ end
32
+
33
+ # The remote host answered, but with a non-success status.
34
+ class ResponseError < FetchError; end
35
+
36
+ # We followed more redirects than the fetcher allows.
37
+ class TooManyRedirectsError < FetchError; end
38
+
39
+ # The host's robots.txt does not allow this client to fetch this path — or
40
+ # could not be read, which comes to the same thing.
41
+ #
42
+ # A FetchError because it is a request that did not happen, which also means
43
+ # an optional extra like an oEmbed lookup degrades instead of failing the
44
+ # whole fetch.
45
+ class RobotsDisallowedError < FetchError; end
46
+ end
@@ -0,0 +1,188 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "net/http"
4
+ require "uri"
5
+ require "resolv"
6
+ require "ipaddr"
7
+
8
+ module Gento
9
+ # The HTTP seam.
10
+ #
11
+ # Everything the library does over the network goes through one of these, so
12
+ # a host that has its own HTTP stack (a Worker's `fetch`, a Faraday
13
+ # connection, a test double) can supply an adapter instead of dragging
14
+ # net/http along.
15
+ #
16
+ # Implementations must return a Gento::Response and raise
17
+ # Gento::FetchError on transport failure.
18
+ class Fetcher
19
+ def get(url, headers: {})
20
+ raise NotImplementedError, "#{self.class} must implement #get"
21
+ end
22
+ end
23
+
24
+ # Default fetcher, built on net/http from the standard library.
25
+ class NetHttpFetcher < Fetcher
26
+ DEFAULT_USER_AGENT =
27
+ "gento/#{VERSION} (+https://github.com/ngram/gento)".freeze
28
+
29
+ # Ranges that a public slide host has no business resolving to. Checked
30
+ # before connecting so a redirect cannot walk us into the private network.
31
+ BLOCKED_RANGES = [
32
+ IPAddr.new("0.0.0.0/8"), IPAddr.new("10.0.0.0/8"),
33
+ IPAddr.new("127.0.0.0/8"), IPAddr.new("169.254.0.0/16"),
34
+ IPAddr.new("172.16.0.0/12"), IPAddr.new("192.168.0.0/16"),
35
+ IPAddr.new("100.64.0.0/10"), IPAddr.new("::1/128"),
36
+ IPAddr.new("fc00::/7"), IPAddr.new("fe80::/10")
37
+ ].freeze
38
+
39
+ attr_reader :user_agent, :open_timeout, :read_timeout, :max_redirects, :proxy
40
+
41
+ def initialize(user_agent: DEFAULT_USER_AGENT, open_timeout: 5, read_timeout: 10,
42
+ max_redirects: 5, block_private_addresses: true, proxy: :env)
43
+ @user_agent = user_agent
44
+ @open_timeout = open_timeout
45
+ @read_timeout = read_timeout
46
+ @max_redirects = max_redirects
47
+ @block_private_addresses = block_private_addresses
48
+ @proxy = proxy == :env ? proxy_from_env : normalize_proxy(proxy)
49
+ super()
50
+ end
51
+
52
+ def get(url, headers: {})
53
+ uri = normalize_uri(url)
54
+ redirects = 0
55
+
56
+ loop do
57
+ response = request(uri, headers)
58
+ location = response["location"]
59
+
60
+ return build_response(response, uri) unless redirect?(response) && location
61
+
62
+ redirects += 1
63
+ if redirects > max_redirects
64
+ raise TooManyRedirectsError.new("more than #{max_redirects} redirects", url: url)
65
+ end
66
+
67
+ uri = normalize_uri(URI.join(uri, location))
68
+ end
69
+ end
70
+
71
+ private
72
+
73
+ def request(uri, headers)
74
+ addresses = resolve(uri.hostname)
75
+ guard_address!(uri, addresses)
76
+
77
+ connection(uri, addresses).start do |http|
78
+ http.request(Net::HTTP::Get.new(uri, request_headers(headers)))
79
+ end
80
+ rescue Timeout::Error, SystemCallError, IOError, OpenSSL::SSL::SSLError,
81
+ Net::HTTPBadResponse, Net::HTTPHeaderSyntaxError => e
82
+ raise FetchError.new("could not fetch #{uri}: #{e.class}: #{e.message}", url: uri.to_s)
83
+ end
84
+
85
+ # Built explicitly rather than through Net::HTTP.start's keyword form:
86
+ # there the proxy arguments are positional, so passing them as keywords
87
+ # puts them in the options hash where they are silently ignored, leaving
88
+ # every request unproxied. Passing nil for the address disables proxying.
89
+ def connection(uri, addresses)
90
+ http = Net::HTTP.new(uri.hostname, uri.port, *proxy_parts(addresses))
91
+ http.use_ssl = uri.scheme == "https"
92
+ http.open_timeout = open_timeout
93
+ http.read_timeout = read_timeout
94
+ http
95
+ end
96
+
97
+ # net/http only consults the lowercase `http_proxy`, and never for HTTPS,
98
+ # so environments that set HTTPS_PROXY — most corporate networks and every
99
+ # sandboxed CI runner — go unproxied and fail. Read both spellings.
100
+ def proxy_from_env
101
+ value = ENV["HTTPS_PROXY"] || ENV["https_proxy"] || ENV["HTTP_PROXY"] || ENV.fetch("http_proxy", nil)
102
+ normalize_proxy(value)
103
+ end
104
+
105
+ def normalize_proxy(value)
106
+ return nil if value.nil? || value.to_s.empty?
107
+
108
+ uri = value.is_a?(URI) ? value : URI.parse(value.to_s)
109
+ uri.host ? uri : nil
110
+ rescue URI::InvalidURIError
111
+ nil
112
+ end
113
+
114
+ # Identity encoding: we decode charsets ourselves and have no reason to
115
+ # also own gzip handling.
116
+ def request_headers(headers)
117
+ { "user-agent" => user_agent, "accept-encoding" => "identity" }
118
+ .merge(headers.transform_keys { |key| key.to_s.downcase })
119
+ end
120
+
121
+ def build_response(response, uri)
122
+ status = response.code.to_i
123
+ unless status.between?(200, 299)
124
+ raise ResponseError.new("#{uri} responded #{status}", url: uri.to_s, status: status)
125
+ end
126
+
127
+ headers = response.each_header.to_h
128
+ Response.new(status: status, body: response.body.to_s, url: uri.to_s, headers: headers)
129
+ end
130
+
131
+ def redirect?(response)
132
+ response.is_a?(Net::HTTPRedirection)
133
+ end
134
+
135
+ def normalize_uri(url)
136
+ uri = url.is_a?(URI) ? url : URI.parse(url.to_s)
137
+ unless uri.is_a?(URI::HTTP)
138
+ raise FetchError.new("only http(s) URLs are supported, got #{url.inspect}", url: url.to_s)
139
+ end
140
+
141
+ uri
142
+ end
143
+
144
+ # Best-effort SSRF guard. There is an unavoidable gap between resolving a
145
+ # name here and net/http resolving it again when it connects; closing it
146
+ # properly needs socket-level pinning. This still stops the ordinary case
147
+ # of a redirect pointed at an internal address.
148
+ def guard_address!(uri, addresses)
149
+ return unless @block_private_addresses
150
+
151
+ blocked = blocked_address(addresses)
152
+ return unless blocked
153
+
154
+ raise FetchError.new("refusing to connect to private address #{blocked} (#{uri.hostname})",
155
+ url: uri.to_s)
156
+ end
157
+
158
+ # A proxy exists to reach the outside world, and loopback or private
159
+ # addresses are not out there. Routing them through one is how a request
160
+ # to a service on this machine ends up answered by the proxy instead.
161
+ def proxy_parts(addresses)
162
+ return [nil, nil, nil, nil] if proxy.nil? || blocked_address(addresses)
163
+
164
+ [proxy.hostname, proxy.port, proxy.user, proxy.password]
165
+ end
166
+
167
+ def blocked_address(addresses)
168
+ addresses.find { |address| BLOCKED_RANGES.any? { |range| range.include?(address) } }
169
+ end
170
+
171
+ def resolve(host)
172
+ return [IPAddr.new(host)] if ip_literal?(host)
173
+
174
+ Resolv.getaddresses(host).filter_map do |address|
175
+ IPAddr.new(address)
176
+ rescue IPAddr::Error
177
+ nil
178
+ end
179
+ end
180
+
181
+ def ip_literal?(host)
182
+ IPAddr.new(host)
183
+ true
184
+ rescue IPAddr::Error
185
+ false
186
+ end
187
+ end
188
+ end