gento 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of gento might be problematic. Click here for more details.
- checksums.yaml +7 -0
- data/CHANGELOG.md +53 -0
- data/LICENSE.txt +21 -0
- data/README.md +143 -0
- data/exe/gento +7 -0
- data/lib/gento/adapters/base.rb +121 -0
- data/lib/gento/adapters/docswell.rb +58 -0
- data/lib/gento/adapters/google_slides.rb +90 -0
- data/lib/gento/adapters/slide_share.rb +82 -0
- data/lib/gento/adapters/speaker_deck.rb +79 -0
- data/lib/gento/charset.rb +45 -0
- data/lib/gento/cli.rb +121 -0
- data/lib/gento/client.rb +55 -0
- data/lib/gento/deck.rb +56 -0
- data/lib/gento/entities.rb +29 -0
- data/lib/gento/errors.rb +46 -0
- data/lib/gento/fetcher.rb +188 -0
- data/lib/gento/html.rb +225 -0
- data/lib/gento/registry.rb +49 -0
- data/lib/gento/response.rb +34 -0
- data/lib/gento/robots.rb +195 -0
- data/lib/gento/robots_fetcher.rb +126 -0
- data/lib/gento/slide.rb +40 -0
- data/lib/gento/version.rb +5 -0
- data/lib/gento.rb +44 -0
- metadata +76 -0
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Gento
|
|
4
|
+
# Turns a raw response body into a UTF-8 string we can safely scan.
|
|
5
|
+
#
|
|
6
|
+
# net/http hands back ASCII-8BIT, and Japanese slide decks are the whole
|
|
7
|
+
# point of this library, so guessing the charset properly is not optional.
|
|
8
|
+
module Charset
|
|
9
|
+
DEFAULT = "UTF-8"
|
|
10
|
+
IN_CONTENT_TYPE = /charset\s*=\s*"?([\w-]+)"?/i
|
|
11
|
+
IN_META = /<meta[^>]+charset\s*=\s*["']?([\w-]+)/i
|
|
12
|
+
# A charset declaration must appear early; do not scan a whole document.
|
|
13
|
+
META_SCAN_BYTES = 4096
|
|
14
|
+
|
|
15
|
+
module_function
|
|
16
|
+
|
|
17
|
+
def normalize(body, content_type: nil)
|
|
18
|
+
body = body.to_s
|
|
19
|
+
name = from_content_type(content_type) || from_meta(body) || DEFAULT
|
|
20
|
+
transcode(body, name)
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
def from_content_type(content_type)
|
|
24
|
+
content_type && content_type[IN_CONTENT_TYPE, 1]
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def from_meta(body)
|
|
28
|
+
head = body.byteslice(0, META_SCAN_BYTES).to_s.dup.force_encoding(::Encoding::ASCII_8BIT)
|
|
29
|
+
head[IN_META, 1]
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
def transcode(body, name)
|
|
33
|
+
encoding = ::Encoding.find(name)
|
|
34
|
+
decoded = body.dup.force_encoding(encoding)
|
|
35
|
+
decoded = decoded.encode(::Encoding::UTF_8, invalid: :replace, undef: :replace) unless utf8?(encoding)
|
|
36
|
+
decoded.valid_encoding? ? decoded : decoded.scrub
|
|
37
|
+
rescue ArgumentError, ::Encoding::ConverterNotFoundError, ::Encoding::UndefinedConversionError
|
|
38
|
+
body.dup.force_encoding(::Encoding::UTF_8).scrub
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
def utf8?(encoding)
|
|
42
|
+
encoding == ::Encoding::UTF_8
|
|
43
|
+
end
|
|
44
|
+
end
|
|
45
|
+
end
|
data/lib/gento/cli.rb
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "optparse"
|
|
4
|
+
require "json"
|
|
5
|
+
|
|
6
|
+
module Gento
|
|
7
|
+
# `gento <URL>` — prints the deck as JSON, or one image URL per line.
|
|
8
|
+
class CLI
|
|
9
|
+
EXIT_SUCCESS = 0
|
|
10
|
+
EXIT_FAILURE = 1
|
|
11
|
+
EXIT_USAGE = 2
|
|
12
|
+
|
|
13
|
+
def self.start(argv, stdout: $stdout, stderr: $stderr)
|
|
14
|
+
new(stdout: stdout, stderr: stderr).run(argv)
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
def initialize(stdout: $stdout, stderr: $stderr)
|
|
18
|
+
@stdout = stdout
|
|
19
|
+
@stderr = stderr
|
|
20
|
+
@options = { format: :json, pretty: true, robots: true }
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
def run(argv)
|
|
24
|
+
urls = parser.parse(argv.dup)
|
|
25
|
+
|
|
26
|
+
if urls.empty?
|
|
27
|
+
@stderr.puts parser.help
|
|
28
|
+
return EXIT_USAGE
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
fetch_all(urls)
|
|
32
|
+
rescue OptionParser::ParseError => e
|
|
33
|
+
@stderr.puts "gento: #{e.message}"
|
|
34
|
+
EXIT_USAGE
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
private
|
|
38
|
+
|
|
39
|
+
def fetch_all(urls)
|
|
40
|
+
client = Client.new(fetcher: build_fetcher, robots: @options[:robots])
|
|
41
|
+
status = EXIT_SUCCESS
|
|
42
|
+
|
|
43
|
+
urls.each do |url|
|
|
44
|
+
emit(client.fetch(url))
|
|
45
|
+
rescue Error => e
|
|
46
|
+
@stderr.puts "gento: #{url}: #{e.message}"
|
|
47
|
+
status = EXIT_FAILURE
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
status
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
def emit(deck)
|
|
54
|
+
case @options[:format]
|
|
55
|
+
when :urls then deck.slides.each { |slide| @stdout.puts slide.url }
|
|
56
|
+
else @stdout.puts serialize(deck)
|
|
57
|
+
end
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
def serialize(deck)
|
|
61
|
+
@options[:pretty] ? JSON.pretty_generate(deck.to_h) : deck.to_h.to_json
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
def build_fetcher
|
|
65
|
+
NetHttpFetcher.new(**{
|
|
66
|
+
user_agent: @options[:user_agent],
|
|
67
|
+
read_timeout: @options[:timeout]
|
|
68
|
+
}.compact)
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
def parser
|
|
72
|
+
@parser ||= OptionParser.new do |opts|
|
|
73
|
+
opts.banner = "Usage: gento [options] URL [URL...]"
|
|
74
|
+
opts.separator ""
|
|
75
|
+
opts.separator "Supported providers: #{Gento.providers.join(', ')}"
|
|
76
|
+
opts.separator ""
|
|
77
|
+
|
|
78
|
+
define_output_options(opts)
|
|
79
|
+
define_request_options(opts)
|
|
80
|
+
define_meta_options(opts)
|
|
81
|
+
end
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
def define_output_options(opts)
|
|
85
|
+
opts.on("-f", "--format FORMAT", %i[json urls],
|
|
86
|
+
"Output format: json (default) or urls") do |format|
|
|
87
|
+
@options[:format] = format
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
opts.on("--[no-]pretty", "Pretty-print JSON output (default: on)") do |pretty|
|
|
91
|
+
@options[:pretty] = pretty
|
|
92
|
+
end
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
def define_request_options(opts)
|
|
96
|
+
opts.on("-t", "--timeout SECONDS", Integer, "Read timeout per request") do |timeout|
|
|
97
|
+
@options[:timeout] = timeout
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
opts.on("-A", "--user-agent STRING", "User-Agent to send") do |agent|
|
|
101
|
+
@options[:user_agent] = agent
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
opts.on("--[no-]robots", "Obey the host's robots.txt (default: on)") do |robots|
|
|
105
|
+
@options[:robots] = robots
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
def define_meta_options(opts)
|
|
110
|
+
opts.on("-h", "--help", "Show this message") do
|
|
111
|
+
@stdout.puts opts
|
|
112
|
+
exit EXIT_SUCCESS
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
opts.on("-v", "--version", "Show the version") do
|
|
116
|
+
@stdout.puts Gento::VERSION
|
|
117
|
+
exit EXIT_SUCCESS
|
|
118
|
+
end
|
|
119
|
+
end
|
|
120
|
+
end
|
|
121
|
+
end
|
data/lib/gento/client.rb
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Gento
|
|
4
|
+
# The entry point: give it a slide URL, get a Deck back.
|
|
5
|
+
#
|
|
6
|
+
# client = Gento::Client.new
|
|
7
|
+
# deck = client.fetch("https://speakerdeck.com/user/talk")
|
|
8
|
+
# deck.slides.map(&:url)
|
|
9
|
+
#
|
|
10
|
+
# Both collaborators are injectable: swap the fetcher to run on a host with
|
|
11
|
+
# its own HTTP stack, swap the registry to limit or extend the supported
|
|
12
|
+
# sites.
|
|
13
|
+
#
|
|
14
|
+
# Every request goes through the host's robots.txt unless robots: false is
|
|
15
|
+
# passed. This holds for an injected fetcher too — the point of the check is
|
|
16
|
+
# the request, not who makes it.
|
|
17
|
+
class Client
|
|
18
|
+
attr_reader :fetcher, :registry
|
|
19
|
+
|
|
20
|
+
def initialize(fetcher: nil, registry: nil, robots: true)
|
|
21
|
+
@registry = registry || Registry.default
|
|
22
|
+
@fetcher = wrap(fetcher || NetHttpFetcher.new, robots: robots)
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def robots?
|
|
26
|
+
fetcher.is_a?(RobotsFetcher)
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
# @raise [UnsupportedURLError] when no adapter claims the URL
|
|
30
|
+
# @raise [RobotsDisallowedError] when robots.txt does not allow the fetch
|
|
31
|
+
# @raise [FetchError] when the remote host cannot be reached
|
|
32
|
+
# @raise [ExtractionError] when the page held no slides
|
|
33
|
+
# @return [Deck]
|
|
34
|
+
def fetch(url)
|
|
35
|
+
adapter = registry.find(url)
|
|
36
|
+
raise UnsupportedURLError, url unless adapter
|
|
37
|
+
|
|
38
|
+
adapter.new(fetcher: fetcher).fetch(url.to_s)
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
def supports?(url)
|
|
42
|
+
registry.supports?(url)
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
private
|
|
46
|
+
|
|
47
|
+
# A fetcher that already checks robots.txt is left alone, so passing one
|
|
48
|
+
# in explicitly does not stack two checks.
|
|
49
|
+
def wrap(fetcher, robots:)
|
|
50
|
+
return fetcher if !robots || fetcher.is_a?(RobotsFetcher)
|
|
51
|
+
|
|
52
|
+
RobotsFetcher.new(fetcher)
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
end
|
data/lib/gento/deck.rb
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
|
|
5
|
+
module Gento
|
|
6
|
+
# The result of fetching one slide URL: metadata plus the ordered pages.
|
|
7
|
+
class Deck
|
|
8
|
+
attr_reader :provider, :source_url, :title, :author, :description,
|
|
9
|
+
:published_at, :slides
|
|
10
|
+
|
|
11
|
+
def initialize(provider:, source_url:, slides:, title: nil, author: nil,
|
|
12
|
+
description: nil, published_at: nil)
|
|
13
|
+
@provider = provider.to_s
|
|
14
|
+
@source_url = source_url.to_s
|
|
15
|
+
@title = normalize(title)
|
|
16
|
+
@author = normalize(author)
|
|
17
|
+
@description = normalize(description)
|
|
18
|
+
@published_at = normalize(published_at)
|
|
19
|
+
@slides = slides.sort_by(&:number).freeze
|
|
20
|
+
freeze
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
def page_count
|
|
24
|
+
slides.size
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def to_h
|
|
28
|
+
{
|
|
29
|
+
provider: provider,
|
|
30
|
+
source_url: source_url,
|
|
31
|
+
title: title,
|
|
32
|
+
author: author,
|
|
33
|
+
description: description,
|
|
34
|
+
published_at: published_at,
|
|
35
|
+
page_count: page_count,
|
|
36
|
+
slides: slides.map(&:to_h)
|
|
37
|
+
}.compact
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
def to_json(*args)
|
|
41
|
+
to_h.to_json(*args)
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
private
|
|
45
|
+
|
|
46
|
+
# Slide titles are typed into a design tool, so they arrive carrying the
|
|
47
|
+
# author's own line breaks — vertical tabs, in Docswell's case. Collapsing
|
|
48
|
+
# them here means every adapter gets it right without remembering to.
|
|
49
|
+
def normalize(text)
|
|
50
|
+
return nil if text.nil?
|
|
51
|
+
|
|
52
|
+
collapsed = text.to_s.gsub(/[[:space:][:cntrl:]]+/, " ").strip
|
|
53
|
+
collapsed.empty? ? nil : collapsed
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
end
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "cgi"
|
|
4
|
+
|
|
5
|
+
module Gento
|
|
6
|
+
# HTML entity decoding.
|
|
7
|
+
#
|
|
8
|
+
# CGI.unescapeHTML handles the five predefined entities and numeric
|
|
9
|
+
# references, which leaves the typographic ones that real slide titles and
|
|
10
|
+
# descriptions are full of. Rather than carry the whole HTML5 entity table,
|
|
11
|
+
# translate the handful that actually show up.
|
|
12
|
+
module Entities
|
|
13
|
+
NAMED = {
|
|
14
|
+
"hellip" => "\u2026", "mdash" => "\u2014", "ndash" => "\u2013", "nbsp" => "\u00A0",
|
|
15
|
+
"lsquo" => "\u2018", "rsquo" => "\u2019", "ldquo" => "\u201C", "rdquo" => "\u201D",
|
|
16
|
+
"laquo" => "\u00AB", "raquo" => "\u00BB", "bull" => "\u2022", "middot" => "\u00B7",
|
|
17
|
+
"times" => "\u00D7", "deg" => "\u00B0", "copy" => "\u00A9", "reg" => "\u00AE",
|
|
18
|
+
"trade" => "\u2122"
|
|
19
|
+
}.freeze
|
|
20
|
+
|
|
21
|
+
PATTERN = /&(#{Regexp.union(NAMED.keys)});/
|
|
22
|
+
|
|
23
|
+
module_function
|
|
24
|
+
|
|
25
|
+
def decode(text)
|
|
26
|
+
CGI.unescapeHTML(text.to_s).gsub(PATTERN) { NAMED.fetch(Regexp.last_match(1)) }
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
end
|
data/lib/gento/errors.rb
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Gento
|
|
4
|
+
# Base class for every error raised by this library. Callers that just want
|
|
5
|
+
# "did it work?" can rescue this one class.
|
|
6
|
+
class Error < StandardError; end
|
|
7
|
+
|
|
8
|
+
# The given URL is not a URL we know how to handle.
|
|
9
|
+
class UnsupportedURLError < Error
|
|
10
|
+
attr_reader :url
|
|
11
|
+
|
|
12
|
+
def initialize(url)
|
|
13
|
+
@url = url
|
|
14
|
+
super("no adapter registered for #{url.inspect}")
|
|
15
|
+
end
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
# The URL looked supported, but the remote document did not contain the
|
|
19
|
+
# slide data we expected. Usually means the site changed its markup.
|
|
20
|
+
class ExtractionError < Error; end
|
|
21
|
+
|
|
22
|
+
# Something went wrong while talking to the remote host.
|
|
23
|
+
class FetchError < Error
|
|
24
|
+
attr_reader :url, :status
|
|
25
|
+
|
|
26
|
+
def initialize(message, url: nil, status: nil)
|
|
27
|
+
@url = url
|
|
28
|
+
@status = status
|
|
29
|
+
super(message)
|
|
30
|
+
end
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
# The remote host answered, but with a non-success status.
|
|
34
|
+
class ResponseError < FetchError; end
|
|
35
|
+
|
|
36
|
+
# We followed more redirects than the fetcher allows.
|
|
37
|
+
class TooManyRedirectsError < FetchError; end
|
|
38
|
+
|
|
39
|
+
# The host's robots.txt does not allow this client to fetch this path — or
|
|
40
|
+
# could not be read, which comes to the same thing.
|
|
41
|
+
#
|
|
42
|
+
# A FetchError because it is a request that did not happen, which also means
|
|
43
|
+
# an optional extra like an oEmbed lookup degrades instead of failing the
|
|
44
|
+
# whole fetch.
|
|
45
|
+
class RobotsDisallowedError < FetchError; end
|
|
46
|
+
end
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "net/http"
|
|
4
|
+
require "uri"
|
|
5
|
+
require "resolv"
|
|
6
|
+
require "ipaddr"
|
|
7
|
+
|
|
8
|
+
module Gento
|
|
9
|
+
# The HTTP seam.
|
|
10
|
+
#
|
|
11
|
+
# Everything the library does over the network goes through one of these, so
|
|
12
|
+
# a host that has its own HTTP stack (a Worker's `fetch`, a Faraday
|
|
13
|
+
# connection, a test double) can supply an adapter instead of dragging
|
|
14
|
+
# net/http along.
|
|
15
|
+
#
|
|
16
|
+
# Implementations must return a Gento::Response and raise
|
|
17
|
+
# Gento::FetchError on transport failure.
|
|
18
|
+
class Fetcher
|
|
19
|
+
def get(url, headers: {})
|
|
20
|
+
raise NotImplementedError, "#{self.class} must implement #get"
|
|
21
|
+
end
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
# Default fetcher, built on net/http from the standard library.
|
|
25
|
+
class NetHttpFetcher < Fetcher
|
|
26
|
+
DEFAULT_USER_AGENT =
|
|
27
|
+
"gento/#{VERSION} (+https://github.com/ngram/gento)".freeze
|
|
28
|
+
|
|
29
|
+
# Ranges that a public slide host has no business resolving to. Checked
|
|
30
|
+
# before connecting so a redirect cannot walk us into the private network.
|
|
31
|
+
BLOCKED_RANGES = [
|
|
32
|
+
IPAddr.new("0.0.0.0/8"), IPAddr.new("10.0.0.0/8"),
|
|
33
|
+
IPAddr.new("127.0.0.0/8"), IPAddr.new("169.254.0.0/16"),
|
|
34
|
+
IPAddr.new("172.16.0.0/12"), IPAddr.new("192.168.0.0/16"),
|
|
35
|
+
IPAddr.new("100.64.0.0/10"), IPAddr.new("::1/128"),
|
|
36
|
+
IPAddr.new("fc00::/7"), IPAddr.new("fe80::/10")
|
|
37
|
+
].freeze
|
|
38
|
+
|
|
39
|
+
attr_reader :user_agent, :open_timeout, :read_timeout, :max_redirects, :proxy
|
|
40
|
+
|
|
41
|
+
def initialize(user_agent: DEFAULT_USER_AGENT, open_timeout: 5, read_timeout: 10,
|
|
42
|
+
max_redirects: 5, block_private_addresses: true, proxy: :env)
|
|
43
|
+
@user_agent = user_agent
|
|
44
|
+
@open_timeout = open_timeout
|
|
45
|
+
@read_timeout = read_timeout
|
|
46
|
+
@max_redirects = max_redirects
|
|
47
|
+
@block_private_addresses = block_private_addresses
|
|
48
|
+
@proxy = proxy == :env ? proxy_from_env : normalize_proxy(proxy)
|
|
49
|
+
super()
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
def get(url, headers: {})
|
|
53
|
+
uri = normalize_uri(url)
|
|
54
|
+
redirects = 0
|
|
55
|
+
|
|
56
|
+
loop do
|
|
57
|
+
response = request(uri, headers)
|
|
58
|
+
location = response["location"]
|
|
59
|
+
|
|
60
|
+
return build_response(response, uri) unless redirect?(response) && location
|
|
61
|
+
|
|
62
|
+
redirects += 1
|
|
63
|
+
if redirects > max_redirects
|
|
64
|
+
raise TooManyRedirectsError.new("more than #{max_redirects} redirects", url: url)
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
uri = normalize_uri(URI.join(uri, location))
|
|
68
|
+
end
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
private
|
|
72
|
+
|
|
73
|
+
def request(uri, headers)
|
|
74
|
+
addresses = resolve(uri.hostname)
|
|
75
|
+
guard_address!(uri, addresses)
|
|
76
|
+
|
|
77
|
+
connection(uri, addresses).start do |http|
|
|
78
|
+
http.request(Net::HTTP::Get.new(uri, request_headers(headers)))
|
|
79
|
+
end
|
|
80
|
+
rescue Timeout::Error, SystemCallError, IOError, OpenSSL::SSL::SSLError,
|
|
81
|
+
Net::HTTPBadResponse, Net::HTTPHeaderSyntaxError => e
|
|
82
|
+
raise FetchError.new("could not fetch #{uri}: #{e.class}: #{e.message}", url: uri.to_s)
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
# Built explicitly rather than through Net::HTTP.start's keyword form:
|
|
86
|
+
# there the proxy arguments are positional, so passing them as keywords
|
|
87
|
+
# puts them in the options hash where they are silently ignored, leaving
|
|
88
|
+
# every request unproxied. Passing nil for the address disables proxying.
|
|
89
|
+
def connection(uri, addresses)
|
|
90
|
+
http = Net::HTTP.new(uri.hostname, uri.port, *proxy_parts(addresses))
|
|
91
|
+
http.use_ssl = uri.scheme == "https"
|
|
92
|
+
http.open_timeout = open_timeout
|
|
93
|
+
http.read_timeout = read_timeout
|
|
94
|
+
http
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
# net/http only consults the lowercase `http_proxy`, and never for HTTPS,
|
|
98
|
+
# so environments that set HTTPS_PROXY — most corporate networks and every
|
|
99
|
+
# sandboxed CI runner — go unproxied and fail. Read both spellings.
|
|
100
|
+
def proxy_from_env
|
|
101
|
+
value = ENV["HTTPS_PROXY"] || ENV["https_proxy"] || ENV["HTTP_PROXY"] || ENV.fetch("http_proxy", nil)
|
|
102
|
+
normalize_proxy(value)
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
def normalize_proxy(value)
|
|
106
|
+
return nil if value.nil? || value.to_s.empty?
|
|
107
|
+
|
|
108
|
+
uri = value.is_a?(URI) ? value : URI.parse(value.to_s)
|
|
109
|
+
uri.host ? uri : nil
|
|
110
|
+
rescue URI::InvalidURIError
|
|
111
|
+
nil
|
|
112
|
+
end
|
|
113
|
+
|
|
114
|
+
# Identity encoding: we decode charsets ourselves and have no reason to
|
|
115
|
+
# also own gzip handling.
|
|
116
|
+
def request_headers(headers)
|
|
117
|
+
{ "user-agent" => user_agent, "accept-encoding" => "identity" }
|
|
118
|
+
.merge(headers.transform_keys { |key| key.to_s.downcase })
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
def build_response(response, uri)
|
|
122
|
+
status = response.code.to_i
|
|
123
|
+
unless status.between?(200, 299)
|
|
124
|
+
raise ResponseError.new("#{uri} responded #{status}", url: uri.to_s, status: status)
|
|
125
|
+
end
|
|
126
|
+
|
|
127
|
+
headers = response.each_header.to_h
|
|
128
|
+
Response.new(status: status, body: response.body.to_s, url: uri.to_s, headers: headers)
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
def redirect?(response)
|
|
132
|
+
response.is_a?(Net::HTTPRedirection)
|
|
133
|
+
end
|
|
134
|
+
|
|
135
|
+
def normalize_uri(url)
|
|
136
|
+
uri = url.is_a?(URI) ? url : URI.parse(url.to_s)
|
|
137
|
+
unless uri.is_a?(URI::HTTP)
|
|
138
|
+
raise FetchError.new("only http(s) URLs are supported, got #{url.inspect}", url: url.to_s)
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
uri
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
# Best-effort SSRF guard. There is an unavoidable gap between resolving a
|
|
145
|
+
# name here and net/http resolving it again when it connects; closing it
|
|
146
|
+
# properly needs socket-level pinning. This still stops the ordinary case
|
|
147
|
+
# of a redirect pointed at an internal address.
|
|
148
|
+
def guard_address!(uri, addresses)
|
|
149
|
+
return unless @block_private_addresses
|
|
150
|
+
|
|
151
|
+
blocked = blocked_address(addresses)
|
|
152
|
+
return unless blocked
|
|
153
|
+
|
|
154
|
+
raise FetchError.new("refusing to connect to private address #{blocked} (#{uri.hostname})",
|
|
155
|
+
url: uri.to_s)
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
# A proxy exists to reach the outside world, and loopback or private
|
|
159
|
+
# addresses are not out there. Routing them through one is how a request
|
|
160
|
+
# to a service on this machine ends up answered by the proxy instead.
|
|
161
|
+
def proxy_parts(addresses)
|
|
162
|
+
return [nil, nil, nil, nil] if proxy.nil? || blocked_address(addresses)
|
|
163
|
+
|
|
164
|
+
[proxy.hostname, proxy.port, proxy.user, proxy.password]
|
|
165
|
+
end
|
|
166
|
+
|
|
167
|
+
def blocked_address(addresses)
|
|
168
|
+
addresses.find { |address| BLOCKED_RANGES.any? { |range| range.include?(address) } }
|
|
169
|
+
end
|
|
170
|
+
|
|
171
|
+
def resolve(host)
|
|
172
|
+
return [IPAddr.new(host)] if ip_literal?(host)
|
|
173
|
+
|
|
174
|
+
Resolv.getaddresses(host).filter_map do |address|
|
|
175
|
+
IPAddr.new(address)
|
|
176
|
+
rescue IPAddr::Error
|
|
177
|
+
nil
|
|
178
|
+
end
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
def ip_literal?(host)
|
|
182
|
+
IPAddr.new(host)
|
|
183
|
+
true
|
|
184
|
+
rescue IPAddr::Error
|
|
185
|
+
false
|
|
186
|
+
end
|
|
187
|
+
end
|
|
188
|
+
end
|