site-inspector 3.2.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. checksums.yaml +4 -4
  2. data/.github/workflows/ci.yml +30 -0
  3. data/.rubocop.yml +1 -5
  4. data/.rubocop_todo.yml +185 -79
  5. data/.ruby-version +1 -1
  6. data/CHANGELOG.md +71 -0
  7. data/Gemfile +0 -2
  8. data/README.md +8 -14
  9. data/bin/site-inspector +13 -13
  10. data/lib/cliver/dependency_ext.rb +2 -2
  11. data/lib/data/well-known.yml +51 -0
  12. data/lib/site-inspector/domain.rb +27 -15
  13. data/lib/site-inspector/domain_parser.rb +11 -0
  14. data/lib/site-inspector/{checks → endpoint}/accessibility.rb +6 -4
  15. data/lib/site-inspector/{checks → endpoint}/check.rb +5 -1
  16. data/lib/site-inspector/endpoint/content.rb +151 -0
  17. data/lib/site-inspector/{checks → endpoint}/cookies.rb +13 -5
  18. data/lib/site-inspector/{checks → endpoint}/dns.rb +42 -18
  19. data/lib/site-inspector/endpoint/headers.rb +55 -0
  20. data/lib/site-inspector/{checks → endpoint}/hsts.rb +36 -5
  21. data/lib/site-inspector/endpoint/wappalyzer.rb +68 -0
  22. data/lib/site-inspector/endpoint/well_known.rb +95 -0
  23. data/lib/site-inspector/endpoint/whois.rb +56 -0
  24. data/lib/site-inspector/endpoint.rb +60 -23
  25. data/lib/site-inspector/formatter.rb +118 -0
  26. data/lib/site-inspector/version.rb +1 -1
  27. data/lib/site-inspector.rb +34 -30
  28. data/package-lock.json +670 -23
  29. data/package.json +2 -1
  30. data/renovate.json +4 -0
  31. data/script/benchmark +8 -0
  32. data/script/cibuild +2 -2
  33. data/script/console +1 -1
  34. data/script/pa11y-version +1 -1
  35. data/script/release +11 -4
  36. data/script/update-well-known +18 -0
  37. data/site-inspector.gemspec +16 -9
  38. data/spec/cliver/dependency_ext_spec.rb +21 -0
  39. data/spec/{site_inspector_domain_spec.rb → site_inspector/domain_spec.rb} +71 -9
  40. data/spec/{checks/site_inspector_endpoint_accessibility_spec.rb → site_inspector/endpoint/accessibility_spec.rb} +9 -7
  41. data/spec/{checks/site_inspector_endpoint_content_spec.rb → site_inspector/endpoint/content_spec.rb} +14 -4
  42. data/spec/{checks/site_inspector_endpoint_cookies_spec.rb → site_inspector/endpoint/cookies_spec.rb} +27 -14
  43. data/spec/{checks/site_inspector_endpoint_dns_spec.rb → site_inspector/endpoint/dns_spec.rb} +87 -9
  44. data/spec/{checks/site_inspector_endpoint_headers_spec.rb → site_inspector/endpoint/headers_spec.rb} +0 -7
  45. data/spec/{checks/site_inspector_endpoint_hsts_spec.rb → site_inspector/endpoint/hsts_spec.rb} +8 -8
  46. data/spec/site_inspector/endpoint/wappalyzer_spec.rb +20 -0
  47. data/spec/site_inspector/endpoint/well_known_spec.rb +12 -0
  48. data/spec/{checks/site_inspector_endpoint_whois_spec.rb → site_inspector/endpoint/whois_spec.rb} +2 -1
  49. data/spec/{site_inspector_endpoint_spec.rb → site_inspector/endpoint_spec.rb} +59 -3
  50. data/spec/site_inspector_spec.rb +58 -11
  51. data/spec/spec_helper.rb +9 -2
  52. metadata +100 -92
  53. data/.travis.yml +0 -9
  54. data/lib/site-inspector/cache.rb +0 -17
  55. data/lib/site-inspector/checks/content.rb +0 -85
  56. data/lib/site-inspector/checks/headers.rb +0 -68
  57. data/lib/site-inspector/checks/sniffer.rb +0 -67
  58. data/lib/site-inspector/checks/wappalyzer.rb +0 -62
  59. data/lib/site-inspector/checks/whois.rb +0 -36
  60. data/lib/site-inspector/disk_cache.rb +0 -42
  61. data/lib/site-inspector/rails_cache.rb +0 -13
  62. data/spec/checks/site_inspector_endpoint_sniffer_spec.rb +0 -150
  63. data/spec/checks/site_inspector_endpoint_wappalyzer_spec.rb +0 -34
  64. data/spec/site_inspector_cache_spec.rb +0 -15
  65. data/spec/site_inspector_disk_cache_spec.rb +0 -39
  66. /data/lib/site-inspector/{checks → endpoint}/https.rb +0 -0
  67. /data/spec/{checks/site_inspector_endpoint_check_spec.rb → site_inspector/endpoint/check_spec.rb} +0 -0
  68. /data/spec/{checks/site_inspector_endpoint_https_spec.rb → site_inspector/endpoint/https_spec.rb} +0 -0
@@ -5,20 +5,16 @@ class SiteInspector
5
5
  attr_reader :host
6
6
 
7
7
  def initialize(host)
8
- host = host.downcase
9
- host = host.sub(/^https?:/, '')
10
- host = host.sub(%r{^/+}, '')
11
- host = host.sub(/^www\./, '')
12
- uri = Addressable::URI.parse "//#{host}"
13
- @host = uri.host
8
+ @host = DomainParser.parse(host)
9
+ @host = strip_www(@host) if @host&.trd
14
10
  end
15
11
 
16
12
  def endpoints
17
13
  @endpoints ||= [
18
- Endpoint.new("https://#{host}", domain: self),
19
- Endpoint.new("https://www.#{host}", domain: self),
20
- Endpoint.new("http://#{host}", domain: self),
21
- Endpoint.new("http://www.#{host}", domain: self)
14
+ Endpoint.build(self, https: true, www: false),
15
+ Endpoint.build(self, https: true, www: true),
16
+ Endpoint.build(self, https: false, www: false),
17
+ Endpoint.build(self, https: false, www: true)
22
18
  ]
23
19
  end
24
20
 
@@ -100,13 +96,14 @@ class SiteInspector
100
96
  # HTTPS is "downgraded" if both:
101
97
  #
102
98
  # * HTTPS is supported, and
103
- # * The 'canonical' endpoint gets an immediate internal redirect to HTTP.
99
+ # * Any HTTPS endpoint redirects to HTTP.
104
100
  #
105
- # TODO: the redirect must be internal.
101
+ # This correctly identifies domains where HTTPS is available but
102
+ # redirects to HTTP, regardless of which endpoint is canonical.
106
103
  def downgrades_https?
107
104
  return false unless https?
108
105
 
109
- canonical_endpoint.redirect? && canonical_endpoint.redirect.http?
106
+ endpoints.select(&:https?).any? { |e| e.redirect&.http? }
110
107
  end
111
108
 
112
109
  # A domain is "canonically" at www if:
@@ -197,7 +194,7 @@ class SiteInspector
197
194
  end
198
195
 
199
196
  def to_s
200
- host
197
+ host.to_s
201
198
  end
202
199
 
203
200
  def inspect
@@ -228,10 +225,15 @@ class SiteInspector
228
225
  #
229
226
  # Returns a complete hash of the domain's information
230
227
  def to_h(options = {})
228
+ return {} unless host
229
+
231
230
  prefetch
232
231
 
233
232
  hash = {
234
- host: host,
233
+ host: canonical_endpoint.host.to_s,
234
+ tld: canonical_endpoint.host.tld,
235
+ sld: canonical_endpoint.host.sld,
236
+ trd: canonical_endpoint.host.trd,
235
237
  up: up?,
236
238
  responds: responds?,
237
239
  www: www?,
@@ -267,5 +269,15 @@ class SiteInspector
267
269
  def to_json(*_args)
268
270
  to_h.to_json
269
271
  end
272
+
273
+ private
274
+
275
+ # Rebuild the domain without a leading `www` label
276
+ def strip_www(domain)
277
+ return domain unless domain.trd == 'www' || domain.trd.start_with?('www.')
278
+
279
+ trd = domain.trd.delete_prefix('www').delete_prefix('.')
280
+ PublicSuffix::Domain.new(domain.tld, domain.sld, trd.empty? ? nil : trd)
281
+ end
270
282
  end
271
283
  end
@@ -0,0 +1,11 @@
1
+ # frozen_string_literal: true
2
+
3
+ class SiteInspector
4
+ class DomainParser
5
+ include NaughtyOrNice
6
+
7
+ def self.parse(domain)
8
+ new(domain).domain
9
+ end
10
+ end
11
+ end
@@ -39,9 +39,8 @@ class SiteInspector
39
39
 
40
40
  def pa11y
41
41
  @pa11y ||= begin
42
- node_bin = File.expand_path('../../../node_modules/pa11y/bin', File.dirname(__FILE__))
43
- path = ['*', node_bin].join(File::PATH_SEPARATOR)
44
- Cliver::Dependency.new('pa11y.js', REQUIRED_PA11Y_VERSION, path: path)
42
+ path = ['*', './bin', './node_modules/.bin'].join(File::PATH_SEPARATOR)
43
+ Cliver::Dependency.new('pa11y', REQUIRED_PA11Y_VERSION, path:)
45
44
  end
46
45
  end
47
46
 
@@ -83,13 +82,16 @@ class SiteInspector
83
82
  end
84
83
 
85
84
  def check
85
+ return {} unless endpoint.up?
86
+ return {} if endpoint.redirect?
87
+
86
88
  @check ||= run_pa11y(standard)
87
89
  rescue Pa11yError
88
90
  nil
89
91
  end
90
92
  alias to_h check
91
93
 
92
- def method_missing(method_sym, *arguments, &block)
94
+ def method_missing(method_sym, *arguments, &)
93
95
  if standard?(method_sym)
94
96
  run_pa11y(method_sym)
95
97
  else
@@ -23,7 +23,7 @@ class SiteInspector
23
23
  end
24
24
 
25
25
  def host
26
- request.base_url.host
26
+ endpoint.host.to_s
27
27
  end
28
28
 
29
29
  def inspect
@@ -34,6 +34,10 @@ class SiteInspector
34
34
  self.class.name
35
35
  end
36
36
 
37
+ def uri_for(_key)
38
+ nil
39
+ end
40
+
37
41
  class << self
38
42
  @@enabled = true
39
43
 
@@ -0,0 +1,151 @@
1
+ # frozen_string_literal: true
2
+
3
+ class SiteInspector
4
+ class Endpoint
5
+ class Content < Check
6
+ PATHS = [
7
+ 'robots.txt', 'sitemap.xml', 'humans.txt', 'vulnerability-disclosure-policy', 'security.txt', '.well-known/security.txt', 'data.json'
8
+ ].freeze
9
+
10
+ class << self
11
+ def paths
12
+ @paths ||= PATHS.to_h { |p| [key_for(p), p] }
13
+ end
14
+
15
+ def path_for(key)
16
+ paths[key]
17
+ end
18
+
19
+ def key_for(path)
20
+ path.gsub(/(\.|-)/, '_').to_sym
21
+ end
22
+ end
23
+
24
+ # Given a path (e.g, "/data"), check if the given path exists on the canonical endpoint
25
+ def path_exists?(path)
26
+ return unless proper_404s?
27
+
28
+ @exists ||= {}
29
+ @exists[path] ||= endpoint.up? && endpoint.request(path:, followlocation: true).success?
30
+ rescue URI::InvalidURIError
31
+ false
32
+ end
33
+
34
+ # The default Check#response method is from a HEAD request
35
+ # The content check has a special response which includes the body from a GET request
36
+ def response
37
+ @response ||= endpoint.request(method: :get, followlocation: true)
38
+ end
39
+
40
+ def document
41
+ require 'nokogiri'
42
+ @doc ||= Nokogiri::HTML response.body if response
43
+ end
44
+ alias doc document
45
+
46
+ def body
47
+ @body ||= document.to_s.force_encoding('UTF-8').encode('UTF-8', invalid: :replace, replace: '')
48
+ end
49
+
50
+ def method_missing(method_sym, *arguments, &)
51
+ key = method_sym.to_s.gsub(/\?$/, '').to_sym
52
+ if respond_to_missing?(key)
53
+ path = self.class.paths[key]
54
+ path_exists?(path)
55
+ else
56
+ super
57
+ end
58
+ end
59
+
60
+ def respond_to_missing?(method_sym, include_private = false)
61
+ if self.class.paths.key?(method_sym)
62
+ true
63
+ else
64
+ super
65
+ end
66
+ end
67
+
68
+ def security_txt?
69
+ @security_txt ||= if proper_404s?
70
+ path_exists?('security.txt') || path_exists?('.well-known/security.txt')
71
+ else
72
+ false
73
+ end
74
+ end
75
+
76
+ def uri_for(key)
77
+ return nil unless self.class.paths.key?(key)
78
+
79
+ endpoint.join(self.class.paths[key]) if proper_404s?
80
+ end
81
+
82
+ def doctype
83
+ document.internal_subset.external_id
84
+ end
85
+
86
+ def generator
87
+ @generator ||= begin
88
+ tag = document.at('meta[name="generator"]')
89
+ tag['content'] if tag
90
+ end
91
+ end
92
+
93
+ def prefetch
94
+ return unless endpoint.up?
95
+
96
+ paths = self.class.paths.values.concat(random_paths)
97
+ paths.each do |path|
98
+ request = endpoint.build_request(path:, followlocation: true)
99
+ SiteInspector.hydra.queue(request)
100
+ end
101
+
102
+ # Request for content is a GET, not a HEAD request
103
+ request = endpoint.build_request(method: :get, followlocation: true)
104
+ SiteInspector.hydra.queue(request)
105
+
106
+ SiteInspector.hydra.run
107
+ end
108
+
109
+ def proper_404s?
110
+ @proper_404s ||= begin
111
+ return unless endpoint.up?
112
+
113
+ random_paths.all? do |random_path|
114
+ endpoint.request(path: random_path, followlocation: true).code == 404
115
+ end
116
+ end
117
+ end
118
+
119
+ def to_h
120
+ return {} unless endpoint.up?
121
+ return {} if endpoint.redirect?
122
+ return @hash if defined?(@hash)
123
+
124
+ prefetch
125
+ @hash = {
126
+ doctype:,
127
+ generator:,
128
+ proper_404s: proper_404s?
129
+ }
130
+
131
+ self.class.paths.each do |key, _path|
132
+ @hash[key] = send(key) unless key.to_s.start_with?('_')
133
+ end
134
+
135
+ @hash[:security_txt] = security_txt?
136
+ @hash
137
+ end
138
+
139
+ private
140
+
141
+ def random_paths
142
+ require 'securerandom'
143
+ @random_paths ||= [
144
+ SecureRandom.hex,
145
+ "#{SecureRandom.hex}.html",
146
+ "#{SecureRandom.hex}.json"
147
+ ]
148
+ end
149
+ end
150
+ end
151
+ end
@@ -1,12 +1,14 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require 'http-cookie'
4
+
3
5
  class SiteInspector
4
6
  class Endpoint
5
7
  class Cookies < Check
6
8
  def any?(&block)
7
9
  if cookie_header.nil? || cookie_header.empty?
8
10
  false
9
- elsif block_given?
11
+ elsif block
10
12
  all.any?(&block)
11
13
  else
12
14
  true
@@ -14,20 +16,26 @@ class SiteInspector
14
16
  end
15
17
  alias cookies? any?
16
18
 
19
+ # Returns an array of HTTP::Cookie objects parsed from the Set-Cookie headers
17
20
  def all
18
- @cookies ||= cookie_header.map { |c| CGI::Cookie.parse(c) } if cookies?
21
+ @cookies ||= cookie_header.flat_map { |c| HTTP::Cookie.parse(c, endpoint.uri.to_s) } if cookies?
19
22
  end
20
23
 
21
24
  def [](key)
22
- all.find { |cookie| cookie.keys.first == key } if cookies?
25
+ all.find { |cookie| cookie.name == key } if cookies?
23
26
  end
24
27
 
28
+ # Does at least one cookie set both the Secure and HttpOnly flags?
25
29
  def secure?
26
- pairs = cookie_header.join('; ').split('; ') # CGI::Cookies#Parse doesn't seem to like secure headers
27
- pairs.any? { |c| c.casecmp('secure').zero? } && pairs.any? { |c| c.casecmp('httponly').zero? }
30
+ return false unless cookies?
31
+
32
+ all.any? { |cookie| cookie.secure? && cookie.httponly? }
28
33
  end
29
34
 
30
35
  def to_h
36
+ return {} unless endpoint.up?
37
+ return {} if endpoint.redirect?
38
+
31
39
  {
32
40
  cookie?: any?,
33
41
  secure?: secure?
@@ -1,10 +1,16 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require 'ipaddr'
4
+
3
5
  class SiteInspector
4
6
  class Endpoint
5
7
  class Dns < Check
6
8
  class LocalhostError < StandardError; end
7
9
 
10
+ # Record types fetched for #records. ANY queries are deprecated
11
+ # (RFC 8482) and most resolvers answer them with a single HINFO record.
12
+ RECORD_TYPES = %w[A AAAA CNAME MX DNSKEY].freeze
13
+
8
14
  def self.resolver
9
15
  require 'dnsruby'
10
16
  @resolver ||= begin
@@ -14,18 +20,19 @@ class SiteInspector
14
20
  end
15
21
  end
16
22
 
17
- def query(type = 'ANY')
18
- SiteInspector::Endpoint::Dns.resolver.query(host.to_s, type).answer
19
- rescue Dnsruby::ResolvTimeout, Dnsruby::ServFail, Dnsruby::NXDomain
20
- []
23
+ def query(type = 'A')
24
+ lookup(host.to_s, type)
21
25
  end
22
26
 
23
27
  def records
24
- @records ||= query
28
+ @records ||= RECORD_TYPES.flat_map { |type| query(type) }.uniq(&:to_s)
25
29
  end
26
30
 
27
31
  def record?(type)
28
- records.any? { |record| record.type == type } || query(type).count != 0
32
+ return true if records.any? { |record| record.type == type }
33
+ return false if RECORD_TYPES.include?(type.to_s)
34
+
35
+ query(type).any?
29
36
  end
30
37
  alias has_record? record?
31
38
 
@@ -60,20 +67,30 @@ class SiteInspector
60
67
  end
61
68
 
62
69
  def localhost?
63
- ip == '127.0.0.1'
70
+ return false unless ip
71
+
72
+ IPAddr.new(ip).loopback?
73
+ rescue IPAddr::InvalidAddressError
74
+ false
64
75
  end
65
76
 
77
+ # The host's first IPv4 address, falling back to IPv6
66
78
  def ip
67
- @ip ||= Resolv.getaddress host
68
- rescue Resolv::ResolvError
69
- nil
79
+ @ip ||= begin
80
+ record = records.find { |r| r.type == 'A' } || records.find { |r| r.type == 'AAAA' }
81
+ record&.address&.to_s
82
+ end
70
83
  end
71
84
 
85
+ # The PTR hostname for #ip
72
86
  def hostname
73
- require 'resolv'
74
- @hostname ||= PublicSuffix.parse(Resolv.getname(ip))
75
- rescue Resolv::ResolvError, PublicSuffix::DomainInvalid
76
- nil
87
+ return @hostname if defined?(@hostname)
88
+ return @hostname = nil unless ip
89
+
90
+ ptr = lookup(ip, 'PTR').find { |record| record.type == 'PTR' }
91
+ @hostname = ptr ? PublicSuffix.parse(ptr.domainname.to_s) : nil
92
+ rescue PublicSuffix::DomainInvalid
93
+ @hostname = nil
77
94
  end
78
95
 
79
96
  def cnames
@@ -92,16 +109,23 @@ class SiteInspector
92
109
  {
93
110
  dnssec: dnssec?,
94
111
  ipv6: ipv6?,
95
- cdn: cdn,
96
- cloud_provider: cloud_provider,
112
+ cdn:,
113
+ cloud_provider:,
97
114
  google_apps: google_apps?,
98
- hostname: hostname,
99
- ip: ip
115
+ hostname: hostname.to_s,
116
+ ip:
100
117
  }
101
118
  end
102
119
 
103
120
  private
104
121
 
122
+ def lookup(name, type)
123
+ SiteInspector::Endpoint::Dns.resolver.query(name, type).answer
124
+ rescue Dnsruby::ResolvTimeout, Dnsruby::ResolvError => e
125
+ SiteInspector.logger.warn e.message
126
+ []
127
+ end
128
+
105
129
  def data
106
130
  @data ||= {}
107
131
  end
@@ -0,0 +1,55 @@
1
+ # frozen_string_literal: true
2
+
3
+ class SiteInspector
4
+ class Endpoint
5
+ class Headers < Check
6
+ COMMON = %w[
7
+ strict-transport-security
8
+ content-security-policy
9
+ x-frame-options
10
+ server
11
+ x-xss-protection
12
+ ].freeze
13
+
14
+ COMMON.each do |header|
15
+ name = header.gsub(/^x-/, '').tr('-', '_')
16
+
17
+ define_method name do
18
+ headers[header]
19
+ end
20
+
21
+ define_method "#{name}?" do
22
+ !!headers[header]
23
+ end
24
+ end
25
+
26
+ # more specific checks than presence of headers
27
+ def xss_protection?
28
+ xss_protection == '1; mode=block'
29
+ end
30
+
31
+ # Returns an array of hashes of downcased key/value header pairs (or an empty hash)
32
+ def all
33
+ @all ||= response&.headers ? response.headers.transform_keys(&:downcase) : {}
34
+ end
35
+ alias headers all
36
+
37
+ def custom_headers
38
+ @custom_headers ||= all.select do |header|
39
+ header.start_with?('x-') && !COMMON.include?(header)
40
+ end
41
+ end
42
+
43
+ def [](header)
44
+ headers[header]
45
+ end
46
+
47
+ def to_h
48
+ return {} if endpoint.redirect?
49
+
50
+ hash = COMMON.to_h { |h| [h, headers[h] || false] }
51
+ hash.merge custom_headers
52
+ end
53
+ end
54
+ end
55
+ end
@@ -5,14 +5,16 @@ class SiteInspector
5
5
  # Utility parser for HSTS headers.
6
6
  # RFC: http://tools.ietf.org/html/rfc6797
7
7
  class Hsts < Check
8
+ STATUS_ENDPOINT = 'https://hstspreload.org/api/v2/status'
9
+
8
10
  def valid?
9
- return false unless header
11
+ return false if header.empty?
10
12
 
11
13
  pairs.none? { |key, value| "#{key}#{value}" =~ /[\s'"]/ }
12
14
  end
13
15
 
14
16
  def max_age
15
- pairs[:"max-age"].to_i
17
+ pairs[:'max-age'].to_i
16
18
  end
17
19
 
18
20
  def include_subdomains?
@@ -37,11 +39,12 @@ class SiteInspector
37
39
  def to_h
38
40
  {
39
41
  valid: valid?,
40
- max_age: max_age,
42
+ max_age:,
41
43
  include_subdomains: include_subdomains?,
42
44
  preload: preload?,
43
45
  enabled: enabled?,
44
- preload_ready: preload_ready?
46
+ preload_ready: preload_ready?,
47
+ preload_list_status: preload_list['status']
45
48
  }
46
49
  end
47
50
 
@@ -52,7 +55,7 @@ class SiteInspector
52
55
  end
53
56
 
54
57
  def header
55
- @header ||= headers['strict-transport-security']
58
+ @header ||= [headers['strict-transport-security']].flatten.join('; ').squeeze(';')
56
59
  end
57
60
 
58
61
  def directives
@@ -76,6 +79,34 @@ class SiteInspector
76
79
  pairs
77
80
  end
78
81
  end
82
+
83
+ def request
84
+ @request ||= begin
85
+ options = SiteInspector.typhoeus_defaults
86
+ options = options.merge(method: :get)
87
+ Typhoeus::Request.new(url, options)
88
+ end
89
+ end
90
+
91
+ def url
92
+ url = Addressable::URI.parse(STATUS_ENDPOINT)
93
+ url.query_values = { domain: endpoint.host.to_s }
94
+ url
95
+ end
96
+
97
+ def preload_list
98
+ @preload_list ||= begin
99
+ SiteInspector.hydra.queue(request)
100
+ SiteInspector.hydra.run
101
+
102
+ response = request.response
103
+ if response.success?
104
+ JSON.parse(response.body)
105
+ else
106
+ {}
107
+ end
108
+ end
109
+ end
79
110
  end
80
111
  end
81
112
  end
@@ -0,0 +1,68 @@
1
+ # frozen_string_literal: true
2
+
3
+ class SiteInspector
4
+ class Endpoint
5
+ class Wappalyzer < Check
6
+ class WappalyzerError < RuntimeError; end
7
+
8
+ class << self
9
+ def wappalyzer?
10
+ return @wappalyzer_detected if defined? @wappalyzer_detected
11
+
12
+ @wappalyzer_detected = !!wappalyzer.detect
13
+ end
14
+
15
+ def enabled?
16
+ @@enabled && wappalyzer?
17
+ end
18
+
19
+ def wappalyzer
20
+ @wappalyzer ||= begin
21
+ path = ['*', './bin', './node_modules/.bin'].join(File::PATH_SEPARATOR)
22
+ Cliver::Dependency.new('wappalyzer', path:)
23
+ end
24
+ end
25
+
26
+ def run_command(args)
27
+ Open3.capture2e(wappalyzer.detect, *args)
28
+ end
29
+ end
30
+
31
+ def to_h
32
+ return {} unless endpoint.up?
33
+ return {} if endpoint.redirect?
34
+
35
+ @to_h ||= begin
36
+ technologies = {}
37
+
38
+ data['technologies']&.each do |t|
39
+ category = t['categories'].first
40
+ category = category ? category['name'] : 'other'
41
+ category = category.downcase.tr(' ', '_').to_sym
42
+ technologies[category] ||= []
43
+ technologies[category].push t['name']
44
+ end
45
+
46
+ technologies
47
+ end
48
+ end
49
+
50
+ private
51
+
52
+ def data
53
+ @data ||= begin
54
+ args = [endpoint.uri.to_s]
55
+ output, _status = self.class.run_command(args)
56
+ JSON.parse(output)
57
+ rescue Timeout::Error, IOError
58
+ { technologies: {
59
+ categories: ['timeout'],
60
+ name: "https://www.wappalyzer.com/lookup/#{endpoint.host}"
61
+ } }
62
+ rescue JSON::ParserError
63
+ raise WappalyzerError, "Command `wappalyzer #{args.join(' ')}` failed: #{output}"
64
+ end
65
+ end
66
+ end
67
+ end
68
+ end