gsc-cli 2.0.2 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/AUTH.md +205 -0
- data/FUNDING.md +120 -0
- data/README.md +463 -299
- data/bin/gsc +29158 -4921
- data/dist/gsc +29158 -4921
- data/lib/gsc/aio_hunter.rb +343 -0
- data/lib/gsc/answer_synthesizer.rb +157 -0
- data/lib/gsc/api.rb +53 -1
- data/lib/gsc/auth.rb +26 -0
- data/lib/gsc/backlinks_manager.rb +96 -0
- data/lib/gsc/brand_segmenter.rb +140 -0
- data/lib/gsc/cache_manager.rb +806 -0
- data/lib/gsc/cannibalization_analyzer.rb +141 -0
- data/lib/gsc/canonical_chains.rb +367 -0
- data/lib/gsc/citation_simulator.rb +339 -0
- data/lib/gsc/cli/aio_hunter.rb +154 -0
- data/lib/gsc/cli/analytics.rb +788 -0
- data/lib/gsc/cli/audit.rb +1976 -0
- data/lib/gsc/cli/base.rb +384 -0
- data/lib/gsc/cli/cache.rb +266 -0
- data/lib/gsc/cli/canonical.rb +223 -0
- data/lib/gsc/cli/citation_simulator.rb +152 -0
- data/lib/gsc/cli/dashboard.rb +354 -0
- data/lib/gsc/cli/doctor.rb +129 -0
- data/lib/gsc/cli/eeat.rb +125 -0
- data/lib/gsc/cli/ga4.rb +852 -0
- data/lib/gsc/cli/growth.rb +650 -0
- data/lib/gsc/cli/hreflang.rb +164 -0
- data/lib/gsc/cli/image_seo.rb +162 -0
- data/lib/gsc/cli/indexing.rb +458 -0
- data/lib/gsc/cli/intent_shift.rb +125 -0
- data/lib/gsc/cli/keyword_value.rb +134 -0
- data/lib/gsc/cli/keywords.rb +795 -0
- data/lib/gsc/cli/landing_roi.rb +308 -0
- data/lib/gsc/cli/low_ctr.rb +213 -0
- data/lib/gsc/cli/mobile_parity.rb +150 -0
- data/lib/gsc/cli/report.rb +100 -0
- data/lib/gsc/cli/rich_results.rb +172 -0
- data/lib/gsc/cli/schema_generate.rb +149 -0
- data/lib/gsc/cli/seasonal.rb +232 -0
- data/lib/gsc/cli/security.rb +153 -0
- data/lib/gsc/cli/setup.rb +1291 -0
- data/lib/gsc/cli/sitemap_tree.rb +143 -0
- data/lib/gsc/cli/skill_pack.rb +62 -0
- data/lib/gsc/cli/soft_404.rb +199 -0
- data/lib/gsc/cli/sparkline.rb +227 -0
- data/lib/gsc/cli/watchdog.rb +150 -0
- data/lib/gsc/cli/zombie_purger.rb +208 -0
- data/lib/gsc/cli.rb +706 -5111
- data/lib/gsc/cli_advanced.rb +1513 -0
- data/lib/gsc/client.rb +17 -2
- data/lib/gsc/color.rb +16 -1
- data/lib/gsc/command_registry.rb +47 -9
- data/lib/gsc/config.rb +11 -2
- data/lib/gsc/content_gap.rb +112 -0
- data/lib/gsc/ctr_curve.rb +115 -0
- data/lib/gsc/decay_predictor.rb +322 -0
- data/lib/gsc/doctor.rb +434 -0
- data/lib/gsc/eeat_auditor.rb +428 -0
- data/lib/gsc/entity_auditor.rb +229 -0
- data/lib/gsc/firewall_scanner.rb +733 -0
- data/lib/gsc/geo_auditor.rb +368 -0
- data/lib/gsc/google_suggest.rb +109 -0
- data/lib/gsc/google_trends.rb +8 -1
- data/lib/gsc/heading_validator.rb +283 -0
- data/lib/gsc/hreflang_validator.rb +412 -0
- data/lib/gsc/image_seo.rb +286 -0
- data/lib/gsc/indexing_queue.rb +179 -0
- data/lib/gsc/indexnow.rb +93 -0
- data/lib/gsc/intent_shift.rb +188 -0
- data/lib/gsc/internal_links.rb +249 -0
- data/lib/gsc/keyword_value.rb +191 -0
- data/lib/gsc/landing_roi.rb +195 -0
- data/lib/gsc/llms_generator.rb +425 -0
- data/lib/gsc/low_ctr_rewriter.rb +408 -0
- data/lib/gsc/mobile_parity.rb +222 -0
- data/lib/gsc/network_tracer.rb +93 -0
- data/lib/gsc/open_page_rank.rb +72 -0
- data/lib/gsc/page_analyzer.rb +47 -7
- data/lib/gsc/page_comparator.rb +108 -0
- data/lib/gsc/page_speed.rb +110 -0
- data/lib/gsc/prompts.rb +38 -29
- data/lib/gsc/questions_harvester.rb +178 -0
- data/lib/gsc/report_generator.rb +461 -0
- data/lib/gsc/rich_results.rb +388 -0
- data/lib/gsc/robots_checker.rb +114 -0
- data/lib/gsc/schema_generator.rb +788 -0
- data/lib/gsc/schema_validator.rb +120 -0
- data/lib/gsc/seasonal_predictor.rb +381 -0
- data/lib/gsc/security_scanner.rb +496 -0
- data/lib/gsc/serp_feature_detector.rb +359 -0
- data/lib/gsc/serp_preview.rb +152 -0
- data/lib/gsc/site_crawler.rb +113 -21
- data/lib/gsc/sitemap_loader.rb +15 -4
- data/lib/gsc/sitemap_tree.rb +301 -0
- data/lib/gsc/skill_pack.rb +195 -0
- data/lib/gsc/soft_404_analyzer.rb +385 -0
- data/lib/gsc/sparkline.rb +171 -0
- data/lib/gsc/speed_correlator.rb +416 -0
- data/lib/gsc/striking_playbook.rb +190 -0
- data/lib/gsc/title_optimizer.rb +420 -0
- data/lib/gsc/vault.rb +260 -0
- data/lib/gsc/version.rb +1 -1
- data/lib/gsc/watchdog.rb +235 -0
- data/lib/gsc/zombie_purger.rb +366 -0
- data/lib/gsc.rb +144 -0
- metadata +91 -2
|
@@ -0,0 +1,733 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'net/http'
|
|
5
|
+
require 'uri'
|
|
6
|
+
require 'json'
|
|
7
|
+
require 'zlib'
|
|
8
|
+
require 'stringio'
|
|
9
|
+
require 'openssl'
|
|
10
|
+
require 'time'
|
|
11
|
+
require_relative 'config' if File.exist?(File.expand_path('config.rb', __dir__))
|
|
12
|
+
require_relative 'color' if File.exist?(File.expand_path('color.rb', __dir__))
|
|
13
|
+
|
|
14
|
+
module GSC
|
|
15
|
+
class FirewallScanner
|
|
16
|
+
AI_BOTS = {
|
|
17
|
+
'GPTBot' => {
|
|
18
|
+
provider: 'OpenAI',
|
|
19
|
+
category: 'search_and_training',
|
|
20
|
+
critical: true,
|
|
21
|
+
default_ua: 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; GPTBot/1.2; +https://openai.com/gptbot)',
|
|
22
|
+
desc: 'ChatGPT Search & OpenAI foundation model training'
|
|
23
|
+
},
|
|
24
|
+
'ChatGPT-User' => {
|
|
25
|
+
provider: 'OpenAI',
|
|
26
|
+
category: 'user_browsing',
|
|
27
|
+
critical: true,
|
|
28
|
+
default_ua: 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; ChatGPT-User/1.0; +https://openai.com/bot)',
|
|
29
|
+
desc: 'Real-time browsing agent for ChatGPT user queries'
|
|
30
|
+
},
|
|
31
|
+
'ClaudeBot' => {
|
|
32
|
+
provider: 'Anthropic',
|
|
33
|
+
category: 'search_and_training',
|
|
34
|
+
critical: true,
|
|
35
|
+
default_ua: 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; ClaudeBot/1.0; +claudebot@anthropic.com)',
|
|
36
|
+
desc: 'Claude AI web search & knowledge retrieval'
|
|
37
|
+
},
|
|
38
|
+
'Claude-Web' => {
|
|
39
|
+
provider: 'Anthropic',
|
|
40
|
+
category: 'user_browsing',
|
|
41
|
+
critical: true,
|
|
42
|
+
default_ua: 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Claude-Web/1.0; +claudebot@anthropic.com)',
|
|
43
|
+
desc: 'Real-time browsing agent for Claude.ai sessions'
|
|
44
|
+
},
|
|
45
|
+
'PerplexityBot' => {
|
|
46
|
+
provider: 'Perplexity AI',
|
|
47
|
+
category: 'search',
|
|
48
|
+
critical: true,
|
|
49
|
+
default_ua: 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)',
|
|
50
|
+
desc: 'Perplexity conversational search crawler'
|
|
51
|
+
},
|
|
52
|
+
'Google-Extended' => {
|
|
53
|
+
provider: 'Google',
|
|
54
|
+
category: 'training',
|
|
55
|
+
critical: false,
|
|
56
|
+
default_ua: 'Mozilla/5.0 (compatible; Google-Extended; +https://developers.google.com/search/docs/crawling-indexing/google-extended)',
|
|
57
|
+
desc: 'Gemini and Vertex AI generative training'
|
|
58
|
+
},
|
|
59
|
+
'Applebot-Extended' => {
|
|
60
|
+
provider: 'Apple',
|
|
61
|
+
category: 'training',
|
|
62
|
+
critical: false,
|
|
63
|
+
default_ua: 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.4 Safari/605.1.15 (Applebot-Extended/0.1)',
|
|
64
|
+
desc: 'Apple Intelligence foundation model training'
|
|
65
|
+
},
|
|
66
|
+
'CCBot' => {
|
|
67
|
+
provider: 'Common Crawl',
|
|
68
|
+
category: 'training',
|
|
69
|
+
critical: false,
|
|
70
|
+
default_ua: 'CCBot/2.0 (https://commoncrawl.org/faq/)',
|
|
71
|
+
desc: 'Open-source web corpus for open LLMs (Llama, Mistral)'
|
|
72
|
+
},
|
|
73
|
+
'Bytespider' => {
|
|
74
|
+
provider: 'ByteDance',
|
|
75
|
+
category: 'search_and_training',
|
|
76
|
+
critical: false,
|
|
77
|
+
default_ua: 'Mozilla/5.0 (compatible; Bytespider; spider-feedback@bytedance.com)',
|
|
78
|
+
desc: 'TikTok & Doubao AI crawling engine'
|
|
79
|
+
},
|
|
80
|
+
'cohere-ai' => {
|
|
81
|
+
provider: 'Cohere',
|
|
82
|
+
category: 'training',
|
|
83
|
+
critical: false,
|
|
84
|
+
default_ua: 'Mozilla/5.0 (compatible; cohere-ai/1.0; +https://cohere.ai/bot)',
|
|
85
|
+
desc: 'Cohere enterprise LLM training crawler'
|
|
86
|
+
},
|
|
87
|
+
'FacebookBot' => {
|
|
88
|
+
provider: 'Meta',
|
|
89
|
+
category: 'training',
|
|
90
|
+
critical: false,
|
|
91
|
+
default_ua: 'Mozilla/5.0 (compatible; FacebookBot/1.0; +https://en-gb.facebook.com/webmasters/crawler)',
|
|
92
|
+
desc: 'Meta Llama web scraping & indexer'
|
|
93
|
+
},
|
|
94
|
+
'Amazonbot' => {
|
|
95
|
+
provider: 'Amazon',
|
|
96
|
+
category: 'search_and_training',
|
|
97
|
+
critical: false,
|
|
98
|
+
default_ua: 'Mozilla/5.0 (compatible; Amazonbot/0.1; +https://developer.amazon.com/support/amazonbot)',
|
|
99
|
+
desc: 'Amazon Alexa & Rufus shopping assistant'
|
|
100
|
+
}
|
|
101
|
+
}.freeze
|
|
102
|
+
|
|
103
|
+
PROBE_CLIENTS = [
|
|
104
|
+
{
|
|
105
|
+
id: :browser,
|
|
106
|
+
name: 'Chrome Desktop',
|
|
107
|
+
persona: 'Standard Browser',
|
|
108
|
+
ua: 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36'
|
|
109
|
+
},
|
|
110
|
+
{
|
|
111
|
+
id: :googlebot,
|
|
112
|
+
name: 'Googlebot Desktop',
|
|
113
|
+
persona: 'Google Search Indexer',
|
|
114
|
+
ua: 'Mozilla/5.0 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)'
|
|
115
|
+
},
|
|
116
|
+
{
|
|
117
|
+
id: :gptbot,
|
|
118
|
+
name: 'GPTBot',
|
|
119
|
+
persona: 'OpenAI Search/Training',
|
|
120
|
+
ua: 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; GPTBot/1.2; +https://openai.com/gptbot)'
|
|
121
|
+
},
|
|
122
|
+
{
|
|
123
|
+
id: :claudebot,
|
|
124
|
+
name: 'ClaudeBot',
|
|
125
|
+
persona: 'Anthropic AI Crawler',
|
|
126
|
+
ua: 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; ClaudeBot/1.0; +claudebot@anthropic.com)'
|
|
127
|
+
},
|
|
128
|
+
{
|
|
129
|
+
id: :perplexity,
|
|
130
|
+
name: 'PerplexityBot',
|
|
131
|
+
persona: 'Perplexity Search',
|
|
132
|
+
ua: 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)'
|
|
133
|
+
}
|
|
134
|
+
].freeze
|
|
135
|
+
|
|
136
|
+
attr_reader :url, :uri, :options
|
|
137
|
+
|
|
138
|
+
def initialize(url_or_domain = nil, options = {})
|
|
139
|
+
raw = url_or_domain.to_s.strip
|
|
140
|
+
raw = options[:domain] || Config.default_domain || '' if raw.empty?
|
|
141
|
+
raise ArgumentError, 'Target URL or domain required. Example: gsc firewall <domain>' if raw.empty?
|
|
142
|
+
raw = "https://#{raw}" unless raw =~ %r{^https?://}
|
|
143
|
+
@url = raw
|
|
144
|
+
@uri = URI.parse(@url)
|
|
145
|
+
@options = options
|
|
146
|
+
end
|
|
147
|
+
|
|
148
|
+
def scan
|
|
149
|
+
# 1. Fetch and parse robots.txt
|
|
150
|
+
robots_data = fetch_and_audit_robots
|
|
151
|
+
|
|
152
|
+
# 2. Run live HTTP probes across multiple user agents
|
|
153
|
+
probe_results = run_live_probes
|
|
154
|
+
|
|
155
|
+
# 3. Detect edge CDN and WAF infrastructure signatures
|
|
156
|
+
edge_infrastructure = detect_edge_infrastructure(probe_results)
|
|
157
|
+
|
|
158
|
+
# 4. Detect inadvertent silent bot blockade
|
|
159
|
+
blockade_diagnosis = analyze_blockade(probe_results, robots_data)
|
|
160
|
+
|
|
161
|
+
# 5. Compute AI Bot Accessibility Score (0-100) and grade
|
|
162
|
+
score_data = calculate_score(robots_data, probe_results, blockade_diagnosis)
|
|
163
|
+
|
|
164
|
+
# 6. Generate WAF Custom Rules & robots.txt remediation recipes
|
|
165
|
+
recipes = generate_remediation_recipes(edge_infrastructure, robots_data, blockade_diagnosis)
|
|
166
|
+
|
|
167
|
+
{
|
|
168
|
+
url: @url,
|
|
169
|
+
host: @uri.host,
|
|
170
|
+
timestamp: Time.now.utc.iso8601,
|
|
171
|
+
score: score_data[:score],
|
|
172
|
+
grade: score_data[:grade],
|
|
173
|
+
verdict: score_data[:verdict],
|
|
174
|
+
score_breakdown: score_data[:breakdown],
|
|
175
|
+
edge_infrastructure: edge_infrastructure,
|
|
176
|
+
blockade_diagnosis: blockade_diagnosis,
|
|
177
|
+
robots_txt: robots_data,
|
|
178
|
+
live_probes: probe_results,
|
|
179
|
+
remediation_recipes: recipes
|
|
180
|
+
}
|
|
181
|
+
end
|
|
182
|
+
|
|
183
|
+
private
|
|
184
|
+
|
|
185
|
+
def fetch_and_audit_robots
|
|
186
|
+
robots_url = "#{@uri.scheme}://#{@uri.host}:#{@uri.port}/robots.txt"
|
|
187
|
+
content = fetch_raw_robots_txt(robots_url)
|
|
188
|
+
|
|
189
|
+
audit_entries = {}
|
|
190
|
+
has_robots = !content.empty?
|
|
191
|
+
|
|
192
|
+
AI_BOTS.each do |bot_name, meta|
|
|
193
|
+
parsed = parse_robots_rules(content, bot_name)
|
|
194
|
+
audit_entries[bot_name] = {
|
|
195
|
+
provider: meta[:provider],
|
|
196
|
+
category: meta[:category],
|
|
197
|
+
critical: meta[:critical],
|
|
198
|
+
desc: meta[:desc],
|
|
199
|
+
status: parsed[:status], # :allowed, :restricted, :blocked
|
|
200
|
+
matched_agent: parsed[:matched_agent],
|
|
201
|
+
reason: parsed[:reason],
|
|
202
|
+
matched_rules: parsed[:rules]
|
|
203
|
+
}
|
|
204
|
+
end
|
|
205
|
+
|
|
206
|
+
# Also inspect baseline Googlebot
|
|
207
|
+
googlebot_parsed = parse_robots_rules(content, 'Googlebot')
|
|
208
|
+
audit_entries['Googlebot'] = {
|
|
209
|
+
provider: 'Google',
|
|
210
|
+
category: 'search',
|
|
211
|
+
critical: true,
|
|
212
|
+
desc: 'Google web search indexing crawler',
|
|
213
|
+
status: googlebot_parsed[:status],
|
|
214
|
+
matched_agent: googlebot_parsed[:matched_agent],
|
|
215
|
+
reason: googlebot_parsed[:reason],
|
|
216
|
+
matched_rules: googlebot_parsed[:rules]
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
{
|
|
220
|
+
url: robots_url,
|
|
221
|
+
present: has_robots,
|
|
222
|
+
byte_size: content.bytesize,
|
|
223
|
+
bots: audit_entries,
|
|
224
|
+
raw_snippet: content.lines.first(15).join
|
|
225
|
+
}
|
|
226
|
+
end
|
|
227
|
+
|
|
228
|
+
def fetch_raw_robots_txt(target_url)
|
|
229
|
+
uri = URI.parse(target_url)
|
|
230
|
+
http = Net::HTTP.new(uri.host, uri.port)
|
|
231
|
+
http.use_ssl = (uri.scheme == 'https')
|
|
232
|
+
http.open_timeout = 5
|
|
233
|
+
http.read_timeout = 7
|
|
234
|
+
|
|
235
|
+
req = Net::HTTP::Get.new(uri.request_uri)
|
|
236
|
+
req['User-Agent'] = "Mozilla/5.0 (compatible; GSC-BotAuditor/#{GSC::VERSION}; +https://apolloswave.com)"
|
|
237
|
+
req['Accept-Encoding'] = 'gzip'
|
|
238
|
+
|
|
239
|
+
res = http.request(req)
|
|
240
|
+
return '' unless res.code == '200'
|
|
241
|
+
|
|
242
|
+
raw = res.body || ''
|
|
243
|
+
if res['content-encoding'] =~ /gzip/i && !raw.empty?
|
|
244
|
+
begin
|
|
245
|
+
Zlib::GzipReader.new(StringIO.new(raw)).read
|
|
246
|
+
rescue StandardError
|
|
247
|
+
raw
|
|
248
|
+
end
|
|
249
|
+
else
|
|
250
|
+
raw
|
|
251
|
+
end.to_s.dup.force_encoding('UTF-8').scrub
|
|
252
|
+
rescue StandardError
|
|
253
|
+
''
|
|
254
|
+
end
|
|
255
|
+
|
|
256
|
+
def parse_robots_rules(robots_text, target_bot)
|
|
257
|
+
return { status: :allowed, matched_agent: nil, rules: [], reason: 'No robots.txt found (Default Allow)' } if robots_text.nil? || robots_text.empty?
|
|
258
|
+
|
|
259
|
+
target_bot_down = target_bot.downcase
|
|
260
|
+
lines = robots_text.lines.map(&:strip).reject { |l| l.start_with?('#') || l.empty? }
|
|
261
|
+
|
|
262
|
+
blocks = []
|
|
263
|
+
current_agents = []
|
|
264
|
+
current_rules = []
|
|
265
|
+
|
|
266
|
+
lines.each do |line|
|
|
267
|
+
if line =~ /^user-agent:\s*(.+)$/i
|
|
268
|
+
agent = $1.strip.downcase
|
|
269
|
+
if current_rules.any?
|
|
270
|
+
blocks << { agents: current_agents, rules: current_rules }
|
|
271
|
+
current_agents = [agent]
|
|
272
|
+
current_rules = []
|
|
273
|
+
else
|
|
274
|
+
current_agents << agent
|
|
275
|
+
end
|
|
276
|
+
elsif line =~ /^(allow|disallow):\s*(.*)$/i
|
|
277
|
+
type = $1.downcase.to_sym
|
|
278
|
+
path = $2.strip
|
|
279
|
+
current_rules << { type: type, path: path }
|
|
280
|
+
end
|
|
281
|
+
end
|
|
282
|
+
blocks << { agents: current_agents, rules: current_rules } if current_agents.any?
|
|
283
|
+
|
|
284
|
+
# Exact match first, then wildcard '*'
|
|
285
|
+
exact_block = blocks.find { |b| b[:agents].include?(target_bot_down) }
|
|
286
|
+
star_block = blocks.find { |b| b[:agents].include?('*') }
|
|
287
|
+
chosen_block = exact_block || star_block
|
|
288
|
+
|
|
289
|
+
if chosen_block.nil?
|
|
290
|
+
return { status: :allowed, matched_agent: nil, rules: [], reason: 'Allowed (No matching agent directive, implicit allow)' }
|
|
291
|
+
end
|
|
292
|
+
|
|
293
|
+
matched_agent = exact_block ? target_bot : '*'
|
|
294
|
+
rules = chosen_block[:rules]
|
|
295
|
+
|
|
296
|
+
root_disallow = rules.find { |r| r[:type] == :disallow && (r[:path] == '/' || r[:path] == '/*') }
|
|
297
|
+
root_allow = rules.find { |r| r[:type] == :allow && (r[:path] == '/' || r[:path] == '/*') }
|
|
298
|
+
|
|
299
|
+
if root_disallow && !root_allow
|
|
300
|
+
{
|
|
301
|
+
status: :blocked,
|
|
302
|
+
matched_agent: matched_agent,
|
|
303
|
+
rules: rules,
|
|
304
|
+
reason: "Disallow: #{root_disallow[:path]} blocks root access"
|
|
305
|
+
}
|
|
306
|
+
elsif rules.any? { |r| r[:type] == :disallow && !r[:path].empty? }
|
|
307
|
+
{
|
|
308
|
+
status: :restricted,
|
|
309
|
+
matched_agent: matched_agent,
|
|
310
|
+
rules: rules,
|
|
311
|
+
reason: "Partial restrictions present (#{rules.count { |r| r[:type] == :disallow }} disallow rules)"
|
|
312
|
+
}
|
|
313
|
+
else
|
|
314
|
+
{
|
|
315
|
+
status: :allowed,
|
|
316
|
+
matched_agent: matched_agent,
|
|
317
|
+
rules: rules,
|
|
318
|
+
reason: matched_agent == '*' ? 'Allowed via wildcard (*) rules' : "Explicitly allowed for #{target_bot}"
|
|
319
|
+
}
|
|
320
|
+
end
|
|
321
|
+
end
|
|
322
|
+
|
|
323
|
+
def run_live_probes
|
|
324
|
+
results = []
|
|
325
|
+
|
|
326
|
+
# Standard suite of probe clients
|
|
327
|
+
clients = PROBE_CLIENTS.dup
|
|
328
|
+
|
|
329
|
+
# If custom user agent specified in options
|
|
330
|
+
if @options[:user_agent] && !@options[:user_agent].to_s.strip.empty?
|
|
331
|
+
clients << {
|
|
332
|
+
id: :custom,
|
|
333
|
+
name: 'Custom User-Agent',
|
|
334
|
+
persona: 'CLI User Override',
|
|
335
|
+
ua: @options[:user_agent].to_s.strip
|
|
336
|
+
}
|
|
337
|
+
end
|
|
338
|
+
|
|
339
|
+
clients.each do |client|
|
|
340
|
+
probe_res = probe_http(@uri, client[:ua])
|
|
341
|
+
results << client.merge(probe_res)
|
|
342
|
+
end
|
|
343
|
+
|
|
344
|
+
results
|
|
345
|
+
end
|
|
346
|
+
|
|
347
|
+
def probe_http(target_uri, user_agent, timeout = 6)
|
|
348
|
+
start_t = Time.now
|
|
349
|
+
http = Net::HTTP.new(target_uri.host, target_uri.port)
|
|
350
|
+
http.use_ssl = (target_uri.scheme == 'https')
|
|
351
|
+
http.open_timeout = timeout
|
|
352
|
+
http.read_timeout = timeout
|
|
353
|
+
|
|
354
|
+
path = target_uri.request_uri.empty? ? '/' : target_uri.request_uri
|
|
355
|
+
req = Net::HTTP::Get.new(path)
|
|
356
|
+
req['User-Agent'] = user_agent
|
|
357
|
+
req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
|
|
358
|
+
req['Accept-Encoding'] = 'gzip'
|
|
359
|
+
|
|
360
|
+
res = http.request(req)
|
|
361
|
+
duration_ms = ((Time.now - start_t) * 1000).round(1)
|
|
362
|
+
|
|
363
|
+
status = res.code.to_i
|
|
364
|
+
raw_body = res.body || ''
|
|
365
|
+
body_str = if res['content-encoding'] =~ /gzip/i && !raw_body.empty?
|
|
366
|
+
begin
|
|
367
|
+
Zlib::GzipReader.new(StringIO.new(raw_body)).read
|
|
368
|
+
rescue StandardError
|
|
369
|
+
raw_body
|
|
370
|
+
end
|
|
371
|
+
else
|
|
372
|
+
raw_body
|
|
373
|
+
end
|
|
374
|
+
body_str = body_str.to_s.dup.force_encoding('UTF-8').scrub
|
|
375
|
+
|
|
376
|
+
headers_hash = {}
|
|
377
|
+
res.each_capitalized { |k, v| headers_hash[k.downcase] = v }
|
|
378
|
+
|
|
379
|
+
challenges = detect_waf_challenges(status, headers_hash, body_str)
|
|
380
|
+
|
|
381
|
+
verdict = if challenges.any?
|
|
382
|
+
:challenge
|
|
383
|
+
elsif status >= 200 && status < 300
|
|
384
|
+
:pass
|
|
385
|
+
elsif status >= 300 && status < 400
|
|
386
|
+
:redirect
|
|
387
|
+
elsif [403, 429, 503].include?(status)
|
|
388
|
+
:blocked
|
|
389
|
+
else
|
|
390
|
+
:error
|
|
391
|
+
end
|
|
392
|
+
|
|
393
|
+
{
|
|
394
|
+
status: status,
|
|
395
|
+
duration_ms: duration_ms,
|
|
396
|
+
verdict: verdict,
|
|
397
|
+
challenges: challenges,
|
|
398
|
+
headers: headers_hash,
|
|
399
|
+
body_length: body_str.bytesize,
|
|
400
|
+
title: extract_page_title(body_str)
|
|
401
|
+
}
|
|
402
|
+
rescue StandardError => e
|
|
403
|
+
duration_ms = ((Time.now - start_t) * 1000).round(1) rescue 0
|
|
404
|
+
{
|
|
405
|
+
status: 0,
|
|
406
|
+
duration_ms: duration_ms,
|
|
407
|
+
verdict: :error,
|
|
408
|
+
challenges: [],
|
|
409
|
+
headers: {},
|
|
410
|
+
body_length: 0,
|
|
411
|
+
title: '',
|
|
412
|
+
error: e.message
|
|
413
|
+
}
|
|
414
|
+
end
|
|
415
|
+
|
|
416
|
+
def detect_waf_challenges(status_code, headers, body)
|
|
417
|
+
indicators = []
|
|
418
|
+
|
|
419
|
+
# Cloudflare Turnstile / Managed Challenge
|
|
420
|
+
if headers['cf-mitigated'] == 'challenge' ||
|
|
421
|
+
body.include?('cf-turnstile') ||
|
|
422
|
+
body.include?('challenges.cloudflare.com') ||
|
|
423
|
+
body.include?('Just a moment...') ||
|
|
424
|
+
body.include?('Attention Required! | Cloudflare') ||
|
|
425
|
+
body.include?('cf-challenge-running')
|
|
426
|
+
indicators << 'Cloudflare Managed Challenge (Turnstile/JS Challenge)'
|
|
427
|
+
end
|
|
428
|
+
|
|
429
|
+
# Cloudflare 1020 / Super Bot Fight Mode block
|
|
430
|
+
if [403, 503].include?(status_code) && (headers['cf-ray'] || headers['server'].to_s.include?('cloudflare'))
|
|
431
|
+
indicators << 'Cloudflare Super Bot Fight Mode (403/503 Block)'
|
|
432
|
+
end
|
|
433
|
+
|
|
434
|
+
# AWS WAF Block
|
|
435
|
+
if headers['x-amzn-waf-action'] == 'block' || body.include?('Request blocked by AWS WAF')
|
|
436
|
+
indicators << 'AWS WAF Automated Block'
|
|
437
|
+
end
|
|
438
|
+
|
|
439
|
+
# Akamai Bot Manager
|
|
440
|
+
if status_code == 403 && headers['server'].to_s.include?('akamaighost')
|
|
441
|
+
indicators << 'Akamai Bot Manager Access Denied'
|
|
442
|
+
end
|
|
443
|
+
|
|
444
|
+
# DataDome
|
|
445
|
+
if headers['x-datadome'] || body.include?('datadome.js')
|
|
446
|
+
indicators << 'DataDome Anti-Bot Challenge'
|
|
447
|
+
end
|
|
448
|
+
|
|
449
|
+
# PerimeterX / HUMAN
|
|
450
|
+
if headers.keys.any? { |k| k.start_with?('x-px-') } || body.include?('captcha.px-cdn.net')
|
|
451
|
+
indicators << 'PerimeterX (HUMAN Security) Challenge'
|
|
452
|
+
end
|
|
453
|
+
|
|
454
|
+
# Generic CAPTCHA
|
|
455
|
+
if body.include?('g-recaptcha') || body.include?('hcaptcha.com')
|
|
456
|
+
indicators << 'Interactive CAPTCHA Challenge'
|
|
457
|
+
end
|
|
458
|
+
|
|
459
|
+
indicators
|
|
460
|
+
end
|
|
461
|
+
|
|
462
|
+
def detect_edge_infrastructure(probe_results)
|
|
463
|
+
# Aggregate headers across all probes
|
|
464
|
+
combined_headers = {}
|
|
465
|
+
probe_results.each do |pr|
|
|
466
|
+
(pr[:headers] || {}).each { |k, v| combined_headers[k] ||= v }
|
|
467
|
+
end
|
|
468
|
+
|
|
469
|
+
server = combined_headers['server'].to_s.downcase
|
|
470
|
+
detected = []
|
|
471
|
+
|
|
472
|
+
# Cloudflare
|
|
473
|
+
if combined_headers['cf-ray'] || server.include?('cloudflare') || combined_headers['cf-cache-status']
|
|
474
|
+
detected << {
|
|
475
|
+
name: 'Cloudflare',
|
|
476
|
+
role: 'Edge CDN, DDoS & WAF',
|
|
477
|
+
indicators: [
|
|
478
|
+
combined_headers['cf-ray'] ? "cf-ray: #{combined_headers['cf-ray']}" : nil,
|
|
479
|
+
combined_headers['cf-cache-status'] ? "cache: #{combined_headers['cf-cache-status']}" : nil,
|
|
480
|
+
server.include?('cloudflare') ? 'server: cloudflare' : nil
|
|
481
|
+
].compact
|
|
482
|
+
}
|
|
483
|
+
end
|
|
484
|
+
|
|
485
|
+
# Shopify Edge
|
|
486
|
+
if combined_headers['x-shopify-stage'] || combined_headers['x-shopid'] || combined_headers['x-shardid']
|
|
487
|
+
detected << {
|
|
488
|
+
name: 'Shopify Cloudflare Enterprise',
|
|
489
|
+
role: 'E-commerce Edge Platform',
|
|
490
|
+
indicators: ['x-shopify-stage header present']
|
|
491
|
+
}
|
|
492
|
+
end
|
|
493
|
+
|
|
494
|
+
# AWS CloudFront / WAF
|
|
495
|
+
if combined_headers['x-amz-cf-id'] || combined_headers['x-amzn-requestid'] || combined_headers['via'].to_s.include?('cloudfront')
|
|
496
|
+
detected << {
|
|
497
|
+
name: 'AWS CloudFront / WAF',
|
|
498
|
+
role: 'Cloud CDN & Security Perimeter',
|
|
499
|
+
indicators: [
|
|
500
|
+
combined_headers['x-amz-cf-id'] ? 'x-amz-cf-id present' : nil,
|
|
501
|
+
combined_headers['x-amzn-waf-action'] ? "WAF action: #{combined_headers['x-amzn-waf-action']}" : nil
|
|
502
|
+
].compact
|
|
503
|
+
}
|
|
504
|
+
end
|
|
505
|
+
|
|
506
|
+
# Fastly
|
|
507
|
+
if combined_headers['x-fastly-request-id'] || combined_headers['x-served-by'].to_s.include?('cache-')
|
|
508
|
+
detected << {
|
|
509
|
+
name: 'Fastly Edge',
|
|
510
|
+
role: 'Edge Cloud CDN',
|
|
511
|
+
indicators: ['x-fastly-request-id present']
|
|
512
|
+
}
|
|
513
|
+
end
|
|
514
|
+
|
|
515
|
+
# Akamai
|
|
516
|
+
if server.include?('akamaighost') || combined_headers['x-akamai-transformed']
|
|
517
|
+
detected << {
|
|
518
|
+
name: 'Akamai Edge',
|
|
519
|
+
role: 'Enterprise Bot Management & CDN',
|
|
520
|
+
indicators: ['Akamai Ghost Server identified']
|
|
521
|
+
}
|
|
522
|
+
end
|
|
523
|
+
|
|
524
|
+
# Vercel
|
|
525
|
+
if combined_headers['x-vercel-id'] || server.include?('vercel')
|
|
526
|
+
detected << {
|
|
527
|
+
name: 'Vercel Edge Network',
|
|
528
|
+
role: 'Serverless Edge & Middleware',
|
|
529
|
+
indicators: ['x-vercel-id present']
|
|
530
|
+
}
|
|
531
|
+
end
|
|
532
|
+
|
|
533
|
+
# DataDome
|
|
534
|
+
if combined_headers['x-datadome']
|
|
535
|
+
detected << {
|
|
536
|
+
name: 'DataDome Bot Protection',
|
|
537
|
+
role: 'AI Bot Mitigation Firewall',
|
|
538
|
+
indicators: ['x-datadome header present']
|
|
539
|
+
}
|
|
540
|
+
end
|
|
541
|
+
|
|
542
|
+
detected << { name: 'Standard Web Origin', role: 'Direct Origin Server', indicators: ['Standard HTTP Server'] } if detected.empty?
|
|
543
|
+
|
|
544
|
+
detected
|
|
545
|
+
end
|
|
546
|
+
|
|
547
|
+
def analyze_blockade(probe_results, robots_data)
|
|
548
|
+
browser_probe = probe_results.find { |p| p[:id] == :browser }
|
|
549
|
+
googlebot_probe = probe_results.find { |p| p[:id] == :googlebot }
|
|
550
|
+
ai_probes = probe_results.reject { |p| [:browser, :googlebot].include?(p[:id]) }
|
|
551
|
+
|
|
552
|
+
browser_ok = browser_probe && [200, 301, 302].include?(browser_probe[:status])
|
|
553
|
+
googlebot_ok = googlebot_probe && [200, 301, 302].include?(googlebot_probe[:status])
|
|
554
|
+
|
|
555
|
+
blocked_ai_probes = ai_probes.select { |p| p[:verdict] == :blocked || p[:verdict] == :challenge }
|
|
556
|
+
silent_blockade = browser_ok && blocked_ai_probes.any?
|
|
557
|
+
|
|
558
|
+
# Check robots.txt disallow conflicts
|
|
559
|
+
robots_blocked_critical = (robots_data[:bots] || {}).select do |bot_name, info|
|
|
560
|
+
info[:critical] && info[:status] == :blocked
|
|
561
|
+
end
|
|
562
|
+
|
|
563
|
+
severity = if silent_blockade && blocked_ai_probes.size >= 2
|
|
564
|
+
:critical
|
|
565
|
+
elsif silent_blockade || robots_blocked_critical.any?
|
|
566
|
+
:high
|
|
567
|
+
elsif ai_probes.any? { |p| p[:verdict] == :challenge }
|
|
568
|
+
:moderate
|
|
569
|
+
else
|
|
570
|
+
:none
|
|
571
|
+
end
|
|
572
|
+
|
|
573
|
+
{
|
|
574
|
+
silent_blockade_detected: silent_blockade,
|
|
575
|
+
severity: severity,
|
|
576
|
+
browser_accessible: browser_ok,
|
|
577
|
+
googlebot_accessible: googlebot_ok,
|
|
578
|
+
blocked_probes: blocked_ai_probes.map { |p| { id: p[:id], name: p[:name], status: p[:status], verdict: p[:verdict], challenges: p[:challenges] } },
|
|
579
|
+
robots_critical_blocked: robots_blocked_critical.keys,
|
|
580
|
+
summary: if silent_blockade
|
|
581
|
+
"🚨 SILENT BOT BLOCKADE: Standard browsers receive HTTP 200, but AI search crawlers (#{blocked_ai_probes.map { |p| p[:name] }.join(', ')}) are challenged or blocked by edge firewall rules."
|
|
582
|
+
elsif robots_blocked_critical.any?
|
|
583
|
+
"⚠️ ROBOTS.TXT BLOCKADE: Critical AI search bots (#{robots_blocked_critical.keys.join(', ')}) are blocked from crawling in /robots.txt."
|
|
584
|
+
else
|
|
585
|
+
"✅ OPEN ACCESS: AI search crawlers pass both edge firewall inspection and robots.txt governance."
|
|
586
|
+
end
|
|
587
|
+
}
|
|
588
|
+
end
|
|
589
|
+
|
|
590
|
+
def calculate_score(robots_data, probe_results, blockade)
|
|
591
|
+
# 1. Robots score (0 - 50 pts)
|
|
592
|
+
critical_bots = ['GPTBot', 'ChatGPT-User', 'ClaudeBot', 'PerplexityBot']
|
|
593
|
+
extended_bots = ['Google-Extended', 'Applebot-Extended']
|
|
594
|
+
|
|
595
|
+
robots_score = 0
|
|
596
|
+
critical_bots.each do |b|
|
|
597
|
+
status = robots_data.dig(:bots, b, :status)
|
|
598
|
+
robots_score += 10 if status == :allowed
|
|
599
|
+
robots_score += 6 if status == :restricted
|
|
600
|
+
end
|
|
601
|
+
|
|
602
|
+
extended_bots.each do |b|
|
|
603
|
+
status = robots_data.dig(:bots, b, :status)
|
|
604
|
+
robots_score += 5 if status == :allowed
|
|
605
|
+
robots_score += 3 if status == :restricted
|
|
606
|
+
end
|
|
607
|
+
|
|
608
|
+
# 2. Live probe score (0 - 50 pts)
|
|
609
|
+
probe_score = 0
|
|
610
|
+
probe_results.each do |pr|
|
|
611
|
+
next if pr[:id] == :custom # Don't skew default scoring
|
|
612
|
+
case pr[:verdict]
|
|
613
|
+
when :pass, :redirect
|
|
614
|
+
probe_score += 10
|
|
615
|
+
when :challenge
|
|
616
|
+
probe_score += 3
|
|
617
|
+
end
|
|
618
|
+
end
|
|
619
|
+
|
|
620
|
+
# 3. Penalties
|
|
621
|
+
penalty = 0
|
|
622
|
+
penalty += 25 if blockade[:silent_blockade_detected]
|
|
623
|
+
penalty += 15 if blockade[:robots_critical_blocked].any?
|
|
624
|
+
|
|
625
|
+
raw_total = robots_score + probe_score - penalty
|
|
626
|
+
final_score = [[0, raw_total].max, 100].min
|
|
627
|
+
|
|
628
|
+
grade = case final_score
|
|
629
|
+
when 90..100 then 'A'
|
|
630
|
+
when 75..89 then 'B'
|
|
631
|
+
when 60..74 then 'C'
|
|
632
|
+
when 40..59 then 'D'
|
|
633
|
+
else 'F'
|
|
634
|
+
end
|
|
635
|
+
|
|
636
|
+
verdict = case grade
|
|
637
|
+
when 'A' then 'EXCELLENT: Fully Accessible & AI Search Ready'
|
|
638
|
+
when 'B' then 'GOOD: Accessible with minor training crawler restrictions'
|
|
639
|
+
when 'C' then 'MODERATE: Partial WAF challenges or restricted bots detected'
|
|
640
|
+
when 'D' then 'POOR: Key AI search assistants (ChatGPT/Claude/Perplexity) blocked'
|
|
641
|
+
else 'CRITICAL: Severe AI Search Blockade active'
|
|
642
|
+
end
|
|
643
|
+
|
|
644
|
+
{
|
|
645
|
+
score: final_score,
|
|
646
|
+
grade: grade,
|
|
647
|
+
verdict: verdict,
|
|
648
|
+
breakdown: {
|
|
649
|
+
robots_governance: { score: robots_score, max: 50 },
|
|
650
|
+
firewall_passthrough: { score: probe_score, max: 50 },
|
|
651
|
+
penalties: penalty
|
|
652
|
+
}
|
|
653
|
+
}
|
|
654
|
+
end
|
|
655
|
+
|
|
656
|
+
def generate_remediation_recipes(edge_infra, robots_data, blockade)
|
|
657
|
+
is_cloudflare = edge_infra.any? { |i| i[:name].include?('Cloudflare') }
|
|
658
|
+
is_aws = edge_infra.any? { |i| i[:name].include?('AWS') }
|
|
659
|
+
|
|
660
|
+
cf_expression = '(http.user_agent contains "GPTBot" or ' \
|
|
661
|
+
'http.user_agent contains "ChatGPT-User" or ' \
|
|
662
|
+
'http.user_agent contains "ClaudeBot" or ' \
|
|
663
|
+
'http.user_agent contains "Claude-Web" or ' \
|
|
664
|
+
'http.user_agent contains "PerplexityBot")'
|
|
665
|
+
|
|
666
|
+
cf_rule = {
|
|
667
|
+
name: 'Allow Legitimate AI Search Engines & User Browsing',
|
|
668
|
+
filter_expression: cf_expression,
|
|
669
|
+
action: 'Skip',
|
|
670
|
+
skip_features: [
|
|
671
|
+
'WAF Managed Rules',
|
|
672
|
+
'Super Bot Fight Mode (Managed Challenge)',
|
|
673
|
+
'Rate Limiting Rules'
|
|
674
|
+
],
|
|
675
|
+
dashboard_instructions: [
|
|
676
|
+
'1. Log into Cloudflare Dashboard -> Security -> WAF -> Custom Rules.',
|
|
677
|
+
'2. Click "Create rule" with Name: "Allow Verified AI Search Engines".',
|
|
678
|
+
"3. Set expression to:\n #{cf_expression}",
|
|
679
|
+
'4. Select Action: "Skip" and check "All remaining custom rules", "Super Bot Fight Mode", and "Managed Rules".',
|
|
680
|
+
'5. Save and Deploy to prevent silent HTTP 403 / Turnstile challenges on AI search bots.'
|
|
681
|
+
]
|
|
682
|
+
}
|
|
683
|
+
|
|
684
|
+
robots_recommendation = <<~ROBOTS
|
|
685
|
+
# ==============================================================================
|
|
686
|
+
# AI Search Engine & Citability Governance (Optimized for ChatGPT, Claude, Perplexity)
|
|
687
|
+
# ==============================================================================
|
|
688
|
+
|
|
689
|
+
# Allow Real-Time AI Search Assistants (ChatGPT Search & Retrieval)
|
|
690
|
+
User-agent: GPTBot
|
|
691
|
+
Allow: /
|
|
692
|
+
|
|
693
|
+
User-agent: ChatGPT-User
|
|
694
|
+
Allow: /
|
|
695
|
+
|
|
696
|
+
# Allow Anthropic Claude Search & Retrieval
|
|
697
|
+
User-agent: ClaudeBot
|
|
698
|
+
Allow: /
|
|
699
|
+
|
|
700
|
+
User-agent: Claude-Web
|
|
701
|
+
Allow: /
|
|
702
|
+
|
|
703
|
+
# Allow Perplexity Conversational Search
|
|
704
|
+
User-agent: PerplexityBot
|
|
705
|
+
Allow: /
|
|
706
|
+
|
|
707
|
+
# (Optional) Restrict model training if proprietary content protection is needed:
|
|
708
|
+
# User-agent: Google-Extended
|
|
709
|
+
# Disallow: /
|
|
710
|
+
# User-agent: CCBot
|
|
711
|
+
# Disallow: /
|
|
712
|
+
ROBOTS
|
|
713
|
+
|
|
714
|
+
{
|
|
715
|
+
cloudflare_waf_rule: is_cloudflare || is_aws ? cf_rule : nil,
|
|
716
|
+
recommended_robots_txt: robots_recommendation,
|
|
717
|
+
action_items: [
|
|
718
|
+
blockade[:silent_blockade_detected] ? 'CRITICAL: Create WAF bypass rule for GPTBot and ClaudeBot to stop 403 challenges.' : nil,
|
|
719
|
+
blockade[:robots_critical_blocked].any? ? "ROBOTS: Remove Disallow: / for #{blockade[:robots_critical_blocked].join(', ')}." : nil,
|
|
720
|
+
'MONITOR: Periodically run `gsc firewall <url>` to verify ongoing AI crawler accessibility.'
|
|
721
|
+
].compact
|
|
722
|
+
}
|
|
723
|
+
end
|
|
724
|
+
|
|
725
|
+
def extract_page_title(html_str)
|
|
726
|
+
if html_str =~ /<title[^>]*>(.*?)<\/title>/im
|
|
727
|
+
$1.to_s.strip.gsub(/\s+/, ' ')
|
|
728
|
+
else
|
|
729
|
+
''
|
|
730
|
+
end
|
|
731
|
+
end
|
|
732
|
+
end
|
|
733
|
+
end
|