gsc-cli 2.1.0 → 2.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. checksums.yaml +4 -4
  2. data/AUTH.md +4 -1
  3. data/README.md +448 -408
  4. data/bin/gsc +28067 -5661
  5. data/dist/gsc +29121 -5046
  6. data/lib/gsc/aio_hunter.rb +343 -0
  7. data/lib/gsc/answer_synthesizer.rb +157 -0
  8. data/lib/gsc/api.rb +53 -1
  9. data/lib/gsc/auth.rb +26 -0
  10. data/lib/gsc/brand_segmenter.rb +140 -0
  11. data/lib/gsc/cache_manager.rb +806 -0
  12. data/lib/gsc/cannibalization_analyzer.rb +141 -0
  13. data/lib/gsc/canonical_chains.rb +367 -0
  14. data/lib/gsc/citation_simulator.rb +339 -0
  15. data/lib/gsc/cli/aio_hunter.rb +154 -0
  16. data/lib/gsc/cli/analytics.rb +788 -0
  17. data/lib/gsc/cli/audit.rb +1976 -0
  18. data/lib/gsc/cli/base.rb +384 -0
  19. data/lib/gsc/cli/cache.rb +266 -0
  20. data/lib/gsc/cli/canonical.rb +223 -0
  21. data/lib/gsc/cli/citation_simulator.rb +152 -0
  22. data/lib/gsc/cli/dashboard.rb +354 -0
  23. data/lib/gsc/cli/doctor.rb +129 -0
  24. data/lib/gsc/cli/eeat.rb +125 -0
  25. data/lib/gsc/cli/ga4.rb +852 -0
  26. data/lib/gsc/cli/growth.rb +650 -0
  27. data/lib/gsc/cli/hreflang.rb +164 -0
  28. data/lib/gsc/cli/image_seo.rb +162 -0
  29. data/lib/gsc/cli/indexing.rb +458 -0
  30. data/lib/gsc/cli/intent_shift.rb +125 -0
  31. data/lib/gsc/cli/keyword_value.rb +134 -0
  32. data/lib/gsc/cli/keywords.rb +795 -0
  33. data/lib/gsc/cli/landing_roi.rb +308 -0
  34. data/lib/gsc/cli/low_ctr.rb +213 -0
  35. data/lib/gsc/cli/mobile_parity.rb +150 -0
  36. data/lib/gsc/cli/report.rb +100 -0
  37. data/lib/gsc/cli/rich_results.rb +172 -0
  38. data/lib/gsc/cli/schema_generate.rb +149 -0
  39. data/lib/gsc/cli/seasonal.rb +232 -0
  40. data/lib/gsc/cli/security.rb +153 -0
  41. data/lib/gsc/cli/setup.rb +1291 -0
  42. data/lib/gsc/cli/sitemap_tree.rb +143 -0
  43. data/lib/gsc/cli/skill_pack.rb +62 -0
  44. data/lib/gsc/cli/soft_404.rb +199 -0
  45. data/lib/gsc/cli/sparkline.rb +227 -0
  46. data/lib/gsc/cli/watchdog.rb +150 -0
  47. data/lib/gsc/cli/zombie_purger.rb +208 -0
  48. data/lib/gsc/cli.rb +707 -5265
  49. data/lib/gsc/cli_advanced.rb +987 -44
  50. data/lib/gsc/client.rb +17 -2
  51. data/lib/gsc/color.rb +16 -1
  52. data/lib/gsc/command_registry.rb +47 -9
  53. data/lib/gsc/config.rb +2 -2
  54. data/lib/gsc/ctr_curve.rb +115 -0
  55. data/lib/gsc/decay_predictor.rb +322 -0
  56. data/lib/gsc/doctor.rb +434 -0
  57. data/lib/gsc/eeat_auditor.rb +428 -0
  58. data/lib/gsc/entity_auditor.rb +229 -0
  59. data/lib/gsc/firewall_scanner.rb +733 -0
  60. data/lib/gsc/geo_auditor.rb +368 -0
  61. data/lib/gsc/google_trends.rb +8 -1
  62. data/lib/gsc/heading_validator.rb +283 -0
  63. data/lib/gsc/hreflang_validator.rb +412 -0
  64. data/lib/gsc/image_seo.rb +286 -0
  65. data/lib/gsc/indexing_queue.rb +179 -0
  66. data/lib/gsc/indexnow.rb +93 -0
  67. data/lib/gsc/intent_shift.rb +188 -0
  68. data/lib/gsc/internal_links.rb +153 -36
  69. data/lib/gsc/keyword_value.rb +191 -0
  70. data/lib/gsc/landing_roi.rb +195 -0
  71. data/lib/gsc/llms_generator.rb +343 -22
  72. data/lib/gsc/low_ctr_rewriter.rb +408 -0
  73. data/lib/gsc/mobile_parity.rb +222 -0
  74. data/lib/gsc/network_tracer.rb +8 -1
  75. data/lib/gsc/page_analyzer.rb +47 -7
  76. data/lib/gsc/prompts.rb +38 -29
  77. data/lib/gsc/questions_harvester.rb +178 -0
  78. data/lib/gsc/report_generator.rb +461 -0
  79. data/lib/gsc/rich_results.rb +388 -0
  80. data/lib/gsc/robots_checker.rb +46 -15
  81. data/lib/gsc/schema_generator.rb +788 -0
  82. data/lib/gsc/schema_validator.rb +36 -38
  83. data/lib/gsc/seasonal_predictor.rb +381 -0
  84. data/lib/gsc/security_scanner.rb +496 -0
  85. data/lib/gsc/serp_feature_detector.rb +359 -0
  86. data/lib/gsc/serp_preview.rb +108 -22
  87. data/lib/gsc/site_crawler.rb +113 -21
  88. data/lib/gsc/sitemap_loader.rb +15 -4
  89. data/lib/gsc/sitemap_tree.rb +301 -0
  90. data/lib/gsc/skill_pack.rb +195 -0
  91. data/lib/gsc/soft_404_analyzer.rb +385 -0
  92. data/lib/gsc/sparkline.rb +171 -0
  93. data/lib/gsc/speed_correlator.rb +416 -0
  94. data/lib/gsc/striking_playbook.rb +190 -0
  95. data/lib/gsc/title_optimizer.rb +420 -0
  96. data/lib/gsc/vault.rb +260 -0
  97. data/lib/gsc/version.rb +1 -1
  98. data/lib/gsc/watchdog.rb +235 -0
  99. data/lib/gsc/zombie_purger.rb +366 -0
  100. data/lib/gsc.rb +118 -0
  101. metadata +75 -1
@@ -0,0 +1,733 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'net/http'
5
+ require 'uri'
6
+ require 'json'
7
+ require 'zlib'
8
+ require 'stringio'
9
+ require 'openssl'
10
+ require 'time'
11
+ require_relative 'config' if File.exist?(File.expand_path('config.rb', __dir__))
12
+ require_relative 'color' if File.exist?(File.expand_path('color.rb', __dir__))
13
+
14
+ module GSC
15
+ class FirewallScanner
16
+ AI_BOTS = {
17
+ 'GPTBot' => {
18
+ provider: 'OpenAI',
19
+ category: 'search_and_training',
20
+ critical: true,
21
+ default_ua: 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; GPTBot/1.2; +https://openai.com/gptbot)',
22
+ desc: 'ChatGPT Search & OpenAI foundation model training'
23
+ },
24
+ 'ChatGPT-User' => {
25
+ provider: 'OpenAI',
26
+ category: 'user_browsing',
27
+ critical: true,
28
+ default_ua: 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; ChatGPT-User/1.0; +https://openai.com/bot)',
29
+ desc: 'Real-time browsing agent for ChatGPT user queries'
30
+ },
31
+ 'ClaudeBot' => {
32
+ provider: 'Anthropic',
33
+ category: 'search_and_training',
34
+ critical: true,
35
+ default_ua: 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; ClaudeBot/1.0; +claudebot@anthropic.com)',
36
+ desc: 'Claude AI web search & knowledge retrieval'
37
+ },
38
+ 'Claude-Web' => {
39
+ provider: 'Anthropic',
40
+ category: 'user_browsing',
41
+ critical: true,
42
+ default_ua: 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Claude-Web/1.0; +claudebot@anthropic.com)',
43
+ desc: 'Real-time browsing agent for Claude.ai sessions'
44
+ },
45
+ 'PerplexityBot' => {
46
+ provider: 'Perplexity AI',
47
+ category: 'search',
48
+ critical: true,
49
+ default_ua: 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)',
50
+ desc: 'Perplexity conversational search crawler'
51
+ },
52
+ 'Google-Extended' => {
53
+ provider: 'Google',
54
+ category: 'training',
55
+ critical: false,
56
+ default_ua: 'Mozilla/5.0 (compatible; Google-Extended; +https://developers.google.com/search/docs/crawling-indexing/google-extended)',
57
+ desc: 'Gemini and Vertex AI generative training'
58
+ },
59
+ 'Applebot-Extended' => {
60
+ provider: 'Apple',
61
+ category: 'training',
62
+ critical: false,
63
+ default_ua: 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.4 Safari/605.1.15 (Applebot-Extended/0.1)',
64
+ desc: 'Apple Intelligence foundation model training'
65
+ },
66
+ 'CCBot' => {
67
+ provider: 'Common Crawl',
68
+ category: 'training',
69
+ critical: false,
70
+ default_ua: 'CCBot/2.0 (https://commoncrawl.org/faq/)',
71
+ desc: 'Open-source web corpus for open LLMs (Llama, Mistral)'
72
+ },
73
+ 'Bytespider' => {
74
+ provider: 'ByteDance',
75
+ category: 'search_and_training',
76
+ critical: false,
77
+ default_ua: 'Mozilla/5.0 (compatible; Bytespider; spider-feedback@bytedance.com)',
78
+ desc: 'TikTok & Doubao AI crawling engine'
79
+ },
80
+ 'cohere-ai' => {
81
+ provider: 'Cohere',
82
+ category: 'training',
83
+ critical: false,
84
+ default_ua: 'Mozilla/5.0 (compatible; cohere-ai/1.0; +https://cohere.ai/bot)',
85
+ desc: 'Cohere enterprise LLM training crawler'
86
+ },
87
+ 'FacebookBot' => {
88
+ provider: 'Meta',
89
+ category: 'training',
90
+ critical: false,
91
+ default_ua: 'Mozilla/5.0 (compatible; FacebookBot/1.0; +https://en-gb.facebook.com/webmasters/crawler)',
92
+ desc: 'Meta Llama web scraping & indexer'
93
+ },
94
+ 'Amazonbot' => {
95
+ provider: 'Amazon',
96
+ category: 'search_and_training',
97
+ critical: false,
98
+ default_ua: 'Mozilla/5.0 (compatible; Amazonbot/0.1; +https://developer.amazon.com/support/amazonbot)',
99
+ desc: 'Amazon Alexa & Rufus shopping assistant'
100
+ }
101
+ }.freeze
102
+
103
+ PROBE_CLIENTS = [
104
+ {
105
+ id: :browser,
106
+ name: 'Chrome Desktop',
107
+ persona: 'Standard Browser',
108
+ ua: 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36'
109
+ },
110
+ {
111
+ id: :googlebot,
112
+ name: 'Googlebot Desktop',
113
+ persona: 'Google Search Indexer',
114
+ ua: 'Mozilla/5.0 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)'
115
+ },
116
+ {
117
+ id: :gptbot,
118
+ name: 'GPTBot',
119
+ persona: 'OpenAI Search/Training',
120
+ ua: 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; GPTBot/1.2; +https://openai.com/gptbot)'
121
+ },
122
+ {
123
+ id: :claudebot,
124
+ name: 'ClaudeBot',
125
+ persona: 'Anthropic AI Crawler',
126
+ ua: 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; ClaudeBot/1.0; +claudebot@anthropic.com)'
127
+ },
128
+ {
129
+ id: :perplexity,
130
+ name: 'PerplexityBot',
131
+ persona: 'Perplexity Search',
132
+ ua: 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)'
133
+ }
134
+ ].freeze
135
+
136
+ attr_reader :url, :uri, :options
137
+
138
+ def initialize(url_or_domain = nil, options = {})
139
+ raw = url_or_domain.to_s.strip
140
+ raw = options[:domain] || Config.default_domain || '' if raw.empty?
141
+ raise ArgumentError, 'Target URL or domain required. Example: gsc firewall <domain>' if raw.empty?
142
+ raw = "https://#{raw}" unless raw =~ %r{^https?://}
143
+ @url = raw
144
+ @uri = URI.parse(@url)
145
+ @options = options
146
+ end
147
+
148
+ def scan
149
+ # 1. Fetch and parse robots.txt
150
+ robots_data = fetch_and_audit_robots
151
+
152
+ # 2. Run live HTTP probes across multiple user agents
153
+ probe_results = run_live_probes
154
+
155
+ # 3. Detect edge CDN and WAF infrastructure signatures
156
+ edge_infrastructure = detect_edge_infrastructure(probe_results)
157
+
158
+ # 4. Detect inadvertent silent bot blockade
159
+ blockade_diagnosis = analyze_blockade(probe_results, robots_data)
160
+
161
+ # 5. Compute AI Bot Accessibility Score (0-100) and grade
162
+ score_data = calculate_score(robots_data, probe_results, blockade_diagnosis)
163
+
164
+ # 6. Generate WAF Custom Rules & robots.txt remediation recipes
165
+ recipes = generate_remediation_recipes(edge_infrastructure, robots_data, blockade_diagnosis)
166
+
167
+ {
168
+ url: @url,
169
+ host: @uri.host,
170
+ timestamp: Time.now.utc.iso8601,
171
+ score: score_data[:score],
172
+ grade: score_data[:grade],
173
+ verdict: score_data[:verdict],
174
+ score_breakdown: score_data[:breakdown],
175
+ edge_infrastructure: edge_infrastructure,
176
+ blockade_diagnosis: blockade_diagnosis,
177
+ robots_txt: robots_data,
178
+ live_probes: probe_results,
179
+ remediation_recipes: recipes
180
+ }
181
+ end
182
+
183
+ private
184
+
185
+ def fetch_and_audit_robots
186
+ robots_url = "#{@uri.scheme}://#{@uri.host}:#{@uri.port}/robots.txt"
187
+ content = fetch_raw_robots_txt(robots_url)
188
+
189
+ audit_entries = {}
190
+ has_robots = !content.empty?
191
+
192
+ AI_BOTS.each do |bot_name, meta|
193
+ parsed = parse_robots_rules(content, bot_name)
194
+ audit_entries[bot_name] = {
195
+ provider: meta[:provider],
196
+ category: meta[:category],
197
+ critical: meta[:critical],
198
+ desc: meta[:desc],
199
+ status: parsed[:status], # :allowed, :restricted, :blocked
200
+ matched_agent: parsed[:matched_agent],
201
+ reason: parsed[:reason],
202
+ matched_rules: parsed[:rules]
203
+ }
204
+ end
205
+
206
+ # Also inspect baseline Googlebot
207
+ googlebot_parsed = parse_robots_rules(content, 'Googlebot')
208
+ audit_entries['Googlebot'] = {
209
+ provider: 'Google',
210
+ category: 'search',
211
+ critical: true,
212
+ desc: 'Google web search indexing crawler',
213
+ status: googlebot_parsed[:status],
214
+ matched_agent: googlebot_parsed[:matched_agent],
215
+ reason: googlebot_parsed[:reason],
216
+ matched_rules: googlebot_parsed[:rules]
217
+ }
218
+
219
+ {
220
+ url: robots_url,
221
+ present: has_robots,
222
+ byte_size: content.bytesize,
223
+ bots: audit_entries,
224
+ raw_snippet: content.lines.first(15).join
225
+ }
226
+ end
227
+
228
+ def fetch_raw_robots_txt(target_url)
229
+ uri = URI.parse(target_url)
230
+ http = Net::HTTP.new(uri.host, uri.port)
231
+ http.use_ssl = (uri.scheme == 'https')
232
+ http.open_timeout = 5
233
+ http.read_timeout = 7
234
+
235
+ req = Net::HTTP::Get.new(uri.request_uri)
236
+ req['User-Agent'] = "Mozilla/5.0 (compatible; GSC-BotAuditor/#{GSC::VERSION}; +https://apolloswave.com)"
237
+ req['Accept-Encoding'] = 'gzip'
238
+
239
+ res = http.request(req)
240
+ return '' unless res.code == '200'
241
+
242
+ raw = res.body || ''
243
+ if res['content-encoding'] =~ /gzip/i && !raw.empty?
244
+ begin
245
+ Zlib::GzipReader.new(StringIO.new(raw)).read
246
+ rescue StandardError
247
+ raw
248
+ end
249
+ else
250
+ raw
251
+ end.to_s.dup.force_encoding('UTF-8').scrub
252
+ rescue StandardError
253
+ ''
254
+ end
255
+
256
+ def parse_robots_rules(robots_text, target_bot)
257
+ return { status: :allowed, matched_agent: nil, rules: [], reason: 'No robots.txt found (Default Allow)' } if robots_text.nil? || robots_text.empty?
258
+
259
+ target_bot_down = target_bot.downcase
260
+ lines = robots_text.lines.map(&:strip).reject { |l| l.start_with?('#') || l.empty? }
261
+
262
+ blocks = []
263
+ current_agents = []
264
+ current_rules = []
265
+
266
+ lines.each do |line|
267
+ if line =~ /^user-agent:\s*(.+)$/i
268
+ agent = $1.strip.downcase
269
+ if current_rules.any?
270
+ blocks << { agents: current_agents, rules: current_rules }
271
+ current_agents = [agent]
272
+ current_rules = []
273
+ else
274
+ current_agents << agent
275
+ end
276
+ elsif line =~ /^(allow|disallow):\s*(.*)$/i
277
+ type = $1.downcase.to_sym
278
+ path = $2.strip
279
+ current_rules << { type: type, path: path }
280
+ end
281
+ end
282
+ blocks << { agents: current_agents, rules: current_rules } if current_agents.any?
283
+
284
+ # Exact match first, then wildcard '*'
285
+ exact_block = blocks.find { |b| b[:agents].include?(target_bot_down) }
286
+ star_block = blocks.find { |b| b[:agents].include?('*') }
287
+ chosen_block = exact_block || star_block
288
+
289
+ if chosen_block.nil?
290
+ return { status: :allowed, matched_agent: nil, rules: [], reason: 'Allowed (No matching agent directive, implicit allow)' }
291
+ end
292
+
293
+ matched_agent = exact_block ? target_bot : '*'
294
+ rules = chosen_block[:rules]
295
+
296
+ root_disallow = rules.find { |r| r[:type] == :disallow && (r[:path] == '/' || r[:path] == '/*') }
297
+ root_allow = rules.find { |r| r[:type] == :allow && (r[:path] == '/' || r[:path] == '/*') }
298
+
299
+ if root_disallow && !root_allow
300
+ {
301
+ status: :blocked,
302
+ matched_agent: matched_agent,
303
+ rules: rules,
304
+ reason: "Disallow: #{root_disallow[:path]} blocks root access"
305
+ }
306
+ elsif rules.any? { |r| r[:type] == :disallow && !r[:path].empty? }
307
+ {
308
+ status: :restricted,
309
+ matched_agent: matched_agent,
310
+ rules: rules,
311
+ reason: "Partial restrictions present (#{rules.count { |r| r[:type] == :disallow }} disallow rules)"
312
+ }
313
+ else
314
+ {
315
+ status: :allowed,
316
+ matched_agent: matched_agent,
317
+ rules: rules,
318
+ reason: matched_agent == '*' ? 'Allowed via wildcard (*) rules' : "Explicitly allowed for #{target_bot}"
319
+ }
320
+ end
321
+ end
322
+
323
+ def run_live_probes
324
+ results = []
325
+
326
+ # Standard suite of probe clients
327
+ clients = PROBE_CLIENTS.dup
328
+
329
+ # If custom user agent specified in options
330
+ if @options[:user_agent] && !@options[:user_agent].to_s.strip.empty?
331
+ clients << {
332
+ id: :custom,
333
+ name: 'Custom User-Agent',
334
+ persona: 'CLI User Override',
335
+ ua: @options[:user_agent].to_s.strip
336
+ }
337
+ end
338
+
339
+ clients.each do |client|
340
+ probe_res = probe_http(@uri, client[:ua])
341
+ results << client.merge(probe_res)
342
+ end
343
+
344
+ results
345
+ end
346
+
347
+ def probe_http(target_uri, user_agent, timeout = 6)
348
+ start_t = Time.now
349
+ http = Net::HTTP.new(target_uri.host, target_uri.port)
350
+ http.use_ssl = (target_uri.scheme == 'https')
351
+ http.open_timeout = timeout
352
+ http.read_timeout = timeout
353
+
354
+ path = target_uri.request_uri.empty? ? '/' : target_uri.request_uri
355
+ req = Net::HTTP::Get.new(path)
356
+ req['User-Agent'] = user_agent
357
+ req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
358
+ req['Accept-Encoding'] = 'gzip'
359
+
360
+ res = http.request(req)
361
+ duration_ms = ((Time.now - start_t) * 1000).round(1)
362
+
363
+ status = res.code.to_i
364
+ raw_body = res.body || ''
365
+ body_str = if res['content-encoding'] =~ /gzip/i && !raw_body.empty?
366
+ begin
367
+ Zlib::GzipReader.new(StringIO.new(raw_body)).read
368
+ rescue StandardError
369
+ raw_body
370
+ end
371
+ else
372
+ raw_body
373
+ end
374
+ body_str = body_str.to_s.dup.force_encoding('UTF-8').scrub
375
+
376
+ headers_hash = {}
377
+ res.each_capitalized { |k, v| headers_hash[k.downcase] = v }
378
+
379
+ challenges = detect_waf_challenges(status, headers_hash, body_str)
380
+
381
+ verdict = if challenges.any?
382
+ :challenge
383
+ elsif status >= 200 && status < 300
384
+ :pass
385
+ elsif status >= 300 && status < 400
386
+ :redirect
387
+ elsif [403, 429, 503].include?(status)
388
+ :blocked
389
+ else
390
+ :error
391
+ end
392
+
393
+ {
394
+ status: status,
395
+ duration_ms: duration_ms,
396
+ verdict: verdict,
397
+ challenges: challenges,
398
+ headers: headers_hash,
399
+ body_length: body_str.bytesize,
400
+ title: extract_page_title(body_str)
401
+ }
402
+ rescue StandardError => e
403
+ duration_ms = ((Time.now - start_t) * 1000).round(1) rescue 0
404
+ {
405
+ status: 0,
406
+ duration_ms: duration_ms,
407
+ verdict: :error,
408
+ challenges: [],
409
+ headers: {},
410
+ body_length: 0,
411
+ title: '',
412
+ error: e.message
413
+ }
414
+ end
415
+
416
+ def detect_waf_challenges(status_code, headers, body)
417
+ indicators = []
418
+
419
+ # Cloudflare Turnstile / Managed Challenge
420
+ if headers['cf-mitigated'] == 'challenge' ||
421
+ body.include?('cf-turnstile') ||
422
+ body.include?('challenges.cloudflare.com') ||
423
+ body.include?('Just a moment...') ||
424
+ body.include?('Attention Required! | Cloudflare') ||
425
+ body.include?('cf-challenge-running')
426
+ indicators << 'Cloudflare Managed Challenge (Turnstile/JS Challenge)'
427
+ end
428
+
429
+ # Cloudflare 1020 / Super Bot Fight Mode block
430
+ if [403, 503].include?(status_code) && (headers['cf-ray'] || headers['server'].to_s.include?('cloudflare'))
431
+ indicators << 'Cloudflare Super Bot Fight Mode (403/503 Block)'
432
+ end
433
+
434
+ # AWS WAF Block
435
+ if headers['x-amzn-waf-action'] == 'block' || body.include?('Request blocked by AWS WAF')
436
+ indicators << 'AWS WAF Automated Block'
437
+ end
438
+
439
+ # Akamai Bot Manager
440
+ if status_code == 403 && headers['server'].to_s.include?('akamaighost')
441
+ indicators << 'Akamai Bot Manager Access Denied'
442
+ end
443
+
444
+ # DataDome
445
+ if headers['x-datadome'] || body.include?('datadome.js')
446
+ indicators << 'DataDome Anti-Bot Challenge'
447
+ end
448
+
449
+ # PerimeterX / HUMAN
450
+ if headers.keys.any? { |k| k.start_with?('x-px-') } || body.include?('captcha.px-cdn.net')
451
+ indicators << 'PerimeterX (HUMAN Security) Challenge'
452
+ end
453
+
454
+ # Generic CAPTCHA
455
+ if body.include?('g-recaptcha') || body.include?('hcaptcha.com')
456
+ indicators << 'Interactive CAPTCHA Challenge'
457
+ end
458
+
459
+ indicators
460
+ end
461
+
462
+ def detect_edge_infrastructure(probe_results)
463
+ # Aggregate headers across all probes
464
+ combined_headers = {}
465
+ probe_results.each do |pr|
466
+ (pr[:headers] || {}).each { |k, v| combined_headers[k] ||= v }
467
+ end
468
+
469
+ server = combined_headers['server'].to_s.downcase
470
+ detected = []
471
+
472
+ # Cloudflare
473
+ if combined_headers['cf-ray'] || server.include?('cloudflare') || combined_headers['cf-cache-status']
474
+ detected << {
475
+ name: 'Cloudflare',
476
+ role: 'Edge CDN, DDoS & WAF',
477
+ indicators: [
478
+ combined_headers['cf-ray'] ? "cf-ray: #{combined_headers['cf-ray']}" : nil,
479
+ combined_headers['cf-cache-status'] ? "cache: #{combined_headers['cf-cache-status']}" : nil,
480
+ server.include?('cloudflare') ? 'server: cloudflare' : nil
481
+ ].compact
482
+ }
483
+ end
484
+
485
+ # Shopify Edge
486
+ if combined_headers['x-shopify-stage'] || combined_headers['x-shopid'] || combined_headers['x-shardid']
487
+ detected << {
488
+ name: 'Shopify Cloudflare Enterprise',
489
+ role: 'E-commerce Edge Platform',
490
+ indicators: ['x-shopify-stage header present']
491
+ }
492
+ end
493
+
494
+ # AWS CloudFront / WAF
495
+ if combined_headers['x-amz-cf-id'] || combined_headers['x-amzn-requestid'] || combined_headers['via'].to_s.include?('cloudfront')
496
+ detected << {
497
+ name: 'AWS CloudFront / WAF',
498
+ role: 'Cloud CDN & Security Perimeter',
499
+ indicators: [
500
+ combined_headers['x-amz-cf-id'] ? 'x-amz-cf-id present' : nil,
501
+ combined_headers['x-amzn-waf-action'] ? "WAF action: #{combined_headers['x-amzn-waf-action']}" : nil
502
+ ].compact
503
+ }
504
+ end
505
+
506
+ # Fastly
507
+ if combined_headers['x-fastly-request-id'] || combined_headers['x-served-by'].to_s.include?('cache-')
508
+ detected << {
509
+ name: 'Fastly Edge',
510
+ role: 'Edge Cloud CDN',
511
+ indicators: ['x-fastly-request-id present']
512
+ }
513
+ end
514
+
515
+ # Akamai
516
+ if server.include?('akamaighost') || combined_headers['x-akamai-transformed']
517
+ detected << {
518
+ name: 'Akamai Edge',
519
+ role: 'Enterprise Bot Management & CDN',
520
+ indicators: ['Akamai Ghost Server identified']
521
+ }
522
+ end
523
+
524
+ # Vercel
525
+ if combined_headers['x-vercel-id'] || server.include?('vercel')
526
+ detected << {
527
+ name: 'Vercel Edge Network',
528
+ role: 'Serverless Edge & Middleware',
529
+ indicators: ['x-vercel-id present']
530
+ }
531
+ end
532
+
533
+ # DataDome
534
+ if combined_headers['x-datadome']
535
+ detected << {
536
+ name: 'DataDome Bot Protection',
537
+ role: 'AI Bot Mitigation Firewall',
538
+ indicators: ['x-datadome header present']
539
+ }
540
+ end
541
+
542
+ detected << { name: 'Standard Web Origin', role: 'Direct Origin Server', indicators: ['Standard HTTP Server'] } if detected.empty?
543
+
544
+ detected
545
+ end
546
+
547
+ def analyze_blockade(probe_results, robots_data)
548
+ browser_probe = probe_results.find { |p| p[:id] == :browser }
549
+ googlebot_probe = probe_results.find { |p| p[:id] == :googlebot }
550
+ ai_probes = probe_results.reject { |p| [:browser, :googlebot].include?(p[:id]) }
551
+
552
+ browser_ok = browser_probe && [200, 301, 302].include?(browser_probe[:status])
553
+ googlebot_ok = googlebot_probe && [200, 301, 302].include?(googlebot_probe[:status])
554
+
555
+ blocked_ai_probes = ai_probes.select { |p| p[:verdict] == :blocked || p[:verdict] == :challenge }
556
+ silent_blockade = browser_ok && blocked_ai_probes.any?
557
+
558
+ # Check robots.txt disallow conflicts
559
+ robots_blocked_critical = (robots_data[:bots] || {}).select do |bot_name, info|
560
+ info[:critical] && info[:status] == :blocked
561
+ end
562
+
563
+ severity = if silent_blockade && blocked_ai_probes.size >= 2
564
+ :critical
565
+ elsif silent_blockade || robots_blocked_critical.any?
566
+ :high
567
+ elsif ai_probes.any? { |p| p[:verdict] == :challenge }
568
+ :moderate
569
+ else
570
+ :none
571
+ end
572
+
573
+ {
574
+ silent_blockade_detected: silent_blockade,
575
+ severity: severity,
576
+ browser_accessible: browser_ok,
577
+ googlebot_accessible: googlebot_ok,
578
+ blocked_probes: blocked_ai_probes.map { |p| { id: p[:id], name: p[:name], status: p[:status], verdict: p[:verdict], challenges: p[:challenges] } },
579
+ robots_critical_blocked: robots_blocked_critical.keys,
580
+ summary: if silent_blockade
581
+ "🚨 SILENT BOT BLOCKADE: Standard browsers receive HTTP 200, but AI search crawlers (#{blocked_ai_probes.map { |p| p[:name] }.join(', ')}) are challenged or blocked by edge firewall rules."
582
+ elsif robots_blocked_critical.any?
583
+ "⚠️ ROBOTS.TXT BLOCKADE: Critical AI search bots (#{robots_blocked_critical.keys.join(', ')}) are blocked from crawling in /robots.txt."
584
+ else
585
+ "✅ OPEN ACCESS: AI search crawlers pass both edge firewall inspection and robots.txt governance."
586
+ end
587
+ }
588
+ end
589
+
590
+ def calculate_score(robots_data, probe_results, blockade)
591
+ # 1. Robots score (0 - 50 pts)
592
+ critical_bots = ['GPTBot', 'ChatGPT-User', 'ClaudeBot', 'PerplexityBot']
593
+ extended_bots = ['Google-Extended', 'Applebot-Extended']
594
+
595
+ robots_score = 0
596
+ critical_bots.each do |b|
597
+ status = robots_data.dig(:bots, b, :status)
598
+ robots_score += 10 if status == :allowed
599
+ robots_score += 6 if status == :restricted
600
+ end
601
+
602
+ extended_bots.each do |b|
603
+ status = robots_data.dig(:bots, b, :status)
604
+ robots_score += 5 if status == :allowed
605
+ robots_score += 3 if status == :restricted
606
+ end
607
+
608
+ # 2. Live probe score (0 - 50 pts)
609
+ probe_score = 0
610
+ probe_results.each do |pr|
611
+ next if pr[:id] == :custom # Don't skew default scoring
612
+ case pr[:verdict]
613
+ when :pass, :redirect
614
+ probe_score += 10
615
+ when :challenge
616
+ probe_score += 3
617
+ end
618
+ end
619
+
620
+ # 3. Penalties
621
+ penalty = 0
622
+ penalty += 25 if blockade[:silent_blockade_detected]
623
+ penalty += 15 if blockade[:robots_critical_blocked].any?
624
+
625
+ raw_total = robots_score + probe_score - penalty
626
+ final_score = [[0, raw_total].max, 100].min
627
+
628
+ grade = case final_score
629
+ when 90..100 then 'A'
630
+ when 75..89 then 'B'
631
+ when 60..74 then 'C'
632
+ when 40..59 then 'D'
633
+ else 'F'
634
+ end
635
+
636
+ verdict = case grade
637
+ when 'A' then 'EXCELLENT: Fully Accessible & AI Search Ready'
638
+ when 'B' then 'GOOD: Accessible with minor training crawler restrictions'
639
+ when 'C' then 'MODERATE: Partial WAF challenges or restricted bots detected'
640
+ when 'D' then 'POOR: Key AI search assistants (ChatGPT/Claude/Perplexity) blocked'
641
+ else 'CRITICAL: Severe AI Search Blockade active'
642
+ end
643
+
644
+ {
645
+ score: final_score,
646
+ grade: grade,
647
+ verdict: verdict,
648
+ breakdown: {
649
+ robots_governance: { score: robots_score, max: 50 },
650
+ firewall_passthrough: { score: probe_score, max: 50 },
651
+ penalties: penalty
652
+ }
653
+ }
654
+ end
655
+
656
+ def generate_remediation_recipes(edge_infra, robots_data, blockade)
657
+ is_cloudflare = edge_infra.any? { |i| i[:name].include?('Cloudflare') }
658
+ is_aws = edge_infra.any? { |i| i[:name].include?('AWS') }
659
+
660
+ cf_expression = '(http.user_agent contains "GPTBot" or ' \
661
+ 'http.user_agent contains "ChatGPT-User" or ' \
662
+ 'http.user_agent contains "ClaudeBot" or ' \
663
+ 'http.user_agent contains "Claude-Web" or ' \
664
+ 'http.user_agent contains "PerplexityBot")'
665
+
666
+ cf_rule = {
667
+ name: 'Allow Legitimate AI Search Engines & User Browsing',
668
+ filter_expression: cf_expression,
669
+ action: 'Skip',
670
+ skip_features: [
671
+ 'WAF Managed Rules',
672
+ 'Super Bot Fight Mode (Managed Challenge)',
673
+ 'Rate Limiting Rules'
674
+ ],
675
+ dashboard_instructions: [
676
+ '1. Log into Cloudflare Dashboard -> Security -> WAF -> Custom Rules.',
677
+ '2. Click "Create rule" with Name: "Allow Verified AI Search Engines".',
678
+ "3. Set expression to:\n #{cf_expression}",
679
+ '4. Select Action: "Skip" and check "All remaining custom rules", "Super Bot Fight Mode", and "Managed Rules".',
680
+ '5. Save and Deploy to prevent silent HTTP 403 / Turnstile challenges on AI search bots.'
681
+ ]
682
+ }
683
+
684
+ robots_recommendation = <<~ROBOTS
685
+ # ==============================================================================
686
+ # AI Search Engine & Citability Governance (Optimized for ChatGPT, Claude, Perplexity)
687
+ # ==============================================================================
688
+
689
+ # Allow Real-Time AI Search Assistants (ChatGPT Search & Retrieval)
690
+ User-agent: GPTBot
691
+ Allow: /
692
+
693
+ User-agent: ChatGPT-User
694
+ Allow: /
695
+
696
+ # Allow Anthropic Claude Search & Retrieval
697
+ User-agent: ClaudeBot
698
+ Allow: /
699
+
700
+ User-agent: Claude-Web
701
+ Allow: /
702
+
703
+ # Allow Perplexity Conversational Search
704
+ User-agent: PerplexityBot
705
+ Allow: /
706
+
707
+ # (Optional) Restrict model training if proprietary content protection is needed:
708
+ # User-agent: Google-Extended
709
+ # Disallow: /
710
+ # User-agent: CCBot
711
+ # Disallow: /
712
+ ROBOTS
713
+
714
+ {
715
+ cloudflare_waf_rule: is_cloudflare || is_aws ? cf_rule : nil,
716
+ recommended_robots_txt: robots_recommendation,
717
+ action_items: [
718
+ blockade[:silent_blockade_detected] ? 'CRITICAL: Create WAF bypass rule for GPTBot and ClaudeBot to stop 403 challenges.' : nil,
719
+ blockade[:robots_critical_blocked].any? ? "ROBOTS: Remove Disallow: / for #{blockade[:robots_critical_blocked].join(', ')}." : nil,
720
+ 'MONITOR: Periodically run `gsc firewall <url>` to verify ongoing AI crawler accessibility.'
721
+ ].compact
722
+ }
723
+ end
724
+
725
+ def extract_page_title(html_str)
726
+ if html_str =~ /<title[^>]*>(.*?)<\/title>/im
727
+ $1.to_s.strip.gsub(/\s+/, ' ')
728
+ else
729
+ ''
730
+ end
731
+ end
732
+ end
733
+ end