gsc-cli 2.0.2 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. checksums.yaml +4 -4
  2. data/AUTH.md +205 -0
  3. data/FUNDING.md +120 -0
  4. data/README.md +463 -299
  5. data/bin/gsc +29158 -4921
  6. data/dist/gsc +29158 -4921
  7. data/lib/gsc/aio_hunter.rb +343 -0
  8. data/lib/gsc/answer_synthesizer.rb +157 -0
  9. data/lib/gsc/api.rb +53 -1
  10. data/lib/gsc/auth.rb +26 -0
  11. data/lib/gsc/backlinks_manager.rb +96 -0
  12. data/lib/gsc/brand_segmenter.rb +140 -0
  13. data/lib/gsc/cache_manager.rb +806 -0
  14. data/lib/gsc/cannibalization_analyzer.rb +141 -0
  15. data/lib/gsc/canonical_chains.rb +367 -0
  16. data/lib/gsc/citation_simulator.rb +339 -0
  17. data/lib/gsc/cli/aio_hunter.rb +154 -0
  18. data/lib/gsc/cli/analytics.rb +788 -0
  19. data/lib/gsc/cli/audit.rb +1976 -0
  20. data/lib/gsc/cli/base.rb +384 -0
  21. data/lib/gsc/cli/cache.rb +266 -0
  22. data/lib/gsc/cli/canonical.rb +223 -0
  23. data/lib/gsc/cli/citation_simulator.rb +152 -0
  24. data/lib/gsc/cli/dashboard.rb +354 -0
  25. data/lib/gsc/cli/doctor.rb +129 -0
  26. data/lib/gsc/cli/eeat.rb +125 -0
  27. data/lib/gsc/cli/ga4.rb +852 -0
  28. data/lib/gsc/cli/growth.rb +650 -0
  29. data/lib/gsc/cli/hreflang.rb +164 -0
  30. data/lib/gsc/cli/image_seo.rb +162 -0
  31. data/lib/gsc/cli/indexing.rb +458 -0
  32. data/lib/gsc/cli/intent_shift.rb +125 -0
  33. data/lib/gsc/cli/keyword_value.rb +134 -0
  34. data/lib/gsc/cli/keywords.rb +795 -0
  35. data/lib/gsc/cli/landing_roi.rb +308 -0
  36. data/lib/gsc/cli/low_ctr.rb +213 -0
  37. data/lib/gsc/cli/mobile_parity.rb +150 -0
  38. data/lib/gsc/cli/report.rb +100 -0
  39. data/lib/gsc/cli/rich_results.rb +172 -0
  40. data/lib/gsc/cli/schema_generate.rb +149 -0
  41. data/lib/gsc/cli/seasonal.rb +232 -0
  42. data/lib/gsc/cli/security.rb +153 -0
  43. data/lib/gsc/cli/setup.rb +1291 -0
  44. data/lib/gsc/cli/sitemap_tree.rb +143 -0
  45. data/lib/gsc/cli/skill_pack.rb +62 -0
  46. data/lib/gsc/cli/soft_404.rb +199 -0
  47. data/lib/gsc/cli/sparkline.rb +227 -0
  48. data/lib/gsc/cli/watchdog.rb +150 -0
  49. data/lib/gsc/cli/zombie_purger.rb +208 -0
  50. data/lib/gsc/cli.rb +706 -5111
  51. data/lib/gsc/cli_advanced.rb +1513 -0
  52. data/lib/gsc/client.rb +17 -2
  53. data/lib/gsc/color.rb +16 -1
  54. data/lib/gsc/command_registry.rb +47 -9
  55. data/lib/gsc/config.rb +11 -2
  56. data/lib/gsc/content_gap.rb +112 -0
  57. data/lib/gsc/ctr_curve.rb +115 -0
  58. data/lib/gsc/decay_predictor.rb +322 -0
  59. data/lib/gsc/doctor.rb +434 -0
  60. data/lib/gsc/eeat_auditor.rb +428 -0
  61. data/lib/gsc/entity_auditor.rb +229 -0
  62. data/lib/gsc/firewall_scanner.rb +733 -0
  63. data/lib/gsc/geo_auditor.rb +368 -0
  64. data/lib/gsc/google_suggest.rb +109 -0
  65. data/lib/gsc/google_trends.rb +8 -1
  66. data/lib/gsc/heading_validator.rb +283 -0
  67. data/lib/gsc/hreflang_validator.rb +412 -0
  68. data/lib/gsc/image_seo.rb +286 -0
  69. data/lib/gsc/indexing_queue.rb +179 -0
  70. data/lib/gsc/indexnow.rb +93 -0
  71. data/lib/gsc/intent_shift.rb +188 -0
  72. data/lib/gsc/internal_links.rb +249 -0
  73. data/lib/gsc/keyword_value.rb +191 -0
  74. data/lib/gsc/landing_roi.rb +195 -0
  75. data/lib/gsc/llms_generator.rb +425 -0
  76. data/lib/gsc/low_ctr_rewriter.rb +408 -0
  77. data/lib/gsc/mobile_parity.rb +222 -0
  78. data/lib/gsc/network_tracer.rb +93 -0
  79. data/lib/gsc/open_page_rank.rb +72 -0
  80. data/lib/gsc/page_analyzer.rb +47 -7
  81. data/lib/gsc/page_comparator.rb +108 -0
  82. data/lib/gsc/page_speed.rb +110 -0
  83. data/lib/gsc/prompts.rb +38 -29
  84. data/lib/gsc/questions_harvester.rb +178 -0
  85. data/lib/gsc/report_generator.rb +461 -0
  86. data/lib/gsc/rich_results.rb +388 -0
  87. data/lib/gsc/robots_checker.rb +114 -0
  88. data/lib/gsc/schema_generator.rb +788 -0
  89. data/lib/gsc/schema_validator.rb +120 -0
  90. data/lib/gsc/seasonal_predictor.rb +381 -0
  91. data/lib/gsc/security_scanner.rb +496 -0
  92. data/lib/gsc/serp_feature_detector.rb +359 -0
  93. data/lib/gsc/serp_preview.rb +152 -0
  94. data/lib/gsc/site_crawler.rb +113 -21
  95. data/lib/gsc/sitemap_loader.rb +15 -4
  96. data/lib/gsc/sitemap_tree.rb +301 -0
  97. data/lib/gsc/skill_pack.rb +195 -0
  98. data/lib/gsc/soft_404_analyzer.rb +385 -0
  99. data/lib/gsc/sparkline.rb +171 -0
  100. data/lib/gsc/speed_correlator.rb +416 -0
  101. data/lib/gsc/striking_playbook.rb +190 -0
  102. data/lib/gsc/title_optimizer.rb +420 -0
  103. data/lib/gsc/vault.rb +260 -0
  104. data/lib/gsc/version.rb +1 -1
  105. data/lib/gsc/watchdog.rb +235 -0
  106. data/lib/gsc/zombie_purger.rb +366 -0
  107. data/lib/gsc.rb +144 -0
  108. metadata +91 -2
@@ -0,0 +1,388 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'net/http'
5
+ require 'uri'
6
+ require 'json'
7
+
8
+ module GSC
9
+ class RichResults
10
+ attr_reader :options, :url, :content, :schemas
11
+
12
+ def initialize(options = {})
13
+ @options = options
14
+ end
15
+
16
+ def self.audit(target, options = {})
17
+ new(options).audit(target)
18
+ end
19
+
20
+ def audit(target)
21
+ @url, @content = load_content(target)
22
+ @schemas = extract_schemas(@content)
23
+ results = @schemas.map.with_index { |s, idx| validate_schema(s, idx) }
24
+ synthesize_report(results)
25
+ end
26
+
27
+ private
28
+
29
+ def load_content(target)
30
+ target_str = target.to_s.strip
31
+ if target_str.match?(%r{^https?://})
32
+ uri = URI.parse(target_str)
33
+ req = Net::HTTP::Get.new(uri)
34
+ req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) gsc-cli/2.1'
35
+ req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
36
+
37
+ res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 8, read_timeout: 10) do |http|
38
+ http.request(req)
39
+ end
40
+ [target_str, res.body.to_s.dup.force_encoding('UTF-8').scrub]
41
+ elsif File.exist?(target_str)
42
+ [target_str, File.read(target_str, encoding: 'UTF-8')]
43
+ else
44
+ [target_str.start_with?('http') ? target_str : 'direct-input', target_str]
45
+ end
46
+ rescue StandardError => e
47
+ [target.to_s, "<!-- fetch error: #{e.message} -->"]
48
+ end
49
+
50
+ def extract_schemas(html_or_json)
51
+ schemas = []
52
+
53
+ # Case 1: Raw JSON input
54
+ trimmed = html_or_json.strip
55
+ if trimmed.start_with?('{', '[')
56
+ begin
57
+ parsed = JSON.parse(trimmed)
58
+ return flatten_schemas(parsed)
59
+ rescue JSON::ParserError
60
+ # Fallback to HTML regex extraction
61
+ end
62
+ end
63
+
64
+ # Case 2: Extract from <script type="application/ld+json">
65
+ html_or_json.scan(%r{<script[^>]*type=["']application/ld\+json["'][^>]*>(.*?)</script>}im) do |match|
66
+ json_str = match[0].to_s.strip
67
+ next if json_str.empty?
68
+
69
+ begin
70
+ parsed = JSON.parse(json_str)
71
+ schemas.concat(flatten_schemas(parsed))
72
+ rescue JSON::ParserError => e
73
+ schemas << {
74
+ '@type' => 'CorruptedJSONLD',
75
+ '_parse_error' => e.message,
76
+ '_raw' => json_str
77
+ }
78
+ end
79
+ end
80
+
81
+ schemas
82
+ end
83
+
84
+ def flatten_schemas(data)
85
+ case data
86
+ when Array
87
+ data.flat_map { |item| flatten_schemas(item) }
88
+ when Hash
89
+ if data['@graph'].is_a?(Array)
90
+ data['@graph'].flat_map { |item| flatten_schemas(item) }
91
+ else
92
+ [data]
93
+ end
94
+ else
95
+ []
96
+ end
97
+ end
98
+
99
+ def validate_schema(schema, idx)
100
+ type = schema['@type'] || 'Unknown'
101
+ errors = []
102
+ warnings = []
103
+ feature_name = nil
104
+
105
+ if schema['_parse_error']
106
+ return {
107
+ index: idx + 1,
108
+ type: 'SyntaxError',
109
+ feature: 'Malformed JSON-LD',
110
+ eligible: false,
111
+ errors: ["JSON Syntax Error: #{schema['_parse_error']}"],
112
+ warnings: [],
113
+ raw: schema['_raw'],
114
+ patch: nil
115
+ }
116
+ end
117
+
118
+ case type
119
+ when 'Product'
120
+ feature_name = 'Merchant Product Rich Card'
121
+ errors << 'Missing "name"' unless schema['name']
122
+ errors << 'Missing "image"' unless schema['image']
123
+
124
+ has_pricing = schema['offers'] || schema['review'] || schema['aggregateRating']
125
+ errors << 'Must provide at least one of "offers", "review", or "aggregateRating"' unless has_pricing
126
+
127
+ if schema['offers']
128
+ offers = schema['offers'].is_a?(Array) ? schema['offers'].first : schema['offers']
129
+ errors << 'Missing "offers.price" or "offers.lowPrice"' unless offers['price'] || offers['lowPrice']
130
+ errors << 'Missing "offers.priceCurrency"' unless offers['priceCurrency']
131
+ warnings << 'Missing "offers.availability" (e.g. InStock)' unless offers['availability']
132
+ warnings << 'Missing "offers.priceValidUntil"' unless offers['priceValidUntil']
133
+ end
134
+
135
+ if schema['aggregateRating']
136
+ ar = schema['aggregateRating']
137
+ errors << 'Missing "aggregateRating.ratingValue"' unless ar['ratingValue']
138
+ errors << 'Missing "aggregateRating.ratingCount" or "aggregateRating.reviewCount"' unless ar['ratingCount'] || ar['reviewCount']
139
+ end
140
+
141
+ when 'FAQPage'
142
+ feature_name = 'Interactive FAQ Accordion'
143
+ entities = schema['mainEntity']
144
+ if !entities.is_a?(Array) || entities.empty?
145
+ errors << 'FAQPage must contain a non-empty "mainEntity" array of Question objects'
146
+ else
147
+ entities.each_with_index do |q, q_idx|
148
+ q_num = q_idx + 1
149
+ errors << "Question ##{q_num} missing 'name'" unless q['name']
150
+ if !q['acceptedAnswer'] || !q['acceptedAnswer']['text']
151
+ errors << "Question ##{q_num} missing 'acceptedAnswer.text'"
152
+ end
153
+ end
154
+ end
155
+
156
+ when 'HowTo'
157
+ feature_name = 'Step-by-Step Guided HowTo Carousel'
158
+ errors << 'Missing "name"' unless schema['name']
159
+ steps = schema['step']
160
+ if !steps.is_a?(Array) || steps.empty?
161
+ errors << 'HowTo must contain a non-empty "step" array'
162
+ else
163
+ steps.each_with_index do |s, s_idx|
164
+ s_num = s_idx + 1
165
+ errors << "Step ##{s_num} missing 'text' or 'name'" unless s['text'] || s['name']
166
+ end
167
+ end
168
+ warnings << 'Missing "totalTime"' unless schema['totalTime']
169
+ warnings << 'Missing "image"' unless schema['image']
170
+
171
+ when 'Article', 'NewsArticle', 'BlogPosting'
172
+ feature_name = 'Top Stories Carousel & Author Card'
173
+ errors << 'Missing "headline"' unless schema['headline']
174
+ errors << 'Missing "image"' unless schema['image']
175
+ errors << 'Missing "datePublished"' unless schema['datePublished']
176
+ errors << 'Missing "author"' unless schema['author']
177
+
178
+ if schema['author']
179
+ authors = schema['author'].is_a?(Array) ? schema['author'] : [schema['author']]
180
+ authors.each do |a|
181
+ if a.is_a?(Hash)
182
+ warnings << 'Author missing "url" (E-E-A-T profile link)' unless a['url']
183
+ end
184
+ end
185
+ end
186
+ warnings << 'Missing "dateModified"' unless schema['dateModified']
187
+ warnings << 'Missing "publisher" with logo' unless schema['publisher']
188
+
189
+ when 'SoftwareApplication', 'WebApplication', 'MobileApplication'
190
+ feature_name = 'Software Application Rich Snippet'
191
+ errors << 'Missing "name"' unless schema['name']
192
+ errors << 'Missing "operatingSystem"' unless schema['operatingSystem']
193
+ errors << 'Missing "applicationCategory"' unless schema['applicationCategory']
194
+ errors << 'Missing "offers"' unless schema['offers']
195
+ warnings << 'Missing "aggregateRating"' unless schema['aggregateRating']
196
+
197
+ when 'LocalBusiness', 'Store', 'Restaurant', 'MedicalBusiness'
198
+ feature_name = 'Local 3-Pack & Knowledge Graph Card'
199
+ errors << 'Missing "name"' unless schema['name']
200
+ errors << 'Missing "address"' unless schema['address']
201
+ errors << 'Missing "telephone"' unless schema['telephone']
202
+ warnings << 'Missing "openingHoursSpecification"' unless schema['openingHoursSpecification'] || schema['openingHours']
203
+ warnings << 'Missing "geo" coordinates (latitude/longitude)' unless schema['geo']
204
+
205
+ when 'BreadcrumbList'
206
+ feature_name = 'Interactive Breadcrumb Path'
207
+ items = schema['itemListElement']
208
+ if !items.is_a?(Array) || items.empty?
209
+ errors << 'BreadcrumbList must contain "itemListElement" array'
210
+ else
211
+ items.each_with_index do |it, i_idx|
212
+ errors << "Breadcrumb ##{i_idx + 1} missing 'position'" unless it['position']
213
+ errors << "Breadcrumb ##{i_idx + 1} missing 'name'" unless it['name']
214
+ errors << "Breadcrumb ##{i_idx + 1} missing 'item'" unless it['item']
215
+ end
216
+ end
217
+
218
+ when 'Recipe'
219
+ feature_name = 'Recipe Rich Card with Cooking Specs'
220
+ errors << 'Missing "name"' unless schema['name']
221
+ errors << 'Missing "image"' unless schema['image']
222
+ errors << 'Missing "recipeIngredient"' unless schema['recipeIngredient']
223
+ errors << 'Missing "recipeInstructions"' unless schema['recipeInstructions']
224
+ warnings << 'Missing "cookTime"' unless schema['cookTime']
225
+ warnings << 'Missing "aggregateRating"' unless schema['aggregateRating']
226
+
227
+ when 'VideoObject'
228
+ feature_name = 'Video SERP Thumbnail & Key Moments'
229
+ errors << 'Missing "name"' unless schema['name']
230
+ errors << 'Missing "description"' unless schema['description']
231
+ errors << 'Missing "thumbnailUrl"' unless schema['thumbnailUrl']
232
+ errors << 'Missing "uploadDate"' unless schema['uploadDate']
233
+
234
+ when 'Course'
235
+ feature_name = 'Course Listing Rich Carousel'
236
+ errors << 'Missing "name"' unless schema['name']
237
+ errors << 'Missing "description"' unless schema['description']
238
+ errors << 'Missing "provider"' unless schema['provider']
239
+
240
+ when 'JobPosting'
241
+ feature_name = 'Google for Jobs Interactive Listing'
242
+ errors << 'Missing "title"' unless schema['title']
243
+ errors << 'Missing "description"' unless schema['description']
244
+ errors << 'Missing "datePosted"' unless schema['datePosted']
245
+ errors << 'Missing "hiringOrganization"' unless schema['hiringOrganization']
246
+
247
+ when 'Event'
248
+ feature_name = 'Event SERP Listing & Tickets'
249
+ errors << 'Missing "name"' unless schema['name']
250
+ errors << 'Missing "startDate"' unless schema['startDate']
251
+ errors << 'Missing "location"' unless schema['location']
252
+
253
+ when 'Review', 'AggregateRating'
254
+ feature_name = 'Star Rating Snippet'
255
+ errors << 'Missing "ratingValue"' unless schema['ratingValue']
256
+ warnings << 'Missing "bestRating" (defaults to 5)' unless schema['bestRating']
257
+
258
+ else
259
+ feature_name = "Schema.org #{type}"
260
+ warnings << "Generic schema type '#{type}' does not trigger a dedicated Google Rich Result"
261
+ end
262
+
263
+ eligible = errors.empty? && !feature_name.start_with?('Schema.org')
264
+
265
+ patch = generate_patch(schema, errors, warnings)
266
+
267
+ {
268
+ index: idx + 1,
269
+ type: type,
270
+ feature: feature_name,
271
+ eligible: eligible,
272
+ errors: errors,
273
+ warnings: warnings,
274
+ raw: schema,
275
+ patch: patch
276
+ }
277
+ end
278
+
279
+ def generate_patch(schema, errors, warnings)
280
+ return nil if errors.empty? && warnings.empty?
281
+
282
+ patched = schema.dup
283
+ type = patched['@type']
284
+
285
+ case type
286
+ when 'Product'
287
+ product_url = @url.to_s.start_with?('http') ? @url : nil
288
+ patched['image'] ||= [product_url ? "#{product_url.sub(%r{/$}, '')}/product.jpg" : 'https://schema.org/ProductImage']
289
+ if !patched['offers'] && !patched['aggregateRating']
290
+ patched['offers'] = {
291
+ '@type' => 'Offer',
292
+ 'price' => '49.99',
293
+ 'priceCurrency' => 'USD',
294
+ 'priceValidUntil' => (Time.now + (365 * 86400)).strftime('%Y-%m-%d'),
295
+ 'availability' => 'https://schema.org/InStock',
296
+ 'url' => product_url || ''
297
+ }
298
+ end
299
+ if patched['offers'].is_a?(Hash)
300
+ patched['offers']['priceValidUntil'] ||= (Time.now + (365 * 86400)).strftime('%Y-%m-%d')
301
+ patched['offers']['availability'] ||= 'https://schema.org/InStock'
302
+ end
303
+
304
+ when 'Article', 'BlogPosting'
305
+ base_url = @url.to_s.start_with?('http') ? @url : nil
306
+ patched['datePublished'] ||= Time.now.strftime('%Y-%m-%dT%H:%M:%S%:z')
307
+ patched['dateModified'] ||= Time.now.strftime('%Y-%m-%dT%H:%M:%S%:z')
308
+ patched['image'] ||= [base_url ? "#{base_url.sub(%r{/$}, '')}/featured.jpg" : 'https://schema.org/ArticleImage']
309
+ if !patched['author']
310
+ author_url = base_url ? "#{base_url.sub(%r{/$}, '')}/author" : ''
311
+ patched['author'] = {
312
+ '@type' => 'Person',
313
+ 'name' => 'Author',
314
+ 'url' => author_url
315
+ }
316
+ end
317
+
318
+ when 'FAQPage'
319
+ if !patched['mainEntity'].is_a?(Array) || patched['mainEntity'].empty?
320
+ patched['mainEntity'] = [
321
+ {
322
+ '@type' => 'Question',
323
+ 'name' => 'Question?',
324
+ 'acceptedAnswer' => {
325
+ '@type' => 'Answer',
326
+ 'text' => 'Answer.'
327
+ }
328
+ }
329
+ ]
330
+ end
331
+ end
332
+
333
+ patched
334
+ end
335
+
336
+ def synthesize_report(results)
337
+ total = results.size
338
+ eligible_count = results.count { |r| r[:eligible] }
339
+ total_errors = results.sum { |r| r[:errors].size }
340
+ total_warnings = results.sum { |r| r[:warnings].size }
341
+
342
+ # Duplicate detection across same entity types
343
+ types = results.map { |r| r[:type] }
344
+ duplicates = types.select { |t| types.count(t) > 1 }.uniq
345
+ if duplicates.any?
346
+ duplicates.each do |dup_type|
347
+ results.each do |r|
348
+ if r[:type] == dup_type
349
+ r[:warnings] << "Duplicate '#{dup_type}' schema detected on page. Google advises consolidating duplicate entities into a single JSON-LD block."
350
+ end
351
+ end
352
+ end
353
+ total_warnings = results.sum { |r| r[:warnings].size }
354
+ end
355
+
356
+ # Score calculation
357
+ score = if total == 0
358
+ 0
359
+ else
360
+ raw = (eligible_count.to_f / total) * 70 + [30 - (total_errors * 10) - (total_warnings * 2), 0].max
361
+ [[raw.round, 100].min, 0].max
362
+ end
363
+
364
+ grade = case score
365
+ when 95..100 then 'A+'
366
+ when 85..94 then 'A'
367
+ when 70..84 then 'B'
368
+ when 50..69 then 'C'
369
+ else 'F'
370
+ end
371
+
372
+ eligible_features = results.select { |r| r[:eligible] }.map { |r| r[:feature] }.uniq
373
+
374
+ {
375
+ url: @url,
376
+ total_schemas_detected: total,
377
+ eligible_for_rich_results: eligible_count > 0,
378
+ eligible_features: eligible_features,
379
+ score: score,
380
+ grade: grade,
381
+ critical_errors_count: total_errors,
382
+ warnings_count: total_warnings,
383
+ duplicate_types: duplicates,
384
+ schemas: results
385
+ }
386
+ end
387
+ end
388
+ end
@@ -0,0 +1,114 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'net/http'
5
+ require 'uri'
6
+
7
+ module GSC
8
+ class RobotsChecker
9
+ attr_reader :base_url, :robots_content
10
+
11
+ def initialize(target_url)
12
+ @target_url = target_url.to_s.strip
13
+ @target_url = "https://#{@target_url}" unless @target_url =~ %r{^https?://}
14
+ @uri = URI.parse(@target_url)
15
+ @robots_url = if (@uri.port == 80 && @uri.scheme == 'http') || (@uri.port == 443 && @uri.scheme == 'https')
16
+ "#{@uri.scheme}://#{@uri.host}/robots.txt"
17
+ else
18
+ "#{@uri.scheme}://#{@uri.host}:#{@uri.port}/robots.txt"
19
+ end
20
+ end
21
+
22
+ def fetch_robots_txt(url = @robots_url, limit = 3)
23
+ return '' if limit <= 0
24
+ uri = URI.parse(url) rescue nil
25
+ return '' unless uri && uri.host
26
+
27
+ http = Net::HTTP.new(uri.host, uri.port)
28
+ http.use_ssl = (uri.scheme == 'https')
29
+ http.open_timeout = 4
30
+ http.read_timeout = 6
31
+ req = Net::HTTP::Get.new(uri.request_uri.empty? ? '/robots.txt' : uri.request_uri)
32
+ req['User-Agent'] = 'Mozilla/5.0 (compatible; GSC-SEO-Auditor/2.2)'
33
+
34
+ res = http.request(req)
35
+ if res.is_a?(Net::HTTPRedirection) && res['location']
36
+ new_url = URI.join(url, res['location']).to_s rescue nil
37
+ return new_url ? fetch_robots_txt(new_url, limit - 1) : ''
38
+ end
39
+
40
+ return '' unless res.code == '200'
41
+ res.body.to_s.force_encoding('UTF-8').scrub
42
+ rescue StandardError
43
+ ''
44
+ end
45
+
46
+ def check(path_to_test = nil, user_agent = 'googlebot')
47
+ @robots_content ||= fetch_robots_txt
48
+ path = path_to_test || @uri.path
49
+ path = '/' if path.empty?
50
+
51
+ ua = user_agent.to_s.downcase
52
+ rules = parse_rules(ua)
53
+
54
+ allowed = true
55
+ matched_rule = nil
56
+
57
+ rules.each do |rule|
58
+ pattern = rule[:path]
59
+ regex = Regexp.new('^' + Regexp.escape(pattern).gsub('\*', '.*'))
60
+ if path =~ regex
61
+ allowed = (rule[:type] == :allow)
62
+ matched_rule = rule
63
+ end
64
+ end
65
+
66
+ {
67
+ robots_url: @robots_url,
68
+ user_agent: user_agent,
69
+ tested_path: path,
70
+ allowed: allowed,
71
+ matched_rule: matched_rule,
72
+ has_robots_txt: !@robots_content.empty?
73
+ }
74
+ end
75
+
76
+ private
77
+
78
+ def parse_rules(target_ua)
79
+ groups = Hash.new { |h, k| h[k] = [] }
80
+ current_uas = []
81
+ in_directives = false
82
+
83
+ @robots_content.each_line do |line|
84
+ line = line.strip.sub(/#.*$/, '')
85
+ next if line.empty?
86
+
87
+ if line =~ /^User-agent:\s*(.+)$/i
88
+ if in_directives
89
+ current_uas = []
90
+ in_directives = false
91
+ end
92
+ current_uas << $1.strip.downcase
93
+ elsif line =~ /^Disallow:\s*(.*)$/i
94
+ in_directives = true
95
+ val = $1.strip
96
+ current_uas.each { |ua| groups[ua] << { type: :disallow, path: val } unless val.empty? }
97
+ elsif line =~ /^Allow:\s*(.*)$/i
98
+ in_directives = true
99
+ val = $1.strip
100
+ current_uas.each { |ua| groups[ua] << { type: :allow, path: val } unless val.empty? }
101
+ end
102
+ end
103
+
104
+ # RFC 9309: Specific user-agent match takes absolute precedence over wildcard '*'
105
+ if groups.key?(target_ua) && groups[target_ua].any?
106
+ groups[target_ua]
107
+ elsif groups.key?('*')
108
+ groups['*']
109
+ else
110
+ []
111
+ end
112
+ end
113
+ end
114
+ end