gsc-cli 2.1.0 → 2.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. checksums.yaml +4 -4
  2. data/AUTH.md +4 -1
  3. data/README.md +448 -408
  4. data/bin/gsc +28067 -5661
  5. data/dist/gsc +29121 -5046
  6. data/lib/gsc/aio_hunter.rb +343 -0
  7. data/lib/gsc/answer_synthesizer.rb +157 -0
  8. data/lib/gsc/api.rb +53 -1
  9. data/lib/gsc/auth.rb +26 -0
  10. data/lib/gsc/brand_segmenter.rb +140 -0
  11. data/lib/gsc/cache_manager.rb +806 -0
  12. data/lib/gsc/cannibalization_analyzer.rb +141 -0
  13. data/lib/gsc/canonical_chains.rb +367 -0
  14. data/lib/gsc/citation_simulator.rb +339 -0
  15. data/lib/gsc/cli/aio_hunter.rb +154 -0
  16. data/lib/gsc/cli/analytics.rb +788 -0
  17. data/lib/gsc/cli/audit.rb +1976 -0
  18. data/lib/gsc/cli/base.rb +384 -0
  19. data/lib/gsc/cli/cache.rb +266 -0
  20. data/lib/gsc/cli/canonical.rb +223 -0
  21. data/lib/gsc/cli/citation_simulator.rb +152 -0
  22. data/lib/gsc/cli/dashboard.rb +354 -0
  23. data/lib/gsc/cli/doctor.rb +129 -0
  24. data/lib/gsc/cli/eeat.rb +125 -0
  25. data/lib/gsc/cli/ga4.rb +852 -0
  26. data/lib/gsc/cli/growth.rb +650 -0
  27. data/lib/gsc/cli/hreflang.rb +164 -0
  28. data/lib/gsc/cli/image_seo.rb +162 -0
  29. data/lib/gsc/cli/indexing.rb +458 -0
  30. data/lib/gsc/cli/intent_shift.rb +125 -0
  31. data/lib/gsc/cli/keyword_value.rb +134 -0
  32. data/lib/gsc/cli/keywords.rb +795 -0
  33. data/lib/gsc/cli/landing_roi.rb +308 -0
  34. data/lib/gsc/cli/low_ctr.rb +213 -0
  35. data/lib/gsc/cli/mobile_parity.rb +150 -0
  36. data/lib/gsc/cli/report.rb +100 -0
  37. data/lib/gsc/cli/rich_results.rb +172 -0
  38. data/lib/gsc/cli/schema_generate.rb +149 -0
  39. data/lib/gsc/cli/seasonal.rb +232 -0
  40. data/lib/gsc/cli/security.rb +153 -0
  41. data/lib/gsc/cli/setup.rb +1291 -0
  42. data/lib/gsc/cli/sitemap_tree.rb +143 -0
  43. data/lib/gsc/cli/skill_pack.rb +62 -0
  44. data/lib/gsc/cli/soft_404.rb +199 -0
  45. data/lib/gsc/cli/sparkline.rb +227 -0
  46. data/lib/gsc/cli/watchdog.rb +150 -0
  47. data/lib/gsc/cli/zombie_purger.rb +208 -0
  48. data/lib/gsc/cli.rb +707 -5265
  49. data/lib/gsc/cli_advanced.rb +987 -44
  50. data/lib/gsc/client.rb +17 -2
  51. data/lib/gsc/color.rb +16 -1
  52. data/lib/gsc/command_registry.rb +47 -9
  53. data/lib/gsc/config.rb +2 -2
  54. data/lib/gsc/ctr_curve.rb +115 -0
  55. data/lib/gsc/decay_predictor.rb +322 -0
  56. data/lib/gsc/doctor.rb +434 -0
  57. data/lib/gsc/eeat_auditor.rb +428 -0
  58. data/lib/gsc/entity_auditor.rb +229 -0
  59. data/lib/gsc/firewall_scanner.rb +733 -0
  60. data/lib/gsc/geo_auditor.rb +368 -0
  61. data/lib/gsc/google_trends.rb +8 -1
  62. data/lib/gsc/heading_validator.rb +283 -0
  63. data/lib/gsc/hreflang_validator.rb +412 -0
  64. data/lib/gsc/image_seo.rb +286 -0
  65. data/lib/gsc/indexing_queue.rb +179 -0
  66. data/lib/gsc/indexnow.rb +93 -0
  67. data/lib/gsc/intent_shift.rb +188 -0
  68. data/lib/gsc/internal_links.rb +153 -36
  69. data/lib/gsc/keyword_value.rb +191 -0
  70. data/lib/gsc/landing_roi.rb +195 -0
  71. data/lib/gsc/llms_generator.rb +343 -22
  72. data/lib/gsc/low_ctr_rewriter.rb +408 -0
  73. data/lib/gsc/mobile_parity.rb +222 -0
  74. data/lib/gsc/network_tracer.rb +8 -1
  75. data/lib/gsc/page_analyzer.rb +47 -7
  76. data/lib/gsc/prompts.rb +38 -29
  77. data/lib/gsc/questions_harvester.rb +178 -0
  78. data/lib/gsc/report_generator.rb +461 -0
  79. data/lib/gsc/rich_results.rb +388 -0
  80. data/lib/gsc/robots_checker.rb +46 -15
  81. data/lib/gsc/schema_generator.rb +788 -0
  82. data/lib/gsc/schema_validator.rb +36 -38
  83. data/lib/gsc/seasonal_predictor.rb +381 -0
  84. data/lib/gsc/security_scanner.rb +496 -0
  85. data/lib/gsc/serp_feature_detector.rb +359 -0
  86. data/lib/gsc/serp_preview.rb +108 -22
  87. data/lib/gsc/site_crawler.rb +113 -21
  88. data/lib/gsc/sitemap_loader.rb +15 -4
  89. data/lib/gsc/sitemap_tree.rb +301 -0
  90. data/lib/gsc/skill_pack.rb +195 -0
  91. data/lib/gsc/soft_404_analyzer.rb +385 -0
  92. data/lib/gsc/sparkline.rb +171 -0
  93. data/lib/gsc/speed_correlator.rb +416 -0
  94. data/lib/gsc/striking_playbook.rb +190 -0
  95. data/lib/gsc/title_optimizer.rb +420 -0
  96. data/lib/gsc/vault.rb +260 -0
  97. data/lib/gsc/version.rb +1 -1
  98. data/lib/gsc/watchdog.rb +235 -0
  99. data/lib/gsc/zombie_purger.rb +366 -0
  100. data/lib/gsc.rb +118 -0
  101. metadata +75 -1
@@ -0,0 +1,388 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'net/http'
5
+ require 'uri'
6
+ require 'json'
7
+
8
+ module GSC
9
+ class RichResults
10
+ attr_reader :options, :url, :content, :schemas
11
+
12
+ def initialize(options = {})
13
+ @options = options
14
+ end
15
+
16
+ def self.audit(target, options = {})
17
+ new(options).audit(target)
18
+ end
19
+
20
+ def audit(target)
21
+ @url, @content = load_content(target)
22
+ @schemas = extract_schemas(@content)
23
+ results = @schemas.map.with_index { |s, idx| validate_schema(s, idx) }
24
+ synthesize_report(results)
25
+ end
26
+
27
+ private
28
+
29
+ def load_content(target)
30
+ target_str = target.to_s.strip
31
+ if target_str.match?(%r{^https?://})
32
+ uri = URI.parse(target_str)
33
+ req = Net::HTTP::Get.new(uri)
34
+ req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) gsc-cli/2.1'
35
+ req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
36
+
37
+ res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 8, read_timeout: 10) do |http|
38
+ http.request(req)
39
+ end
40
+ [target_str, res.body.to_s.dup.force_encoding('UTF-8').scrub]
41
+ elsif File.exist?(target_str)
42
+ [target_str, File.read(target_str, encoding: 'UTF-8')]
43
+ else
44
+ [target_str.start_with?('http') ? target_str : 'direct-input', target_str]
45
+ end
46
+ rescue StandardError => e
47
+ [target.to_s, "<!-- fetch error: #{e.message} -->"]
48
+ end
49
+
50
+ def extract_schemas(html_or_json)
51
+ schemas = []
52
+
53
+ # Case 1: Raw JSON input
54
+ trimmed = html_or_json.strip
55
+ if trimmed.start_with?('{', '[')
56
+ begin
57
+ parsed = JSON.parse(trimmed)
58
+ return flatten_schemas(parsed)
59
+ rescue JSON::ParserError
60
+ # Fallback to HTML regex extraction
61
+ end
62
+ end
63
+
64
+ # Case 2: Extract from <script type="application/ld+json">
65
+ html_or_json.scan(%r{<script[^>]*type=["']application/ld\+json["'][^>]*>(.*?)</script>}im) do |match|
66
+ json_str = match[0].to_s.strip
67
+ next if json_str.empty?
68
+
69
+ begin
70
+ parsed = JSON.parse(json_str)
71
+ schemas.concat(flatten_schemas(parsed))
72
+ rescue JSON::ParserError => e
73
+ schemas << {
74
+ '@type' => 'CorruptedJSONLD',
75
+ '_parse_error' => e.message,
76
+ '_raw' => json_str
77
+ }
78
+ end
79
+ end
80
+
81
+ schemas
82
+ end
83
+
84
+ def flatten_schemas(data)
85
+ case data
86
+ when Array
87
+ data.flat_map { |item| flatten_schemas(item) }
88
+ when Hash
89
+ if data['@graph'].is_a?(Array)
90
+ data['@graph'].flat_map { |item| flatten_schemas(item) }
91
+ else
92
+ [data]
93
+ end
94
+ else
95
+ []
96
+ end
97
+ end
98
+
99
+ def validate_schema(schema, idx)
100
+ type = schema['@type'] || 'Unknown'
101
+ errors = []
102
+ warnings = []
103
+ feature_name = nil
104
+
105
+ if schema['_parse_error']
106
+ return {
107
+ index: idx + 1,
108
+ type: 'SyntaxError',
109
+ feature: 'Malformed JSON-LD',
110
+ eligible: false,
111
+ errors: ["JSON Syntax Error: #{schema['_parse_error']}"],
112
+ warnings: [],
113
+ raw: schema['_raw'],
114
+ patch: nil
115
+ }
116
+ end
117
+
118
+ case type
119
+ when 'Product'
120
+ feature_name = 'Merchant Product Rich Card'
121
+ errors << 'Missing "name"' unless schema['name']
122
+ errors << 'Missing "image"' unless schema['image']
123
+
124
+ has_pricing = schema['offers'] || schema['review'] || schema['aggregateRating']
125
+ errors << 'Must provide at least one of "offers", "review", or "aggregateRating"' unless has_pricing
126
+
127
+ if schema['offers']
128
+ offers = schema['offers'].is_a?(Array) ? schema['offers'].first : schema['offers']
129
+ errors << 'Missing "offers.price" or "offers.lowPrice"' unless offers['price'] || offers['lowPrice']
130
+ errors << 'Missing "offers.priceCurrency"' unless offers['priceCurrency']
131
+ warnings << 'Missing "offers.availability" (e.g. InStock)' unless offers['availability']
132
+ warnings << 'Missing "offers.priceValidUntil"' unless offers['priceValidUntil']
133
+ end
134
+
135
+ if schema['aggregateRating']
136
+ ar = schema['aggregateRating']
137
+ errors << 'Missing "aggregateRating.ratingValue"' unless ar['ratingValue']
138
+ errors << 'Missing "aggregateRating.ratingCount" or "aggregateRating.reviewCount"' unless ar['ratingCount'] || ar['reviewCount']
139
+ end
140
+
141
+ when 'FAQPage'
142
+ feature_name = 'Interactive FAQ Accordion'
143
+ entities = schema['mainEntity']
144
+ if !entities.is_a?(Array) || entities.empty?
145
+ errors << 'FAQPage must contain a non-empty "mainEntity" array of Question objects'
146
+ else
147
+ entities.each_with_index do |q, q_idx|
148
+ q_num = q_idx + 1
149
+ errors << "Question ##{q_num} missing 'name'" unless q['name']
150
+ if !q['acceptedAnswer'] || !q['acceptedAnswer']['text']
151
+ errors << "Question ##{q_num} missing 'acceptedAnswer.text'"
152
+ end
153
+ end
154
+ end
155
+
156
+ when 'HowTo'
157
+ feature_name = 'Step-by-Step Guided HowTo Carousel'
158
+ errors << 'Missing "name"' unless schema['name']
159
+ steps = schema['step']
160
+ if !steps.is_a?(Array) || steps.empty?
161
+ errors << 'HowTo must contain a non-empty "step" array'
162
+ else
163
+ steps.each_with_index do |s, s_idx|
164
+ s_num = s_idx + 1
165
+ errors << "Step ##{s_num} missing 'text' or 'name'" unless s['text'] || s['name']
166
+ end
167
+ end
168
+ warnings << 'Missing "totalTime"' unless schema['totalTime']
169
+ warnings << 'Missing "image"' unless schema['image']
170
+
171
+ when 'Article', 'NewsArticle', 'BlogPosting'
172
+ feature_name = 'Top Stories Carousel & Author Card'
173
+ errors << 'Missing "headline"' unless schema['headline']
174
+ errors << 'Missing "image"' unless schema['image']
175
+ errors << 'Missing "datePublished"' unless schema['datePublished']
176
+ errors << 'Missing "author"' unless schema['author']
177
+
178
+ if schema['author']
179
+ authors = schema['author'].is_a?(Array) ? schema['author'] : [schema['author']]
180
+ authors.each do |a|
181
+ if a.is_a?(Hash)
182
+ warnings << 'Author missing "url" (E-E-A-T profile link)' unless a['url']
183
+ end
184
+ end
185
+ end
186
+ warnings << 'Missing "dateModified"' unless schema['dateModified']
187
+ warnings << 'Missing "publisher" with logo' unless schema['publisher']
188
+
189
+ when 'SoftwareApplication', 'WebApplication', 'MobileApplication'
190
+ feature_name = 'Software Application Rich Snippet'
191
+ errors << 'Missing "name"' unless schema['name']
192
+ errors << 'Missing "operatingSystem"' unless schema['operatingSystem']
193
+ errors << 'Missing "applicationCategory"' unless schema['applicationCategory']
194
+ errors << 'Missing "offers"' unless schema['offers']
195
+ warnings << 'Missing "aggregateRating"' unless schema['aggregateRating']
196
+
197
+ when 'LocalBusiness', 'Store', 'Restaurant', 'MedicalBusiness'
198
+ feature_name = 'Local 3-Pack & Knowledge Graph Card'
199
+ errors << 'Missing "name"' unless schema['name']
200
+ errors << 'Missing "address"' unless schema['address']
201
+ errors << 'Missing "telephone"' unless schema['telephone']
202
+ warnings << 'Missing "openingHoursSpecification"' unless schema['openingHoursSpecification'] || schema['openingHours']
203
+ warnings << 'Missing "geo" coordinates (latitude/longitude)' unless schema['geo']
204
+
205
+ when 'BreadcrumbList'
206
+ feature_name = 'Interactive Breadcrumb Path'
207
+ items = schema['itemListElement']
208
+ if !items.is_a?(Array) || items.empty?
209
+ errors << 'BreadcrumbList must contain "itemListElement" array'
210
+ else
211
+ items.each_with_index do |it, i_idx|
212
+ errors << "Breadcrumb ##{i_idx + 1} missing 'position'" unless it['position']
213
+ errors << "Breadcrumb ##{i_idx + 1} missing 'name'" unless it['name']
214
+ errors << "Breadcrumb ##{i_idx + 1} missing 'item'" unless it['item']
215
+ end
216
+ end
217
+
218
+ when 'Recipe'
219
+ feature_name = 'Recipe Rich Card with Cooking Specs'
220
+ errors << 'Missing "name"' unless schema['name']
221
+ errors << 'Missing "image"' unless schema['image']
222
+ errors << 'Missing "recipeIngredient"' unless schema['recipeIngredient']
223
+ errors << 'Missing "recipeInstructions"' unless schema['recipeInstructions']
224
+ warnings << 'Missing "cookTime"' unless schema['cookTime']
225
+ warnings << 'Missing "aggregateRating"' unless schema['aggregateRating']
226
+
227
+ when 'VideoObject'
228
+ feature_name = 'Video SERP Thumbnail & Key Moments'
229
+ errors << 'Missing "name"' unless schema['name']
230
+ errors << 'Missing "description"' unless schema['description']
231
+ errors << 'Missing "thumbnailUrl"' unless schema['thumbnailUrl']
232
+ errors << 'Missing "uploadDate"' unless schema['uploadDate']
233
+
234
+ when 'Course'
235
+ feature_name = 'Course Listing Rich Carousel'
236
+ errors << 'Missing "name"' unless schema['name']
237
+ errors << 'Missing "description"' unless schema['description']
238
+ errors << 'Missing "provider"' unless schema['provider']
239
+
240
+ when 'JobPosting'
241
+ feature_name = 'Google for Jobs Interactive Listing'
242
+ errors << 'Missing "title"' unless schema['title']
243
+ errors << 'Missing "description"' unless schema['description']
244
+ errors << 'Missing "datePosted"' unless schema['datePosted']
245
+ errors << 'Missing "hiringOrganization"' unless schema['hiringOrganization']
246
+
247
+ when 'Event'
248
+ feature_name = 'Event SERP Listing & Tickets'
249
+ errors << 'Missing "name"' unless schema['name']
250
+ errors << 'Missing "startDate"' unless schema['startDate']
251
+ errors << 'Missing "location"' unless schema['location']
252
+
253
+ when 'Review', 'AggregateRating'
254
+ feature_name = 'Star Rating Snippet'
255
+ errors << 'Missing "ratingValue"' unless schema['ratingValue']
256
+ warnings << 'Missing "bestRating" (defaults to 5)' unless schema['bestRating']
257
+
258
+ else
259
+ feature_name = "Schema.org #{type}"
260
+ warnings << "Generic schema type '#{type}' does not trigger a dedicated Google Rich Result"
261
+ end
262
+
263
+ eligible = errors.empty? && !feature_name.start_with?('Schema.org')
264
+
265
+ patch = generate_patch(schema, errors, warnings)
266
+
267
+ {
268
+ index: idx + 1,
269
+ type: type,
270
+ feature: feature_name,
271
+ eligible: eligible,
272
+ errors: errors,
273
+ warnings: warnings,
274
+ raw: schema,
275
+ patch: patch
276
+ }
277
+ end
278
+
279
+ def generate_patch(schema, errors, warnings)
280
+ return nil if errors.empty? && warnings.empty?
281
+
282
+ patched = schema.dup
283
+ type = patched['@type']
284
+
285
+ case type
286
+ when 'Product'
287
+ product_url = @url.to_s.start_with?('http') ? @url : nil
288
+ patched['image'] ||= [product_url ? "#{product_url.sub(%r{/$}, '')}/product.jpg" : 'https://schema.org/ProductImage']
289
+ if !patched['offers'] && !patched['aggregateRating']
290
+ patched['offers'] = {
291
+ '@type' => 'Offer',
292
+ 'price' => '49.99',
293
+ 'priceCurrency' => 'USD',
294
+ 'priceValidUntil' => (Time.now + (365 * 86400)).strftime('%Y-%m-%d'),
295
+ 'availability' => 'https://schema.org/InStock',
296
+ 'url' => product_url || ''
297
+ }
298
+ end
299
+ if patched['offers'].is_a?(Hash)
300
+ patched['offers']['priceValidUntil'] ||= (Time.now + (365 * 86400)).strftime('%Y-%m-%d')
301
+ patched['offers']['availability'] ||= 'https://schema.org/InStock'
302
+ end
303
+
304
+ when 'Article', 'BlogPosting'
305
+ base_url = @url.to_s.start_with?('http') ? @url : nil
306
+ patched['datePublished'] ||= Time.now.strftime('%Y-%m-%dT%H:%M:%S%:z')
307
+ patched['dateModified'] ||= Time.now.strftime('%Y-%m-%dT%H:%M:%S%:z')
308
+ patched['image'] ||= [base_url ? "#{base_url.sub(%r{/$}, '')}/featured.jpg" : 'https://schema.org/ArticleImage']
309
+ if !patched['author']
310
+ author_url = base_url ? "#{base_url.sub(%r{/$}, '')}/author" : ''
311
+ patched['author'] = {
312
+ '@type' => 'Person',
313
+ 'name' => 'Author',
314
+ 'url' => author_url
315
+ }
316
+ end
317
+
318
+ when 'FAQPage'
319
+ if !patched['mainEntity'].is_a?(Array) || patched['mainEntity'].empty?
320
+ patched['mainEntity'] = [
321
+ {
322
+ '@type' => 'Question',
323
+ 'name' => 'Question?',
324
+ 'acceptedAnswer' => {
325
+ '@type' => 'Answer',
326
+ 'text' => 'Answer.'
327
+ }
328
+ }
329
+ ]
330
+ end
331
+ end
332
+
333
+ patched
334
+ end
335
+
336
+ def synthesize_report(results)
337
+ total = results.size
338
+ eligible_count = results.count { |r| r[:eligible] }
339
+ total_errors = results.sum { |r| r[:errors].size }
340
+ total_warnings = results.sum { |r| r[:warnings].size }
341
+
342
+ # Duplicate detection across same entity types
343
+ types = results.map { |r| r[:type] }
344
+ duplicates = types.select { |t| types.count(t) > 1 }.uniq
345
+ if duplicates.any?
346
+ duplicates.each do |dup_type|
347
+ results.each do |r|
348
+ if r[:type] == dup_type
349
+ r[:warnings] << "Duplicate '#{dup_type}' schema detected on page. Google advises consolidating duplicate entities into a single JSON-LD block."
350
+ end
351
+ end
352
+ end
353
+ total_warnings = results.sum { |r| r[:warnings].size }
354
+ end
355
+
356
+ # Score calculation
357
+ score = if total == 0
358
+ 0
359
+ else
360
+ raw = (eligible_count.to_f / total) * 70 + [30 - (total_errors * 10) - (total_warnings * 2), 0].max
361
+ [[raw.round, 100].min, 0].max
362
+ end
363
+
364
+ grade = case score
365
+ when 95..100 then 'A+'
366
+ when 85..94 then 'A'
367
+ when 70..84 then 'B'
368
+ when 50..69 then 'C'
369
+ else 'F'
370
+ end
371
+
372
+ eligible_features = results.select { |r| r[:eligible] }.map { |r| r[:feature] }.uniq
373
+
374
+ {
375
+ url: @url,
376
+ total_schemas_detected: total,
377
+ eligible_for_rich_results: eligible_count > 0,
378
+ eligible_features: eligible_features,
379
+ score: score,
380
+ grade: grade,
381
+ critical_errors_count: total_errors,
382
+ warnings_count: total_warnings,
383
+ duplicate_types: duplicates,
384
+ schemas: results
385
+ }
386
+ end
387
+ end
388
+ end
@@ -12,14 +12,33 @@ module GSC
12
12
  @target_url = target_url.to_s.strip
13
13
  @target_url = "https://#{@target_url}" unless @target_url =~ %r{^https?://}
14
14
  @uri = URI.parse(@target_url)
15
- @robots_url = "#{@uri.scheme}://#{@uri.host}:#{@uri.port}/robots.txt"
15
+ @robots_url = if (@uri.port == 80 && @uri.scheme == 'http') || (@uri.port == 443 && @uri.scheme == 'https')
16
+ "#{@uri.scheme}://#{@uri.host}/robots.txt"
17
+ else
18
+ "#{@uri.scheme}://#{@uri.host}:#{@uri.port}/robots.txt"
19
+ end
16
20
  end
17
21
 
18
- def fetch_robots_txt
19
- res = Net::HTTP.get_response(URI.parse(@robots_url))
20
- return '' unless res.code == '200'
22
+ def fetch_robots_txt(url = @robots_url, limit = 3)
23
+ return '' if limit <= 0
24
+ uri = URI.parse(url) rescue nil
25
+ return '' unless uri && uri.host
26
+
27
+ http = Net::HTTP.new(uri.host, uri.port)
28
+ http.use_ssl = (uri.scheme == 'https')
29
+ http.open_timeout = 4
30
+ http.read_timeout = 6
31
+ req = Net::HTTP::Get.new(uri.request_uri.empty? ? '/robots.txt' : uri.request_uri)
32
+ req['User-Agent'] = 'Mozilla/5.0 (compatible; GSC-SEO-Auditor/2.2)'
33
+
34
+ res = http.request(req)
35
+ if res.is_a?(Net::HTTPRedirection) && res['location']
36
+ new_url = URI.join(url, res['location']).to_s rescue nil
37
+ return new_url ? fetch_robots_txt(new_url, limit - 1) : ''
38
+ end
21
39
 
22
- res.body.force_encoding('UTF-8')
40
+ return '' unless res.code == '200'
41
+ res.body.to_s.force_encoding('UTF-8').scrub
23
42
  rescue StandardError
24
43
  ''
25
44
  end
@@ -57,27 +76,39 @@ module GSC
57
76
  private
58
77
 
59
78
  def parse_rules(target_ua)
60
- rules = []
61
- current_ua = nil
62
- applies = false
79
+ groups = Hash.new { |h, k| h[k] = [] }
80
+ current_uas = []
81
+ in_directives = false
63
82
 
64
83
  @robots_content.each_line do |line|
65
84
  line = line.strip.sub(/#.*$/, '')
66
85
  next if line.empty?
67
86
 
68
87
  if line =~ /^User-agent:\s*(.+)$/i
69
- current_ua = $1.strip.downcase
70
- applies = (current_ua == '*' || current_ua == target_ua)
71
- elsif applies && line =~ /^Disallow:\s*(.*)$/i
88
+ if in_directives
89
+ current_uas = []
90
+ in_directives = false
91
+ end
92
+ current_uas << $1.strip.downcase
93
+ elsif line =~ /^Disallow:\s*(.*)$/i
94
+ in_directives = true
72
95
  val = $1.strip
73
- rules << { type: :disallow, path: val } unless val.empty?
74
- elsif applies && line =~ /^Allow:\s*(.*)$/i
96
+ current_uas.each { |ua| groups[ua] << { type: :disallow, path: val } unless val.empty? }
97
+ elsif line =~ /^Allow:\s*(.*)$/i
98
+ in_directives = true
75
99
  val = $1.strip
76
- rules << { type: :allow, path: val } unless val.empty?
100
+ current_uas.each { |ua| groups[ua] << { type: :allow, path: val } unless val.empty? }
77
101
  end
78
102
  end
79
103
 
80
- rules
104
+ # RFC 9309: Specific user-agent match takes absolute precedence over wildcard '*'
105
+ if groups.key?(target_ua) && groups[target_ua].any?
106
+ groups[target_ua]
107
+ elsif groups.key?('*')
108
+ groups['*']
109
+ else
110
+ []
111
+ end
81
112
  end
82
113
  end
83
114
  end