gsc-cli 2.1.0 → 2.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. checksums.yaml +4 -4
  2. data/AUTH.md +4 -1
  3. data/README.md +448 -408
  4. data/bin/gsc +28067 -5661
  5. data/dist/gsc +29121 -5046
  6. data/lib/gsc/aio_hunter.rb +343 -0
  7. data/lib/gsc/answer_synthesizer.rb +157 -0
  8. data/lib/gsc/api.rb +53 -1
  9. data/lib/gsc/auth.rb +26 -0
  10. data/lib/gsc/brand_segmenter.rb +140 -0
  11. data/lib/gsc/cache_manager.rb +806 -0
  12. data/lib/gsc/cannibalization_analyzer.rb +141 -0
  13. data/lib/gsc/canonical_chains.rb +367 -0
  14. data/lib/gsc/citation_simulator.rb +339 -0
  15. data/lib/gsc/cli/aio_hunter.rb +154 -0
  16. data/lib/gsc/cli/analytics.rb +788 -0
  17. data/lib/gsc/cli/audit.rb +1976 -0
  18. data/lib/gsc/cli/base.rb +384 -0
  19. data/lib/gsc/cli/cache.rb +266 -0
  20. data/lib/gsc/cli/canonical.rb +223 -0
  21. data/lib/gsc/cli/citation_simulator.rb +152 -0
  22. data/lib/gsc/cli/dashboard.rb +354 -0
  23. data/lib/gsc/cli/doctor.rb +129 -0
  24. data/lib/gsc/cli/eeat.rb +125 -0
  25. data/lib/gsc/cli/ga4.rb +852 -0
  26. data/lib/gsc/cli/growth.rb +650 -0
  27. data/lib/gsc/cli/hreflang.rb +164 -0
  28. data/lib/gsc/cli/image_seo.rb +162 -0
  29. data/lib/gsc/cli/indexing.rb +458 -0
  30. data/lib/gsc/cli/intent_shift.rb +125 -0
  31. data/lib/gsc/cli/keyword_value.rb +134 -0
  32. data/lib/gsc/cli/keywords.rb +795 -0
  33. data/lib/gsc/cli/landing_roi.rb +308 -0
  34. data/lib/gsc/cli/low_ctr.rb +213 -0
  35. data/lib/gsc/cli/mobile_parity.rb +150 -0
  36. data/lib/gsc/cli/report.rb +100 -0
  37. data/lib/gsc/cli/rich_results.rb +172 -0
  38. data/lib/gsc/cli/schema_generate.rb +149 -0
  39. data/lib/gsc/cli/seasonal.rb +232 -0
  40. data/lib/gsc/cli/security.rb +153 -0
  41. data/lib/gsc/cli/setup.rb +1291 -0
  42. data/lib/gsc/cli/sitemap_tree.rb +143 -0
  43. data/lib/gsc/cli/skill_pack.rb +62 -0
  44. data/lib/gsc/cli/soft_404.rb +199 -0
  45. data/lib/gsc/cli/sparkline.rb +227 -0
  46. data/lib/gsc/cli/watchdog.rb +150 -0
  47. data/lib/gsc/cli/zombie_purger.rb +208 -0
  48. data/lib/gsc/cli.rb +707 -5265
  49. data/lib/gsc/cli_advanced.rb +987 -44
  50. data/lib/gsc/client.rb +17 -2
  51. data/lib/gsc/color.rb +16 -1
  52. data/lib/gsc/command_registry.rb +47 -9
  53. data/lib/gsc/config.rb +2 -2
  54. data/lib/gsc/ctr_curve.rb +115 -0
  55. data/lib/gsc/decay_predictor.rb +322 -0
  56. data/lib/gsc/doctor.rb +434 -0
  57. data/lib/gsc/eeat_auditor.rb +428 -0
  58. data/lib/gsc/entity_auditor.rb +229 -0
  59. data/lib/gsc/firewall_scanner.rb +733 -0
  60. data/lib/gsc/geo_auditor.rb +368 -0
  61. data/lib/gsc/google_trends.rb +8 -1
  62. data/lib/gsc/heading_validator.rb +283 -0
  63. data/lib/gsc/hreflang_validator.rb +412 -0
  64. data/lib/gsc/image_seo.rb +286 -0
  65. data/lib/gsc/indexing_queue.rb +179 -0
  66. data/lib/gsc/indexnow.rb +93 -0
  67. data/lib/gsc/intent_shift.rb +188 -0
  68. data/lib/gsc/internal_links.rb +153 -36
  69. data/lib/gsc/keyword_value.rb +191 -0
  70. data/lib/gsc/landing_roi.rb +195 -0
  71. data/lib/gsc/llms_generator.rb +343 -22
  72. data/lib/gsc/low_ctr_rewriter.rb +408 -0
  73. data/lib/gsc/mobile_parity.rb +222 -0
  74. data/lib/gsc/network_tracer.rb +8 -1
  75. data/lib/gsc/page_analyzer.rb +47 -7
  76. data/lib/gsc/prompts.rb +38 -29
  77. data/lib/gsc/questions_harvester.rb +178 -0
  78. data/lib/gsc/report_generator.rb +461 -0
  79. data/lib/gsc/rich_results.rb +388 -0
  80. data/lib/gsc/robots_checker.rb +46 -15
  81. data/lib/gsc/schema_generator.rb +788 -0
  82. data/lib/gsc/schema_validator.rb +36 -38
  83. data/lib/gsc/seasonal_predictor.rb +381 -0
  84. data/lib/gsc/security_scanner.rb +496 -0
  85. data/lib/gsc/serp_feature_detector.rb +359 -0
  86. data/lib/gsc/serp_preview.rb +108 -22
  87. data/lib/gsc/site_crawler.rb +113 -21
  88. data/lib/gsc/sitemap_loader.rb +15 -4
  89. data/lib/gsc/sitemap_tree.rb +301 -0
  90. data/lib/gsc/skill_pack.rb +195 -0
  91. data/lib/gsc/soft_404_analyzer.rb +385 -0
  92. data/lib/gsc/sparkline.rb +171 -0
  93. data/lib/gsc/speed_correlator.rb +416 -0
  94. data/lib/gsc/striking_playbook.rb +190 -0
  95. data/lib/gsc/title_optimizer.rb +420 -0
  96. data/lib/gsc/vault.rb +260 -0
  97. data/lib/gsc/version.rb +1 -1
  98. data/lib/gsc/watchdog.rb +235 -0
  99. data/lib/gsc/zombie_purger.rb +366 -0
  100. data/lib/gsc.rb +118 -0
  101. metadata +75 -1
@@ -0,0 +1,359 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'uri'
5
+ require 'net/http'
6
+ require 'json'
7
+ require 'time'
8
+ require_relative 'google_suggest' if File.exist?(File.expand_path('google_suggest.rb', __dir__))
9
+
10
+ module GSC
11
+ class SerpFeatureDetector
12
+ attr_reader :query, :options
13
+
14
+ def initialize(query, options = {})
15
+ @query = query.to_s.dup.force_encoding('UTF-8').scrub.strip
16
+ @options = options
17
+ end
18
+
19
+ def detect
20
+ intent = classify_intent(@query)
21
+ paa_questions = extract_live_paa_questions(@query)
22
+ live_serp = sample_live_serp(@query)
23
+
24
+ ai_overview = detect_ai_overview(@query, intent)
25
+ featured_snippet = detect_featured_snippet(@query, intent)
26
+ local_pack = detect_local_pack(@query)
27
+ video_carousel = detect_video_carousel(@query, live_serp)
28
+ forum_discussions = detect_forum_discussions(@query, live_serp)
29
+ shopping_pack = detect_shopping_pack(@query, intent)
30
+ sitelinks = detect_sitelinks(@query, intent)
31
+
32
+ zero_click = calculate_zero_click_risk(
33
+ ai_overview: ai_overview,
34
+ featured_snippet: featured_snippet,
35
+ paa_count: paa_questions.size,
36
+ local_pack: local_pack,
37
+ shopping_pack: shopping_pack,
38
+ video_carousel: video_carousel
39
+ )
40
+
41
+ playbook = generate_capture_playbook(
42
+ query: @query,
43
+ intent: intent,
44
+ ai_overview: ai_overview,
45
+ featured_snippet: featured_snippet,
46
+ paa_questions: paa_questions
47
+ )
48
+
49
+ {
50
+ query: @query,
51
+ timestamp: Time.now.utc.iso8601,
52
+ intent: intent,
53
+ zero_click_threat: zero_click,
54
+ features: {
55
+ ai_overview: ai_overview,
56
+ featured_snippet: featured_snippet,
57
+ people_also_ask: {
58
+ detected: !paa_questions.empty?,
59
+ count: paa_questions.size,
60
+ questions: paa_questions
61
+ },
62
+ local_3_pack: local_pack,
63
+ video_carousel: video_carousel,
64
+ discussions_and_forums: forum_discussions,
65
+ shopping_pack: shopping_pack,
66
+ sitelinks: sitelinks
67
+ },
68
+ serp_sampling: live_serp,
69
+ playbook: playbook
70
+ }
71
+ end
72
+
73
+ private
74
+
75
+ def classify_intent(query)
76
+ q = query.downcase
77
+
78
+ if q =~ /\b(near me|in [a-z]+|city|store|repair|dentist|plumber|gym|shop|restaurant|agency)\b/i
79
+ { primary: 'Local', secondary: 'Commercial', description: 'User seeking physical or regional local service' }
80
+ elsif q =~ /\b(buy|order|purchase|coupon|discount|deal|cheap|for sale|pricing|price|cost)\b/i
81
+ { primary: 'Transactional', secondary: 'Commercial', description: 'User is ready to make an immediate purchase' }
82
+ elsif q =~ /\b(best|top|review|vs|versus|compare|alternative|alternatives|software|tool|app|guide)\b/i
83
+ { primary: 'Commercial', secondary: 'Informational', description: 'User evaluating products or services before purchasing' }
84
+ elsif q =~ /\b(login|portal|website|account|dashboard|sign in|app\.|\.com)\b/i
85
+ { primary: 'Navigational', secondary: 'Brand', description: 'User navigating to a specific brand destination' }
86
+ else
87
+ { primary: 'Informational', secondary: 'Research', description: 'User seeking knowledge, definitions, answers or tutorials' }
88
+ end
89
+ end
90
+
91
+ def extract_live_paa_questions(query)
92
+ suggest = GSC::GoogleSuggest.new(query)
93
+ raw = suggest.fetch(questions: true)
94
+ extracted = []
95
+
96
+ if raw.is_a?(Hash)
97
+ raw.each_value do |items|
98
+ Array(items).each do |item|
99
+ term = item.is_a?(Hash) ? item[:term] : item.to_s
100
+ clean = term.to_s.strip
101
+ extracted << clean if !clean.empty? && !extracted.include?(clean)
102
+ end
103
+ end
104
+ elsif raw.is_a?(Array)
105
+ raw.each do |entry|
106
+ if entry.is_a?(Array) && entry[1].is_a?(Array)
107
+ entry[1].each do |item|
108
+ term = item.is_a?(Hash) ? item[:term] : item.to_s
109
+ clean = term.to_s.strip
110
+ extracted << clean if !clean.empty? && !extracted.include?(clean)
111
+ end
112
+ elsif entry.is_a?(Hash) && entry[:term]
113
+ term = entry[:term].to_s.strip
114
+ extracted << term if !term.empty? && !extracted.include?(term)
115
+ end
116
+ end
117
+ end
118
+
119
+ # Filter for relevance to query tokens
120
+ tokens = query.downcase.split(/\s+/).reject { |t| t.length < 3 }
121
+ relevant = if tokens.empty?
122
+ extracted
123
+ else
124
+ matched = extracted.select { |q| tokens.any? { |t| q.downcase.include?(t) } }
125
+ matched.empty? ? extracted : matched
126
+ end
127
+
128
+ relevant.first(8)
129
+ rescue StandardError
130
+ []
131
+ end
132
+
133
+ def sample_live_serp(query)
134
+ results = []
135
+ uri = URI("https://www.bing.com/search?q=#{URI.encode_www_form_component(query)}&setlang=en-US&cc=US")
136
+ http = Net::HTTP.new(uri.host, uri.port)
137
+ http.use_ssl = true
138
+ http.open_timeout = 3
139
+ http.read_timeout = 3
140
+ req = Net::HTTP::Get.new(uri.request_uri)
141
+ req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36'
142
+ req['Accept-Language'] = 'en-US,en;q=0.9'
143
+
144
+ res = http.request(req)
145
+ if res.is_a?(Net::HTTPSuccess)
146
+ body = res.body.to_s.dup.force_encoding('UTF-8').scrub
147
+ matches = body.scan(/<li class="b_algo"[^>]*>.*?<h2><a[^>]*href="([^"]+)"[^>]*>(.*?)<\/a><\/h2>(?:.*?<p[^>]*>(.*?)<\/p>)?/im)
148
+ matches.first(6).each_with_index do |(url, raw_title, raw_snippet), idx|
149
+ title = raw_title.to_s.gsub(/<[^>]+>/, '').strip
150
+ snippet = raw_snippet.to_s.gsub(/<[^>]+>/, '').strip
151
+ domain = URI.parse(url).host rescue url
152
+ results << {
153
+ position: idx + 1,
154
+ title: title,
155
+ url: url,
156
+ domain: domain,
157
+ snippet: snippet[0..140]
158
+ }
159
+ end
160
+ end
161
+ results
162
+ rescue StandardError
163
+ []
164
+ end
165
+
166
+ def detect_ai_overview(query, intent)
167
+ q = query.downcase
168
+ triggers = []
169
+ probability = 15
170
+
171
+ if intent[:primary] == 'Informational'
172
+ probability += 40
173
+ triggers << 'Informational search intent'
174
+ end
175
+
176
+ if q =~ /^(what is|how to|why does|how do|what are|difference between|steps to|guide to)\b/i
177
+ probability += 35
178
+ triggers << 'Interrogative / procedural query prefix'
179
+ elsif q.include?(' vs ') || q.include?(' versus ')
180
+ probability += 30
181
+ triggers << 'Direct comparative product/concept syntax'
182
+ end
183
+
184
+ if q.split(/\s+/).size >= 4
185
+ probability += 10
186
+ triggers << 'Long-tail semantic query depth'
187
+ end
188
+
189
+ probability = [probability, 95].min
190
+ detected = probability >= 60
191
+
192
+ {
193
+ detected: detected,
194
+ probability_pct: probability,
195
+ triggers: triggers,
196
+ displacement_pixels: detected ? 850 : 0,
197
+ impact: detected ? 'Severe: Google Gemini AI Overview displaces Top 1 organic result below the viewport fold.' : 'Low: Traditional organic listings occupy top positions.',
198
+ counter_strategy: 'Structure content with direct 40–50 word definition answer, bulleted takeaways, and FAQPage/Question JSON-LD markup.'
199
+ }
200
+ end
201
+
202
+ def detect_featured_snippet(query, intent)
203
+ q = query.downcase
204
+ format = :none
205
+ probability = 10
206
+ target_recipe = ''
207
+
208
+ if q =~ /^(how to|steps to|guide|tutorial|how do i)\b/i
209
+ format = :ordered_list
210
+ probability = 90
211
+ target_recipe = 'Use an ordered list (<ol><li>) with 5–8 actionable sequential steps directly under an H2 heading.'
212
+ elsif q.include?(' vs ') || q.include?('difference between') || q =~ /\b(pricing|cost|plans|tiers|specs)\b/i
213
+ format = :table
214
+ probability = 75
215
+ target_recipe = 'Provide a structured HTML <table> with clear column headers (<th>) comparing key attributes and metrics.'
216
+ elsif q =~ /\b(best|top|types of|examples of|list of)\b/i
217
+ format = :unordered_list
218
+ probability = 80
219
+ target_recipe = 'Use an unordered bullet list (<ul><li>) highlighting 5–10 items with bold introductory headers.'
220
+ elsif q =~ /^(what is|who is|definition of|meaning of|what does)\b/i || intent[:primary] == 'Informational'
221
+ format = :paragraph
222
+ probability = 85
223
+ target_recipe = 'Place target query in <h2>, followed immediately by a concise 42–55 word direct definition answer block.'
224
+ end
225
+
226
+ {
227
+ detected: format != :none,
228
+ target_format: format.to_s,
229
+ probability_pct: probability,
230
+ optimal_length: format == :paragraph ? '40–60 words (~280–350 chars)' : '5–8 structured items',
231
+ capture_prescription: target_recipe
232
+ }
233
+ end
234
+
235
+ def detect_local_pack(query)
236
+ q = query.downcase
237
+ is_local = q =~ /\b(near me|in [a-z]+|city|store|repair|dentist|plumber|gym|shop|restaurant|agency|services|near)\b/i
238
+ {
239
+ detected: !!is_local,
240
+ probability_pct: is_local ? 90 : 5,
241
+ notes: is_local ? 'Triggers Google Maps 3-Pack; local organic listings appear below maps.' : 'No local intent detected.'
242
+ }
243
+ end
244
+
245
+ def detect_video_carousel(query, live_serp)
246
+ q = query.downcase
247
+ has_video_kw = q =~ /\b(how to|tutorial|review|walkthrough|guide|setup|install|demo|video|diy|unboxing)\b/i
248
+ serp_has_youtube = live_serp.any? { |r| r[:url].to_s.include?('youtube.com') }
249
+ detected = !!(has_video_kw || serp_has_youtube)
250
+
251
+ {
252
+ detected: detected,
253
+ probability_pct: detected ? 80 : 15,
254
+ notes: detected ? 'Video carousel likely above or between organic results. YouTube videos dominate.' : 'Low video intent.'
255
+ }
256
+ end
257
+
258
+ def detect_forum_discussions(query, live_serp)
259
+ q = query.downcase
260
+ has_forum_kw = q =~ /\b(reddit|quora|worth it|opinions|review|anyone tried|experiences|recommendations|issues)\b/i
261
+ serp_has_forum = live_serp.any? { |r| r[:url].to_s =~ /(reddit\.com|quora\.com|community\.)/i }
262
+ detected = !!(has_forum_kw || serp_has_forum)
263
+
264
+ {
265
+ detected: detected,
266
+ probability_pct: detected ? 85 : 20,
267
+ notes: detected ? 'Google "Discussions and Forums" module active. Authentic first-person experiences prioritized.' : 'Standard commercial/informational listings dominate.'
268
+ }
269
+ end
270
+
271
+ def detect_shopping_pack(query, intent)
272
+ q = query.downcase
273
+ is_shopping = (intent[:primary] == 'Transactional') || (q =~ /\b(buy|price|cost|cheap|best|discount|sale|store|deals|shop|coupon)\b/i)
274
+ {
275
+ detected: !!is_shopping,
276
+ probability_pct: is_shopping ? 85 : 10,
277
+ notes: is_shopping ? 'Product grid / Google Shopping carousels occupy top above-the-fold position.' : 'Non-e-commerce SERP.'
278
+ }
279
+ end
280
+
281
+ def detect_sitelinks(query, intent)
282
+ is_brand = (intent[:primary] == 'Navigational') || (query.split(/\s+/).size <= 2 && query =~ /^[A-Z][a-zA-Z0-9]+$/)
283
+ {
284
+ detected: !!is_brand,
285
+ probability_pct: is_brand ? 90 : 15,
286
+ notes: is_brand ? 'Expanded 6-pack or 4-pack branded sitelinks trigger for primary domain.' : 'Standard single-line snippets.'
287
+ }
288
+ end
289
+
290
+ def calculate_zero_click_risk(ai_overview:, featured_snippet:, paa_count:, local_pack:, shopping_pack:, video_carousel:)
291
+ score = 10
292
+ score += 35 if ai_overview[:detected]
293
+ score += 25 if featured_snippet[:detected]
294
+ score += 15 if paa_count >= 3
295
+ score += 10 if local_pack[:detected]
296
+ score += 10 if shopping_pack[:detected]
297
+ score += 5 if video_carousel[:detected]
298
+
299
+ score = [score, 100].min
300
+
301
+ level = case score
302
+ when 0..30 then 'LOW'
303
+ when 31..55 then 'MODERATE'
304
+ when 56..79 then 'HIGH'
305
+ else 'SEVERE'
306
+ end
307
+
308
+ ctr_drop = case level
309
+ when 'LOW' then '-5% to -10%'
310
+ when 'MODERATE' then '-15% to -25%'
311
+ when 'HIGH' then '-30% to -45%'
312
+ when 'SEVERE' then '-50% to -65%'
313
+ end
314
+
315
+ {
316
+ score: score,
317
+ level: level,
318
+ estimated_organic_ctr_suppression: ctr_drop,
319
+ summary: "Zero-Click Threat is #{level} (#{score}/100). SERP features push standard organic rankings down."
320
+ }
321
+ end
322
+
323
+ def generate_capture_playbook(query:, intent:, ai_overview:, featured_snippet:, paa_questions:)
324
+ playbook = []
325
+
326
+ if featured_snippet[:detected]
327
+ playbook << {
328
+ target: "Featured Snippet (#{featured_snippet[:target_format].capitalize})",
329
+ action: featured_snippet[:capture_prescription],
330
+ priority: 'P1 - High Impact'
331
+ }
332
+ end
333
+
334
+ if ai_overview[:detected]
335
+ playbook << {
336
+ target: 'Google Gemini AI Overview Citation',
337
+ action: 'Provide authoritative factual statistics with citability markers (author bio, published date, quantitative percentages).',
338
+ priority: 'P1 - High Impact'
339
+ }
340
+ end
341
+
342
+ if !paa_questions.empty?
343
+ playbook << {
344
+ target: 'People Also Ask (PAA) Inclusion',
345
+ action: "Inject H3 headings answering top questions: \"#{paa_questions.first(3).join('", "')}\" using Schema.org FAQPage JSON-LD.",
346
+ priority: 'P2 - Traffic Expansion'
347
+ }
348
+ end
349
+
350
+ playbook << {
351
+ target: 'Entity Disambiguation',
352
+ action: 'Include sameAs Wikidata / Wikipedia references and Organization structured data to anchor topical authority.',
353
+ priority: 'P3 - Topical Authority'
354
+ }
355
+
356
+ playbook
357
+ end
358
+ end
359
+ end
@@ -1,54 +1,140 @@
1
1
  # encoding: utf-8
2
2
  # frozen_string_literal: true
3
3
 
4
+ require 'uri'
5
+
4
6
  module GSC
5
7
  class SerpPreview
6
- attr_reader :url, :data
8
+ attr_reader :url, :title, :desc, :data
9
+
10
+ DESKTOP_TITLE_PIXEL_LIMIT = 580.0
11
+ MOBILE_TITLE_PIXEL_LIMIT = 650.0
12
+ DESKTOP_DESC_PIXEL_LIMIT = 960.0
13
+ MOBILE_DESC_PIXEL_LIMIT = 680.0
7
14
 
8
- def initialize(url)
9
- @url = url.to_s.strip
15
+ def initialize(url = nil, title: nil, desc: nil)
16
+ @url = url.to_s.dup.force_encoding('UTF-8').scrub.strip
17
+ @custom_title = title ? title.to_s.dup.force_encoding('UTF-8').scrub : nil
18
+ @custom_desc = desc ? desc.to_s.dup.force_encoding('UTF-8').scrub : nil
10
19
  end
11
20
 
12
21
  def generate
22
+ if @custom_title || @custom_desc
23
+ generate_custom
24
+ elsif !@url.empty? && @url.start_with?('http://', 'https://')
25
+ generate_from_url
26
+ else
27
+ generate_custom
28
+ end
29
+ end
30
+
31
+ def self.estimate_pixel_width(str)
32
+ width = 0.0
33
+ str.to_s.each_char do |ch|
34
+ width += case ch
35
+ when /[WMwm]/ then 13.5
36
+ when /[ABCDEFGHKNOPQRSTUVXYZ]/ then 10.5
37
+ when /[abcdeghnopqrsuvxyz]/ then 8.5
38
+ when /[fIjt1l\|\ \.\:\;]/ then 4.5
39
+ else 9.0
40
+ end
41
+ end
42
+ width.round(1)
43
+ end
44
+
45
+ private
46
+
47
+ def generate_from_url
13
48
  pa = GSC::PageAnalyzer.new(@url)
14
49
  @data = pa.fetch_and_analyze
15
50
 
16
- title = @data.dig(:title, :text) || 'Untitled Page'
17
- desc = @data.dig(:meta_description, :text) || 'No meta description found.'
51
+ title = @custom_title || @data.dig(:title, :text) || 'Untitled Page'
52
+ desc = @custom_desc || @data.dig(:meta_description, :text) || 'No meta description found.'
18
53
  canonical = @data.dig(:canonical, :url) || @url
19
54
 
20
- # SERP pixel calculation approximation:
21
- # ~10px per character average for Arial 18px title
22
- title_chars = title.length
23
- is_truncated = title_chars > 60
24
-
25
- desktop_title = is_truncated ? "#{title[0..56]}..." : title
26
- desktop_snippet = desc.length > 155 ? "#{desc[0..152]}..." : desc
27
-
28
55
  og = @data[:open_graph] || {}
29
56
  twitter = @data[:twitter_card] || {}
30
57
 
31
- {
58
+ build_result(
32
59
  url: @url,
33
60
  canonical: canonical,
34
61
  title: title,
62
+ desc: desc,
63
+ og_title: og['og:title'],
64
+ og_description: og['og:description'],
65
+ og_image: og['og:image'],
66
+ twitter_card: twitter['twitter:card']
67
+ )
68
+ end
69
+
70
+ def generate_custom
71
+ url = @url.empty? ? (Config.default_domain ? "https://#{Config.default_domain}/page" : '/') : @url
72
+ title = @custom_title || 'Untitled Page'
73
+ desc = @custom_desc || ''
74
+
75
+ build_result(
76
+ url: url,
77
+ canonical: url,
78
+ title: title,
79
+ desc: desc
80
+ )
81
+ end
82
+
83
+ def build_result(url:, canonical:, title:, desc:, og_title: nil, og_description: nil, og_image: nil, twitter_card: nil)
84
+ t_px = self.class.estimate_pixel_width(title)
85
+ d_px = self.class.estimate_pixel_width(desc)
86
+
87
+ title_truncated = t_px > DESKTOP_TITLE_PIXEL_LIMIT
88
+ desc_truncated = d_px > DESKTOP_DESC_PIXEL_LIMIT
89
+
90
+ desktop_title = title_truncated ? truncate_to_pixels(title, DESKTOP_TITLE_PIXEL_LIMIT) : title
91
+ desktop_desc = desc_truncated ? truncate_to_pixels(desc, DESKTOP_DESC_PIXEL_LIMIT) : desc
92
+
93
+ {
94
+ url: url,
95
+ canonical: canonical,
96
+ title: title,
35
97
  meta_description: desc,
36
- truncation_risk: is_truncated,
98
+ metrics: {
99
+ title_chars: title.length,
100
+ title_pixel_est: t_px,
101
+ title_desktop_limit: DESKTOP_TITLE_PIXEL_LIMIT,
102
+ title_truncated: title_truncated,
103
+ desc_chars: desc.length,
104
+ desc_pixel_est: d_px,
105
+ desc_desktop_limit: DESKTOP_DESC_PIXEL_LIMIT,
106
+ desc_truncated: desc_truncated
107
+ },
37
108
  desktop_serp: {
109
+ breadcrumb: format_breadcrumb(canonical),
38
110
  title: desktop_title,
39
- snippet: desktop_snippet,
40
- breadcrumb: format_breadcrumb(canonical)
111
+ snippet: desktop_desc
112
+ },
113
+ mobile_serp: {
114
+ breadcrumb: format_breadcrumb(canonical),
115
+ title: t_px > MOBILE_TITLE_PIXEL_LIMIT ? truncate_to_pixels(title, MOBILE_TITLE_PIXEL_LIMIT) : title,
116
+ snippet: d_px > MOBILE_DESC_PIXEL_LIMIT ? truncate_to_pixels(desc, MOBILE_DESC_PIXEL_LIMIT) : desc
41
117
  },
42
118
  social: {
43
- og_title: og['og:title'] || title,
44
- og_description: og['og:description'] || desc,
45
- og_image: og['og:image'],
46
- twitter_card: twitter['twitter:card'] || 'summary_large_image'
119
+ og_title: og_title || title,
120
+ og_description: og_description || desc,
121
+ og_image: og_image,
122
+ twitter_card: twitter_card || 'summary_large_image'
47
123
  }
48
124
  }
49
125
  end
50
126
 
51
- private
127
+ def truncate_to_pixels(str, limit)
128
+ return str if self.class.estimate_pixel_width(str) <= limit
129
+
130
+ chars = []
131
+ str.each_char do |c|
132
+ candidate = "#{chars.join}#{c}..."
133
+ break if self.class.estimate_pixel_width(candidate) > limit
134
+ chars << c
135
+ end
136
+ "#{chars.join}..."
137
+ end
52
138
 
53
139
  def format_breadcrumb(url_str)
54
140
  uri = URI.parse(url_str) rescue nil
@@ -26,18 +26,53 @@ module GSC
26
26
  urls = urls.first(@options[:limit]) if @options[:limit] && @options[:limit] > 0
27
27
 
28
28
  total = urls.size
29
- urls.each_with_index do |url, idx|
30
- progress_block.call(url, idx + 1, total) if block_given?
31
-
32
- analyzer = PageAnalyzer.new(url)
33
- page_data = analyzer.fetch_and_analyze(
34
- check_links: @options[:check_links] || false,
35
- gsc_api: @options[:gsc_api],
36
- active_domain: @options[:active_domain]
37
- )
38
-
39
- @results << page_data
40
- categorize_page_issues(page_data)
29
+ concurrency = (@options[:concurrency] || 5).to_i
30
+ concurrency = 1 if concurrency < 1
31
+ concurrency = [concurrency, 20].min
32
+ concurrency = [concurrency, total].min if total > 0
33
+
34
+ if concurrency <= 1 || total <= 1
35
+ urls.each_with_index do |url, idx|
36
+ progress_block.call(url, idx + 1, total) if block_given?
37
+
38
+ page_data = safe_analyze_page(url)
39
+ @results << page_data
40
+ categorize_page_issues(page_data)
41
+ end
42
+ else
43
+ require 'thread'
44
+ queue = Queue.new
45
+ urls.each_with_index { |url, idx| queue << [url, idx] }
46
+
47
+ indexed_results = []
48
+ mutex = Mutex.new
49
+ completed_count = 0
50
+
51
+ workers = Array.new(concurrency) do
52
+ Thread.new do
53
+ loop do
54
+ item = begin
55
+ queue.pop(true)
56
+ rescue ThreadError
57
+ nil
58
+ end
59
+ break unless item
60
+
61
+ url, idx = item
62
+ page_data = safe_analyze_page(url)
63
+
64
+ mutex.synchronize do
65
+ completed_count += 1
66
+ indexed_results << [idx, page_data]
67
+ categorize_page_issues(page_data)
68
+ progress_block.call(url, completed_count, total) if block_given?
69
+ end
70
+ end
71
+ end
72
+ end
73
+
74
+ workers.each(&:join)
75
+ @results = indexed_results.sort_by(&:first).map(&:last)
41
76
  end
42
77
 
43
78
  aggregate_summary
@@ -157,7 +192,7 @@ module GSC
157
192
  end
158
193
 
159
194
  def aggregate_summary
160
- critical_errors = @broken_links.size + @results.count { |r| r[:indexability][:noindex] }
195
+ critical_errors = @broken_links.size + @results.count { |r| r.dig(:indexability, :noindex) || false }
161
196
  total_issues = critical_errors + @missing_alts.size + @heading_issues.size + @title_issues.size + @canonical_issues.size
162
197
 
163
198
  {
@@ -174,6 +209,28 @@ module GSC
174
209
 
175
210
  private
176
211
 
212
+ def safe_analyze_page(url)
213
+ analyzer = PageAnalyzer.new(url)
214
+ analyzer.fetch_and_analyze(
215
+ check_links: @options[:check_links] || false,
216
+ gsc_api: @options[:gsc_api],
217
+ active_domain: @options[:active_domain]
218
+ )
219
+ rescue StandardError => e
220
+ {
221
+ url: url,
222
+ http_status: 0,
223
+ response_time_ms: 0,
224
+ title: { text: '', length: 0, pixel_est: 0.0, ok: false },
225
+ meta_description: { text: '', length: 0, ok: false },
226
+ canonical: { url: nil, self_referencing: false },
227
+ headings: { count: 0, h1_count: 0, score: 0, grade: 'F', violations: [], list: [] },
228
+ images: { total: 0, missing_alt_count: 0, missing_alt: [] },
229
+ links: { total: 0, internal_count: 0, external_count: 0, internal: [], external: [] },
230
+ issues: [{ level: :error, type: :network, message: "Crawl failure: #{e.message}" }]
231
+ }
232
+ end
233
+
177
234
  def discover_urls(target)
178
235
  if target.end_with?('.xml') || target.include?('sitemap')
179
236
  SitemapLoader.load_urls(target)
@@ -217,18 +274,42 @@ module GSC
217
274
  end
218
275
 
219
276
  # Headings
220
- h1_count = data.dig(:headings, :h1_count) || 0
277
+ headings_obj = data[:headings]
278
+ h1_count = if headings_obj.is_a?(Hash)
279
+ headings_obj[:h1_count] || headings_obj['h1_count'] || 0
280
+ elsif data[:h1].is_a?(Array)
281
+ data[:h1].size
282
+ else
283
+ 0
284
+ end
285
+
221
286
  if h1_count == 0
222
287
  @heading_issues << { page_url: page_url, issue: "Missing <h1> tag (0 found)" }
223
288
  elsif h1_count > 1
224
289
  @heading_issues << { page_url: page_url, issue: "Multiple <h1> tags (#{h1_count} found)" }
225
290
  end
226
291
 
227
- # Title & Meta
228
- title_chars = data.dig(:title, :length) || 0
229
- meta_chars = data.dig(:meta_description, :length) || 0
230
- title_text = data.dig(:title, :text) || ''
231
- meta_text = data.dig(:meta_description, :text) || ''
292
+ # Title
293
+ title_obj = data[:title]
294
+ title_text, title_chars = if title_obj.is_a?(Hash)
295
+ t = (title_obj[:text] || title_obj['text']).to_s
296
+ [t, (title_obj[:length] || title_obj['length'] || t.length).to_i]
297
+ elsif title_obj.is_a?(String)
298
+ [title_obj, title_obj.length]
299
+ else
300
+ ['', 0]
301
+ end
302
+
303
+ # Meta Description
304
+ meta_obj = data[:meta_description]
305
+ meta_text, meta_chars = if meta_obj.is_a?(Hash)
306
+ m = (meta_obj[:text] || meta_obj['text']).to_s
307
+ [m, (meta_obj[:length] || meta_obj['length'] || m.length).to_i]
308
+ elsif meta_obj.is_a?(String)
309
+ [meta_obj, meta_obj.length]
310
+ else
311
+ ['', 0]
312
+ end
232
313
 
233
314
  flaws = []
234
315
  flaws << "Title > 60 chars" if title_chars > 60
@@ -249,10 +330,21 @@ module GSC
249
330
  end
250
331
 
251
332
  # Canonical
252
- if data.dig(:canonical, :url) && !data.dig(:canonical, :self_referencing)
333
+ canon_obj = data[:canonical]
334
+ canon_url = nil
335
+ self_ref = false
336
+ if canon_obj.is_a?(Hash)
337
+ canon_url = canon_obj[:url] || canon_obj['url']
338
+ self_ref = canon_obj[:self_referencing] || false
339
+ elsif canon_obj.is_a?(String)
340
+ canon_url = canon_obj
341
+ self_ref = (canon_url == page_url)
342
+ end
343
+
344
+ if canon_url && !self_ref
253
345
  @canonical_issues << {
254
346
  page_url: page_url,
255
- canonical_url: data[:canonical][:url]
347
+ canonical_url: canon_url
256
348
  }
257
349
  end
258
350
  end