gsc-cli 2.1.0 → 2.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/AUTH.md +4 -1
- data/README.md +448 -408
- data/bin/gsc +28067 -5661
- data/dist/gsc +29121 -5046
- data/lib/gsc/aio_hunter.rb +343 -0
- data/lib/gsc/answer_synthesizer.rb +157 -0
- data/lib/gsc/api.rb +53 -1
- data/lib/gsc/auth.rb +26 -0
- data/lib/gsc/brand_segmenter.rb +140 -0
- data/lib/gsc/cache_manager.rb +806 -0
- data/lib/gsc/cannibalization_analyzer.rb +141 -0
- data/lib/gsc/canonical_chains.rb +367 -0
- data/lib/gsc/citation_simulator.rb +339 -0
- data/lib/gsc/cli/aio_hunter.rb +154 -0
- data/lib/gsc/cli/analytics.rb +788 -0
- data/lib/gsc/cli/audit.rb +1976 -0
- data/lib/gsc/cli/base.rb +384 -0
- data/lib/gsc/cli/cache.rb +266 -0
- data/lib/gsc/cli/canonical.rb +223 -0
- data/lib/gsc/cli/citation_simulator.rb +152 -0
- data/lib/gsc/cli/dashboard.rb +354 -0
- data/lib/gsc/cli/doctor.rb +129 -0
- data/lib/gsc/cli/eeat.rb +125 -0
- data/lib/gsc/cli/ga4.rb +852 -0
- data/lib/gsc/cli/growth.rb +650 -0
- data/lib/gsc/cli/hreflang.rb +164 -0
- data/lib/gsc/cli/image_seo.rb +162 -0
- data/lib/gsc/cli/indexing.rb +458 -0
- data/lib/gsc/cli/intent_shift.rb +125 -0
- data/lib/gsc/cli/keyword_value.rb +134 -0
- data/lib/gsc/cli/keywords.rb +795 -0
- data/lib/gsc/cli/landing_roi.rb +308 -0
- data/lib/gsc/cli/low_ctr.rb +213 -0
- data/lib/gsc/cli/mobile_parity.rb +150 -0
- data/lib/gsc/cli/report.rb +100 -0
- data/lib/gsc/cli/rich_results.rb +172 -0
- data/lib/gsc/cli/schema_generate.rb +149 -0
- data/lib/gsc/cli/seasonal.rb +232 -0
- data/lib/gsc/cli/security.rb +153 -0
- data/lib/gsc/cli/setup.rb +1291 -0
- data/lib/gsc/cli/sitemap_tree.rb +143 -0
- data/lib/gsc/cli/skill_pack.rb +62 -0
- data/lib/gsc/cli/soft_404.rb +199 -0
- data/lib/gsc/cli/sparkline.rb +227 -0
- data/lib/gsc/cli/watchdog.rb +150 -0
- data/lib/gsc/cli/zombie_purger.rb +208 -0
- data/lib/gsc/cli.rb +707 -5265
- data/lib/gsc/cli_advanced.rb +987 -44
- data/lib/gsc/client.rb +17 -2
- data/lib/gsc/color.rb +16 -1
- data/lib/gsc/command_registry.rb +47 -9
- data/lib/gsc/config.rb +2 -2
- data/lib/gsc/ctr_curve.rb +115 -0
- data/lib/gsc/decay_predictor.rb +322 -0
- data/lib/gsc/doctor.rb +434 -0
- data/lib/gsc/eeat_auditor.rb +428 -0
- data/lib/gsc/entity_auditor.rb +229 -0
- data/lib/gsc/firewall_scanner.rb +733 -0
- data/lib/gsc/geo_auditor.rb +368 -0
- data/lib/gsc/google_trends.rb +8 -1
- data/lib/gsc/heading_validator.rb +283 -0
- data/lib/gsc/hreflang_validator.rb +412 -0
- data/lib/gsc/image_seo.rb +286 -0
- data/lib/gsc/indexing_queue.rb +179 -0
- data/lib/gsc/indexnow.rb +93 -0
- data/lib/gsc/intent_shift.rb +188 -0
- data/lib/gsc/internal_links.rb +153 -36
- data/lib/gsc/keyword_value.rb +191 -0
- data/lib/gsc/landing_roi.rb +195 -0
- data/lib/gsc/llms_generator.rb +343 -22
- data/lib/gsc/low_ctr_rewriter.rb +408 -0
- data/lib/gsc/mobile_parity.rb +222 -0
- data/lib/gsc/network_tracer.rb +8 -1
- data/lib/gsc/page_analyzer.rb +47 -7
- data/lib/gsc/prompts.rb +38 -29
- data/lib/gsc/questions_harvester.rb +178 -0
- data/lib/gsc/report_generator.rb +461 -0
- data/lib/gsc/rich_results.rb +388 -0
- data/lib/gsc/robots_checker.rb +46 -15
- data/lib/gsc/schema_generator.rb +788 -0
- data/lib/gsc/schema_validator.rb +36 -38
- data/lib/gsc/seasonal_predictor.rb +381 -0
- data/lib/gsc/security_scanner.rb +496 -0
- data/lib/gsc/serp_feature_detector.rb +359 -0
- data/lib/gsc/serp_preview.rb +108 -22
- data/lib/gsc/site_crawler.rb +113 -21
- data/lib/gsc/sitemap_loader.rb +15 -4
- data/lib/gsc/sitemap_tree.rb +301 -0
- data/lib/gsc/skill_pack.rb +195 -0
- data/lib/gsc/soft_404_analyzer.rb +385 -0
- data/lib/gsc/sparkline.rb +171 -0
- data/lib/gsc/speed_correlator.rb +416 -0
- data/lib/gsc/striking_playbook.rb +190 -0
- data/lib/gsc/title_optimizer.rb +420 -0
- data/lib/gsc/vault.rb +260 -0
- data/lib/gsc/version.rb +1 -1
- data/lib/gsc/watchdog.rb +235 -0
- data/lib/gsc/zombie_purger.rb +366 -0
- data/lib/gsc.rb +118 -0
- metadata +75 -1
|
@@ -0,0 +1,359 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'uri'
|
|
5
|
+
require 'net/http'
|
|
6
|
+
require 'json'
|
|
7
|
+
require 'time'
|
|
8
|
+
require_relative 'google_suggest' if File.exist?(File.expand_path('google_suggest.rb', __dir__))
|
|
9
|
+
|
|
10
|
+
module GSC
|
|
11
|
+
class SerpFeatureDetector
|
|
12
|
+
attr_reader :query, :options
|
|
13
|
+
|
|
14
|
+
def initialize(query, options = {})
|
|
15
|
+
@query = query.to_s.dup.force_encoding('UTF-8').scrub.strip
|
|
16
|
+
@options = options
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def detect
|
|
20
|
+
intent = classify_intent(@query)
|
|
21
|
+
paa_questions = extract_live_paa_questions(@query)
|
|
22
|
+
live_serp = sample_live_serp(@query)
|
|
23
|
+
|
|
24
|
+
ai_overview = detect_ai_overview(@query, intent)
|
|
25
|
+
featured_snippet = detect_featured_snippet(@query, intent)
|
|
26
|
+
local_pack = detect_local_pack(@query)
|
|
27
|
+
video_carousel = detect_video_carousel(@query, live_serp)
|
|
28
|
+
forum_discussions = detect_forum_discussions(@query, live_serp)
|
|
29
|
+
shopping_pack = detect_shopping_pack(@query, intent)
|
|
30
|
+
sitelinks = detect_sitelinks(@query, intent)
|
|
31
|
+
|
|
32
|
+
zero_click = calculate_zero_click_risk(
|
|
33
|
+
ai_overview: ai_overview,
|
|
34
|
+
featured_snippet: featured_snippet,
|
|
35
|
+
paa_count: paa_questions.size,
|
|
36
|
+
local_pack: local_pack,
|
|
37
|
+
shopping_pack: shopping_pack,
|
|
38
|
+
video_carousel: video_carousel
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
playbook = generate_capture_playbook(
|
|
42
|
+
query: @query,
|
|
43
|
+
intent: intent,
|
|
44
|
+
ai_overview: ai_overview,
|
|
45
|
+
featured_snippet: featured_snippet,
|
|
46
|
+
paa_questions: paa_questions
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
{
|
|
50
|
+
query: @query,
|
|
51
|
+
timestamp: Time.now.utc.iso8601,
|
|
52
|
+
intent: intent,
|
|
53
|
+
zero_click_threat: zero_click,
|
|
54
|
+
features: {
|
|
55
|
+
ai_overview: ai_overview,
|
|
56
|
+
featured_snippet: featured_snippet,
|
|
57
|
+
people_also_ask: {
|
|
58
|
+
detected: !paa_questions.empty?,
|
|
59
|
+
count: paa_questions.size,
|
|
60
|
+
questions: paa_questions
|
|
61
|
+
},
|
|
62
|
+
local_3_pack: local_pack,
|
|
63
|
+
video_carousel: video_carousel,
|
|
64
|
+
discussions_and_forums: forum_discussions,
|
|
65
|
+
shopping_pack: shopping_pack,
|
|
66
|
+
sitelinks: sitelinks
|
|
67
|
+
},
|
|
68
|
+
serp_sampling: live_serp,
|
|
69
|
+
playbook: playbook
|
|
70
|
+
}
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
private
|
|
74
|
+
|
|
75
|
+
def classify_intent(query)
|
|
76
|
+
q = query.downcase
|
|
77
|
+
|
|
78
|
+
if q =~ /\b(near me|in [a-z]+|city|store|repair|dentist|plumber|gym|shop|restaurant|agency)\b/i
|
|
79
|
+
{ primary: 'Local', secondary: 'Commercial', description: 'User seeking physical or regional local service' }
|
|
80
|
+
elsif q =~ /\b(buy|order|purchase|coupon|discount|deal|cheap|for sale|pricing|price|cost)\b/i
|
|
81
|
+
{ primary: 'Transactional', secondary: 'Commercial', description: 'User is ready to make an immediate purchase' }
|
|
82
|
+
elsif q =~ /\b(best|top|review|vs|versus|compare|alternative|alternatives|software|tool|app|guide)\b/i
|
|
83
|
+
{ primary: 'Commercial', secondary: 'Informational', description: 'User evaluating products or services before purchasing' }
|
|
84
|
+
elsif q =~ /\b(login|portal|website|account|dashboard|sign in|app\.|\.com)\b/i
|
|
85
|
+
{ primary: 'Navigational', secondary: 'Brand', description: 'User navigating to a specific brand destination' }
|
|
86
|
+
else
|
|
87
|
+
{ primary: 'Informational', secondary: 'Research', description: 'User seeking knowledge, definitions, answers or tutorials' }
|
|
88
|
+
end
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
def extract_live_paa_questions(query)
|
|
92
|
+
suggest = GSC::GoogleSuggest.new(query)
|
|
93
|
+
raw = suggest.fetch(questions: true)
|
|
94
|
+
extracted = []
|
|
95
|
+
|
|
96
|
+
if raw.is_a?(Hash)
|
|
97
|
+
raw.each_value do |items|
|
|
98
|
+
Array(items).each do |item|
|
|
99
|
+
term = item.is_a?(Hash) ? item[:term] : item.to_s
|
|
100
|
+
clean = term.to_s.strip
|
|
101
|
+
extracted << clean if !clean.empty? && !extracted.include?(clean)
|
|
102
|
+
end
|
|
103
|
+
end
|
|
104
|
+
elsif raw.is_a?(Array)
|
|
105
|
+
raw.each do |entry|
|
|
106
|
+
if entry.is_a?(Array) && entry[1].is_a?(Array)
|
|
107
|
+
entry[1].each do |item|
|
|
108
|
+
term = item.is_a?(Hash) ? item[:term] : item.to_s
|
|
109
|
+
clean = term.to_s.strip
|
|
110
|
+
extracted << clean if !clean.empty? && !extracted.include?(clean)
|
|
111
|
+
end
|
|
112
|
+
elsif entry.is_a?(Hash) && entry[:term]
|
|
113
|
+
term = entry[:term].to_s.strip
|
|
114
|
+
extracted << term if !term.empty? && !extracted.include?(term)
|
|
115
|
+
end
|
|
116
|
+
end
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
# Filter for relevance to query tokens
|
|
120
|
+
tokens = query.downcase.split(/\s+/).reject { |t| t.length < 3 }
|
|
121
|
+
relevant = if tokens.empty?
|
|
122
|
+
extracted
|
|
123
|
+
else
|
|
124
|
+
matched = extracted.select { |q| tokens.any? { |t| q.downcase.include?(t) } }
|
|
125
|
+
matched.empty? ? extracted : matched
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
relevant.first(8)
|
|
129
|
+
rescue StandardError
|
|
130
|
+
[]
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
def sample_live_serp(query)
|
|
134
|
+
results = []
|
|
135
|
+
uri = URI("https://www.bing.com/search?q=#{URI.encode_www_form_component(query)}&setlang=en-US&cc=US")
|
|
136
|
+
http = Net::HTTP.new(uri.host, uri.port)
|
|
137
|
+
http.use_ssl = true
|
|
138
|
+
http.open_timeout = 3
|
|
139
|
+
http.read_timeout = 3
|
|
140
|
+
req = Net::HTTP::Get.new(uri.request_uri)
|
|
141
|
+
req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36'
|
|
142
|
+
req['Accept-Language'] = 'en-US,en;q=0.9'
|
|
143
|
+
|
|
144
|
+
res = http.request(req)
|
|
145
|
+
if res.is_a?(Net::HTTPSuccess)
|
|
146
|
+
body = res.body.to_s.dup.force_encoding('UTF-8').scrub
|
|
147
|
+
matches = body.scan(/<li class="b_algo"[^>]*>.*?<h2><a[^>]*href="([^"]+)"[^>]*>(.*?)<\/a><\/h2>(?:.*?<p[^>]*>(.*?)<\/p>)?/im)
|
|
148
|
+
matches.first(6).each_with_index do |(url, raw_title, raw_snippet), idx|
|
|
149
|
+
title = raw_title.to_s.gsub(/<[^>]+>/, '').strip
|
|
150
|
+
snippet = raw_snippet.to_s.gsub(/<[^>]+>/, '').strip
|
|
151
|
+
domain = URI.parse(url).host rescue url
|
|
152
|
+
results << {
|
|
153
|
+
position: idx + 1,
|
|
154
|
+
title: title,
|
|
155
|
+
url: url,
|
|
156
|
+
domain: domain,
|
|
157
|
+
snippet: snippet[0..140]
|
|
158
|
+
}
|
|
159
|
+
end
|
|
160
|
+
end
|
|
161
|
+
results
|
|
162
|
+
rescue StandardError
|
|
163
|
+
[]
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
def detect_ai_overview(query, intent)
|
|
167
|
+
q = query.downcase
|
|
168
|
+
triggers = []
|
|
169
|
+
probability = 15
|
|
170
|
+
|
|
171
|
+
if intent[:primary] == 'Informational'
|
|
172
|
+
probability += 40
|
|
173
|
+
triggers << 'Informational search intent'
|
|
174
|
+
end
|
|
175
|
+
|
|
176
|
+
if q =~ /^(what is|how to|why does|how do|what are|difference between|steps to|guide to)\b/i
|
|
177
|
+
probability += 35
|
|
178
|
+
triggers << 'Interrogative / procedural query prefix'
|
|
179
|
+
elsif q.include?(' vs ') || q.include?(' versus ')
|
|
180
|
+
probability += 30
|
|
181
|
+
triggers << 'Direct comparative product/concept syntax'
|
|
182
|
+
end
|
|
183
|
+
|
|
184
|
+
if q.split(/\s+/).size >= 4
|
|
185
|
+
probability += 10
|
|
186
|
+
triggers << 'Long-tail semantic query depth'
|
|
187
|
+
end
|
|
188
|
+
|
|
189
|
+
probability = [probability, 95].min
|
|
190
|
+
detected = probability >= 60
|
|
191
|
+
|
|
192
|
+
{
|
|
193
|
+
detected: detected,
|
|
194
|
+
probability_pct: probability,
|
|
195
|
+
triggers: triggers,
|
|
196
|
+
displacement_pixels: detected ? 850 : 0,
|
|
197
|
+
impact: detected ? 'Severe: Google Gemini AI Overview displaces Top 1 organic result below the viewport fold.' : 'Low: Traditional organic listings occupy top positions.',
|
|
198
|
+
counter_strategy: 'Structure content with direct 40–50 word definition answer, bulleted takeaways, and FAQPage/Question JSON-LD markup.'
|
|
199
|
+
}
|
|
200
|
+
end
|
|
201
|
+
|
|
202
|
+
def detect_featured_snippet(query, intent)
|
|
203
|
+
q = query.downcase
|
|
204
|
+
format = :none
|
|
205
|
+
probability = 10
|
|
206
|
+
target_recipe = ''
|
|
207
|
+
|
|
208
|
+
if q =~ /^(how to|steps to|guide|tutorial|how do i)\b/i
|
|
209
|
+
format = :ordered_list
|
|
210
|
+
probability = 90
|
|
211
|
+
target_recipe = 'Use an ordered list (<ol><li>) with 5–8 actionable sequential steps directly under an H2 heading.'
|
|
212
|
+
elsif q.include?(' vs ') || q.include?('difference between') || q =~ /\b(pricing|cost|plans|tiers|specs)\b/i
|
|
213
|
+
format = :table
|
|
214
|
+
probability = 75
|
|
215
|
+
target_recipe = 'Provide a structured HTML <table> with clear column headers (<th>) comparing key attributes and metrics.'
|
|
216
|
+
elsif q =~ /\b(best|top|types of|examples of|list of)\b/i
|
|
217
|
+
format = :unordered_list
|
|
218
|
+
probability = 80
|
|
219
|
+
target_recipe = 'Use an unordered bullet list (<ul><li>) highlighting 5–10 items with bold introductory headers.'
|
|
220
|
+
elsif q =~ /^(what is|who is|definition of|meaning of|what does)\b/i || intent[:primary] == 'Informational'
|
|
221
|
+
format = :paragraph
|
|
222
|
+
probability = 85
|
|
223
|
+
target_recipe = 'Place target query in <h2>, followed immediately by a concise 42–55 word direct definition answer block.'
|
|
224
|
+
end
|
|
225
|
+
|
|
226
|
+
{
|
|
227
|
+
detected: format != :none,
|
|
228
|
+
target_format: format.to_s,
|
|
229
|
+
probability_pct: probability,
|
|
230
|
+
optimal_length: format == :paragraph ? '40–60 words (~280–350 chars)' : '5–8 structured items',
|
|
231
|
+
capture_prescription: target_recipe
|
|
232
|
+
}
|
|
233
|
+
end
|
|
234
|
+
|
|
235
|
+
def detect_local_pack(query)
|
|
236
|
+
q = query.downcase
|
|
237
|
+
is_local = q =~ /\b(near me|in [a-z]+|city|store|repair|dentist|plumber|gym|shop|restaurant|agency|services|near)\b/i
|
|
238
|
+
{
|
|
239
|
+
detected: !!is_local,
|
|
240
|
+
probability_pct: is_local ? 90 : 5,
|
|
241
|
+
notes: is_local ? 'Triggers Google Maps 3-Pack; local organic listings appear below maps.' : 'No local intent detected.'
|
|
242
|
+
}
|
|
243
|
+
end
|
|
244
|
+
|
|
245
|
+
def detect_video_carousel(query, live_serp)
|
|
246
|
+
q = query.downcase
|
|
247
|
+
has_video_kw = q =~ /\b(how to|tutorial|review|walkthrough|guide|setup|install|demo|video|diy|unboxing)\b/i
|
|
248
|
+
serp_has_youtube = live_serp.any? { |r| r[:url].to_s.include?('youtube.com') }
|
|
249
|
+
detected = !!(has_video_kw || serp_has_youtube)
|
|
250
|
+
|
|
251
|
+
{
|
|
252
|
+
detected: detected,
|
|
253
|
+
probability_pct: detected ? 80 : 15,
|
|
254
|
+
notes: detected ? 'Video carousel likely above or between organic results. YouTube videos dominate.' : 'Low video intent.'
|
|
255
|
+
}
|
|
256
|
+
end
|
|
257
|
+
|
|
258
|
+
def detect_forum_discussions(query, live_serp)
|
|
259
|
+
q = query.downcase
|
|
260
|
+
has_forum_kw = q =~ /\b(reddit|quora|worth it|opinions|review|anyone tried|experiences|recommendations|issues)\b/i
|
|
261
|
+
serp_has_forum = live_serp.any? { |r| r[:url].to_s =~ /(reddit\.com|quora\.com|community\.)/i }
|
|
262
|
+
detected = !!(has_forum_kw || serp_has_forum)
|
|
263
|
+
|
|
264
|
+
{
|
|
265
|
+
detected: detected,
|
|
266
|
+
probability_pct: detected ? 85 : 20,
|
|
267
|
+
notes: detected ? 'Google "Discussions and Forums" module active. Authentic first-person experiences prioritized.' : 'Standard commercial/informational listings dominate.'
|
|
268
|
+
}
|
|
269
|
+
end
|
|
270
|
+
|
|
271
|
+
def detect_shopping_pack(query, intent)
|
|
272
|
+
q = query.downcase
|
|
273
|
+
is_shopping = (intent[:primary] == 'Transactional') || (q =~ /\b(buy|price|cost|cheap|best|discount|sale|store|deals|shop|coupon)\b/i)
|
|
274
|
+
{
|
|
275
|
+
detected: !!is_shopping,
|
|
276
|
+
probability_pct: is_shopping ? 85 : 10,
|
|
277
|
+
notes: is_shopping ? 'Product grid / Google Shopping carousels occupy top above-the-fold position.' : 'Non-e-commerce SERP.'
|
|
278
|
+
}
|
|
279
|
+
end
|
|
280
|
+
|
|
281
|
+
def detect_sitelinks(query, intent)
|
|
282
|
+
is_brand = (intent[:primary] == 'Navigational') || (query.split(/\s+/).size <= 2 && query =~ /^[A-Z][a-zA-Z0-9]+$/)
|
|
283
|
+
{
|
|
284
|
+
detected: !!is_brand,
|
|
285
|
+
probability_pct: is_brand ? 90 : 15,
|
|
286
|
+
notes: is_brand ? 'Expanded 6-pack or 4-pack branded sitelinks trigger for primary domain.' : 'Standard single-line snippets.'
|
|
287
|
+
}
|
|
288
|
+
end
|
|
289
|
+
|
|
290
|
+
def calculate_zero_click_risk(ai_overview:, featured_snippet:, paa_count:, local_pack:, shopping_pack:, video_carousel:)
|
|
291
|
+
score = 10
|
|
292
|
+
score += 35 if ai_overview[:detected]
|
|
293
|
+
score += 25 if featured_snippet[:detected]
|
|
294
|
+
score += 15 if paa_count >= 3
|
|
295
|
+
score += 10 if local_pack[:detected]
|
|
296
|
+
score += 10 if shopping_pack[:detected]
|
|
297
|
+
score += 5 if video_carousel[:detected]
|
|
298
|
+
|
|
299
|
+
score = [score, 100].min
|
|
300
|
+
|
|
301
|
+
level = case score
|
|
302
|
+
when 0..30 then 'LOW'
|
|
303
|
+
when 31..55 then 'MODERATE'
|
|
304
|
+
when 56..79 then 'HIGH'
|
|
305
|
+
else 'SEVERE'
|
|
306
|
+
end
|
|
307
|
+
|
|
308
|
+
ctr_drop = case level
|
|
309
|
+
when 'LOW' then '-5% to -10%'
|
|
310
|
+
when 'MODERATE' then '-15% to -25%'
|
|
311
|
+
when 'HIGH' then '-30% to -45%'
|
|
312
|
+
when 'SEVERE' then '-50% to -65%'
|
|
313
|
+
end
|
|
314
|
+
|
|
315
|
+
{
|
|
316
|
+
score: score,
|
|
317
|
+
level: level,
|
|
318
|
+
estimated_organic_ctr_suppression: ctr_drop,
|
|
319
|
+
summary: "Zero-Click Threat is #{level} (#{score}/100). SERP features push standard organic rankings down."
|
|
320
|
+
}
|
|
321
|
+
end
|
|
322
|
+
|
|
323
|
+
def generate_capture_playbook(query:, intent:, ai_overview:, featured_snippet:, paa_questions:)
|
|
324
|
+
playbook = []
|
|
325
|
+
|
|
326
|
+
if featured_snippet[:detected]
|
|
327
|
+
playbook << {
|
|
328
|
+
target: "Featured Snippet (#{featured_snippet[:target_format].capitalize})",
|
|
329
|
+
action: featured_snippet[:capture_prescription],
|
|
330
|
+
priority: 'P1 - High Impact'
|
|
331
|
+
}
|
|
332
|
+
end
|
|
333
|
+
|
|
334
|
+
if ai_overview[:detected]
|
|
335
|
+
playbook << {
|
|
336
|
+
target: 'Google Gemini AI Overview Citation',
|
|
337
|
+
action: 'Provide authoritative factual statistics with citability markers (author bio, published date, quantitative percentages).',
|
|
338
|
+
priority: 'P1 - High Impact'
|
|
339
|
+
}
|
|
340
|
+
end
|
|
341
|
+
|
|
342
|
+
if !paa_questions.empty?
|
|
343
|
+
playbook << {
|
|
344
|
+
target: 'People Also Ask (PAA) Inclusion',
|
|
345
|
+
action: "Inject H3 headings answering top questions: \"#{paa_questions.first(3).join('", "')}\" using Schema.org FAQPage JSON-LD.",
|
|
346
|
+
priority: 'P2 - Traffic Expansion'
|
|
347
|
+
}
|
|
348
|
+
end
|
|
349
|
+
|
|
350
|
+
playbook << {
|
|
351
|
+
target: 'Entity Disambiguation',
|
|
352
|
+
action: 'Include sameAs Wikidata / Wikipedia references and Organization structured data to anchor topical authority.',
|
|
353
|
+
priority: 'P3 - Topical Authority'
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
playbook
|
|
357
|
+
end
|
|
358
|
+
end
|
|
359
|
+
end
|
data/lib/gsc/serp_preview.rb
CHANGED
|
@@ -1,54 +1,140 @@
|
|
|
1
1
|
# encoding: utf-8
|
|
2
2
|
# frozen_string_literal: true
|
|
3
3
|
|
|
4
|
+
require 'uri'
|
|
5
|
+
|
|
4
6
|
module GSC
|
|
5
7
|
class SerpPreview
|
|
6
|
-
attr_reader :url, :data
|
|
8
|
+
attr_reader :url, :title, :desc, :data
|
|
9
|
+
|
|
10
|
+
DESKTOP_TITLE_PIXEL_LIMIT = 580.0
|
|
11
|
+
MOBILE_TITLE_PIXEL_LIMIT = 650.0
|
|
12
|
+
DESKTOP_DESC_PIXEL_LIMIT = 960.0
|
|
13
|
+
MOBILE_DESC_PIXEL_LIMIT = 680.0
|
|
7
14
|
|
|
8
|
-
def initialize(url)
|
|
9
|
-
@url = url.to_s.strip
|
|
15
|
+
def initialize(url = nil, title: nil, desc: nil)
|
|
16
|
+
@url = url.to_s.dup.force_encoding('UTF-8').scrub.strip
|
|
17
|
+
@custom_title = title ? title.to_s.dup.force_encoding('UTF-8').scrub : nil
|
|
18
|
+
@custom_desc = desc ? desc.to_s.dup.force_encoding('UTF-8').scrub : nil
|
|
10
19
|
end
|
|
11
20
|
|
|
12
21
|
def generate
|
|
22
|
+
if @custom_title || @custom_desc
|
|
23
|
+
generate_custom
|
|
24
|
+
elsif !@url.empty? && @url.start_with?('http://', 'https://')
|
|
25
|
+
generate_from_url
|
|
26
|
+
else
|
|
27
|
+
generate_custom
|
|
28
|
+
end
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
def self.estimate_pixel_width(str)
|
|
32
|
+
width = 0.0
|
|
33
|
+
str.to_s.each_char do |ch|
|
|
34
|
+
width += case ch
|
|
35
|
+
when /[WMwm]/ then 13.5
|
|
36
|
+
when /[ABCDEFGHKNOPQRSTUVXYZ]/ then 10.5
|
|
37
|
+
when /[abcdeghnopqrsuvxyz]/ then 8.5
|
|
38
|
+
when /[fIjt1l\|\ \.\:\;]/ then 4.5
|
|
39
|
+
else 9.0
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
width.round(1)
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
private
|
|
46
|
+
|
|
47
|
+
def generate_from_url
|
|
13
48
|
pa = GSC::PageAnalyzer.new(@url)
|
|
14
49
|
@data = pa.fetch_and_analyze
|
|
15
50
|
|
|
16
|
-
title = @data.dig(:title, :text) || 'Untitled Page'
|
|
17
|
-
desc = @data.dig(:meta_description, :text) || 'No meta description found.'
|
|
51
|
+
title = @custom_title || @data.dig(:title, :text) || 'Untitled Page'
|
|
52
|
+
desc = @custom_desc || @data.dig(:meta_description, :text) || 'No meta description found.'
|
|
18
53
|
canonical = @data.dig(:canonical, :url) || @url
|
|
19
54
|
|
|
20
|
-
# SERP pixel calculation approximation:
|
|
21
|
-
# ~10px per character average for Arial 18px title
|
|
22
|
-
title_chars = title.length
|
|
23
|
-
is_truncated = title_chars > 60
|
|
24
|
-
|
|
25
|
-
desktop_title = is_truncated ? "#{title[0..56]}..." : title
|
|
26
|
-
desktop_snippet = desc.length > 155 ? "#{desc[0..152]}..." : desc
|
|
27
|
-
|
|
28
55
|
og = @data[:open_graph] || {}
|
|
29
56
|
twitter = @data[:twitter_card] || {}
|
|
30
57
|
|
|
31
|
-
|
|
58
|
+
build_result(
|
|
32
59
|
url: @url,
|
|
33
60
|
canonical: canonical,
|
|
34
61
|
title: title,
|
|
62
|
+
desc: desc,
|
|
63
|
+
og_title: og['og:title'],
|
|
64
|
+
og_description: og['og:description'],
|
|
65
|
+
og_image: og['og:image'],
|
|
66
|
+
twitter_card: twitter['twitter:card']
|
|
67
|
+
)
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
def generate_custom
|
|
71
|
+
url = @url.empty? ? (Config.default_domain ? "https://#{Config.default_domain}/page" : '/') : @url
|
|
72
|
+
title = @custom_title || 'Untitled Page'
|
|
73
|
+
desc = @custom_desc || ''
|
|
74
|
+
|
|
75
|
+
build_result(
|
|
76
|
+
url: url,
|
|
77
|
+
canonical: url,
|
|
78
|
+
title: title,
|
|
79
|
+
desc: desc
|
|
80
|
+
)
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
def build_result(url:, canonical:, title:, desc:, og_title: nil, og_description: nil, og_image: nil, twitter_card: nil)
|
|
84
|
+
t_px = self.class.estimate_pixel_width(title)
|
|
85
|
+
d_px = self.class.estimate_pixel_width(desc)
|
|
86
|
+
|
|
87
|
+
title_truncated = t_px > DESKTOP_TITLE_PIXEL_LIMIT
|
|
88
|
+
desc_truncated = d_px > DESKTOP_DESC_PIXEL_LIMIT
|
|
89
|
+
|
|
90
|
+
desktop_title = title_truncated ? truncate_to_pixels(title, DESKTOP_TITLE_PIXEL_LIMIT) : title
|
|
91
|
+
desktop_desc = desc_truncated ? truncate_to_pixels(desc, DESKTOP_DESC_PIXEL_LIMIT) : desc
|
|
92
|
+
|
|
93
|
+
{
|
|
94
|
+
url: url,
|
|
95
|
+
canonical: canonical,
|
|
96
|
+
title: title,
|
|
35
97
|
meta_description: desc,
|
|
36
|
-
|
|
98
|
+
metrics: {
|
|
99
|
+
title_chars: title.length,
|
|
100
|
+
title_pixel_est: t_px,
|
|
101
|
+
title_desktop_limit: DESKTOP_TITLE_PIXEL_LIMIT,
|
|
102
|
+
title_truncated: title_truncated,
|
|
103
|
+
desc_chars: desc.length,
|
|
104
|
+
desc_pixel_est: d_px,
|
|
105
|
+
desc_desktop_limit: DESKTOP_DESC_PIXEL_LIMIT,
|
|
106
|
+
desc_truncated: desc_truncated
|
|
107
|
+
},
|
|
37
108
|
desktop_serp: {
|
|
109
|
+
breadcrumb: format_breadcrumb(canonical),
|
|
38
110
|
title: desktop_title,
|
|
39
|
-
snippet:
|
|
40
|
-
|
|
111
|
+
snippet: desktop_desc
|
|
112
|
+
},
|
|
113
|
+
mobile_serp: {
|
|
114
|
+
breadcrumb: format_breadcrumb(canonical),
|
|
115
|
+
title: t_px > MOBILE_TITLE_PIXEL_LIMIT ? truncate_to_pixels(title, MOBILE_TITLE_PIXEL_LIMIT) : title,
|
|
116
|
+
snippet: d_px > MOBILE_DESC_PIXEL_LIMIT ? truncate_to_pixels(desc, MOBILE_DESC_PIXEL_LIMIT) : desc
|
|
41
117
|
},
|
|
42
118
|
social: {
|
|
43
|
-
og_title:
|
|
44
|
-
og_description:
|
|
45
|
-
og_image:
|
|
46
|
-
twitter_card:
|
|
119
|
+
og_title: og_title || title,
|
|
120
|
+
og_description: og_description || desc,
|
|
121
|
+
og_image: og_image,
|
|
122
|
+
twitter_card: twitter_card || 'summary_large_image'
|
|
47
123
|
}
|
|
48
124
|
}
|
|
49
125
|
end
|
|
50
126
|
|
|
51
|
-
|
|
127
|
+
def truncate_to_pixels(str, limit)
|
|
128
|
+
return str if self.class.estimate_pixel_width(str) <= limit
|
|
129
|
+
|
|
130
|
+
chars = []
|
|
131
|
+
str.each_char do |c|
|
|
132
|
+
candidate = "#{chars.join}#{c}..."
|
|
133
|
+
break if self.class.estimate_pixel_width(candidate) > limit
|
|
134
|
+
chars << c
|
|
135
|
+
end
|
|
136
|
+
"#{chars.join}..."
|
|
137
|
+
end
|
|
52
138
|
|
|
53
139
|
def format_breadcrumb(url_str)
|
|
54
140
|
uri = URI.parse(url_str) rescue nil
|
data/lib/gsc/site_crawler.rb
CHANGED
|
@@ -26,18 +26,53 @@ module GSC
|
|
|
26
26
|
urls = urls.first(@options[:limit]) if @options[:limit] && @options[:limit] > 0
|
|
27
27
|
|
|
28
28
|
total = urls.size
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
29
|
+
concurrency = (@options[:concurrency] || 5).to_i
|
|
30
|
+
concurrency = 1 if concurrency < 1
|
|
31
|
+
concurrency = [concurrency, 20].min
|
|
32
|
+
concurrency = [concurrency, total].min if total > 0
|
|
33
|
+
|
|
34
|
+
if concurrency <= 1 || total <= 1
|
|
35
|
+
urls.each_with_index do |url, idx|
|
|
36
|
+
progress_block.call(url, idx + 1, total) if block_given?
|
|
37
|
+
|
|
38
|
+
page_data = safe_analyze_page(url)
|
|
39
|
+
@results << page_data
|
|
40
|
+
categorize_page_issues(page_data)
|
|
41
|
+
end
|
|
42
|
+
else
|
|
43
|
+
require 'thread'
|
|
44
|
+
queue = Queue.new
|
|
45
|
+
urls.each_with_index { |url, idx| queue << [url, idx] }
|
|
46
|
+
|
|
47
|
+
indexed_results = []
|
|
48
|
+
mutex = Mutex.new
|
|
49
|
+
completed_count = 0
|
|
50
|
+
|
|
51
|
+
workers = Array.new(concurrency) do
|
|
52
|
+
Thread.new do
|
|
53
|
+
loop do
|
|
54
|
+
item = begin
|
|
55
|
+
queue.pop(true)
|
|
56
|
+
rescue ThreadError
|
|
57
|
+
nil
|
|
58
|
+
end
|
|
59
|
+
break unless item
|
|
60
|
+
|
|
61
|
+
url, idx = item
|
|
62
|
+
page_data = safe_analyze_page(url)
|
|
63
|
+
|
|
64
|
+
mutex.synchronize do
|
|
65
|
+
completed_count += 1
|
|
66
|
+
indexed_results << [idx, page_data]
|
|
67
|
+
categorize_page_issues(page_data)
|
|
68
|
+
progress_block.call(url, completed_count, total) if block_given?
|
|
69
|
+
end
|
|
70
|
+
end
|
|
71
|
+
end
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
workers.each(&:join)
|
|
75
|
+
@results = indexed_results.sort_by(&:first).map(&:last)
|
|
41
76
|
end
|
|
42
77
|
|
|
43
78
|
aggregate_summary
|
|
@@ -157,7 +192,7 @@ module GSC
|
|
|
157
192
|
end
|
|
158
193
|
|
|
159
194
|
def aggregate_summary
|
|
160
|
-
critical_errors = @broken_links.size + @results.count { |r| r
|
|
195
|
+
critical_errors = @broken_links.size + @results.count { |r| r.dig(:indexability, :noindex) || false }
|
|
161
196
|
total_issues = critical_errors + @missing_alts.size + @heading_issues.size + @title_issues.size + @canonical_issues.size
|
|
162
197
|
|
|
163
198
|
{
|
|
@@ -174,6 +209,28 @@ module GSC
|
|
|
174
209
|
|
|
175
210
|
private
|
|
176
211
|
|
|
212
|
+
def safe_analyze_page(url)
|
|
213
|
+
analyzer = PageAnalyzer.new(url)
|
|
214
|
+
analyzer.fetch_and_analyze(
|
|
215
|
+
check_links: @options[:check_links] || false,
|
|
216
|
+
gsc_api: @options[:gsc_api],
|
|
217
|
+
active_domain: @options[:active_domain]
|
|
218
|
+
)
|
|
219
|
+
rescue StandardError => e
|
|
220
|
+
{
|
|
221
|
+
url: url,
|
|
222
|
+
http_status: 0,
|
|
223
|
+
response_time_ms: 0,
|
|
224
|
+
title: { text: '', length: 0, pixel_est: 0.0, ok: false },
|
|
225
|
+
meta_description: { text: '', length: 0, ok: false },
|
|
226
|
+
canonical: { url: nil, self_referencing: false },
|
|
227
|
+
headings: { count: 0, h1_count: 0, score: 0, grade: 'F', violations: [], list: [] },
|
|
228
|
+
images: { total: 0, missing_alt_count: 0, missing_alt: [] },
|
|
229
|
+
links: { total: 0, internal_count: 0, external_count: 0, internal: [], external: [] },
|
|
230
|
+
issues: [{ level: :error, type: :network, message: "Crawl failure: #{e.message}" }]
|
|
231
|
+
}
|
|
232
|
+
end
|
|
233
|
+
|
|
177
234
|
def discover_urls(target)
|
|
178
235
|
if target.end_with?('.xml') || target.include?('sitemap')
|
|
179
236
|
SitemapLoader.load_urls(target)
|
|
@@ -217,18 +274,42 @@ module GSC
|
|
|
217
274
|
end
|
|
218
275
|
|
|
219
276
|
# Headings
|
|
220
|
-
|
|
277
|
+
headings_obj = data[:headings]
|
|
278
|
+
h1_count = if headings_obj.is_a?(Hash)
|
|
279
|
+
headings_obj[:h1_count] || headings_obj['h1_count'] || 0
|
|
280
|
+
elsif data[:h1].is_a?(Array)
|
|
281
|
+
data[:h1].size
|
|
282
|
+
else
|
|
283
|
+
0
|
|
284
|
+
end
|
|
285
|
+
|
|
221
286
|
if h1_count == 0
|
|
222
287
|
@heading_issues << { page_url: page_url, issue: "Missing <h1> tag (0 found)" }
|
|
223
288
|
elsif h1_count > 1
|
|
224
289
|
@heading_issues << { page_url: page_url, issue: "Multiple <h1> tags (#{h1_count} found)" }
|
|
225
290
|
end
|
|
226
291
|
|
|
227
|
-
# Title
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
292
|
+
# Title
|
|
293
|
+
title_obj = data[:title]
|
|
294
|
+
title_text, title_chars = if title_obj.is_a?(Hash)
|
|
295
|
+
t = (title_obj[:text] || title_obj['text']).to_s
|
|
296
|
+
[t, (title_obj[:length] || title_obj['length'] || t.length).to_i]
|
|
297
|
+
elsif title_obj.is_a?(String)
|
|
298
|
+
[title_obj, title_obj.length]
|
|
299
|
+
else
|
|
300
|
+
['', 0]
|
|
301
|
+
end
|
|
302
|
+
|
|
303
|
+
# Meta Description
|
|
304
|
+
meta_obj = data[:meta_description]
|
|
305
|
+
meta_text, meta_chars = if meta_obj.is_a?(Hash)
|
|
306
|
+
m = (meta_obj[:text] || meta_obj['text']).to_s
|
|
307
|
+
[m, (meta_obj[:length] || meta_obj['length'] || m.length).to_i]
|
|
308
|
+
elsif meta_obj.is_a?(String)
|
|
309
|
+
[meta_obj, meta_obj.length]
|
|
310
|
+
else
|
|
311
|
+
['', 0]
|
|
312
|
+
end
|
|
232
313
|
|
|
233
314
|
flaws = []
|
|
234
315
|
flaws << "Title > 60 chars" if title_chars > 60
|
|
@@ -249,10 +330,21 @@ module GSC
|
|
|
249
330
|
end
|
|
250
331
|
|
|
251
332
|
# Canonical
|
|
252
|
-
|
|
333
|
+
canon_obj = data[:canonical]
|
|
334
|
+
canon_url = nil
|
|
335
|
+
self_ref = false
|
|
336
|
+
if canon_obj.is_a?(Hash)
|
|
337
|
+
canon_url = canon_obj[:url] || canon_obj['url']
|
|
338
|
+
self_ref = canon_obj[:self_referencing] || false
|
|
339
|
+
elsif canon_obj.is_a?(String)
|
|
340
|
+
canon_url = canon_obj
|
|
341
|
+
self_ref = (canon_url == page_url)
|
|
342
|
+
end
|
|
343
|
+
|
|
344
|
+
if canon_url && !self_ref
|
|
253
345
|
@canonical_issues << {
|
|
254
346
|
page_url: page_url,
|
|
255
|
-
canonical_url:
|
|
347
|
+
canonical_url: canon_url
|
|
256
348
|
}
|
|
257
349
|
end
|
|
258
350
|
end
|