gsc-cli 2.0.2 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/AUTH.md +205 -0
- data/FUNDING.md +120 -0
- data/README.md +463 -299
- data/bin/gsc +29158 -4921
- data/dist/gsc +29158 -4921
- data/lib/gsc/aio_hunter.rb +343 -0
- data/lib/gsc/answer_synthesizer.rb +157 -0
- data/lib/gsc/api.rb +53 -1
- data/lib/gsc/auth.rb +26 -0
- data/lib/gsc/backlinks_manager.rb +96 -0
- data/lib/gsc/brand_segmenter.rb +140 -0
- data/lib/gsc/cache_manager.rb +806 -0
- data/lib/gsc/cannibalization_analyzer.rb +141 -0
- data/lib/gsc/canonical_chains.rb +367 -0
- data/lib/gsc/citation_simulator.rb +339 -0
- data/lib/gsc/cli/aio_hunter.rb +154 -0
- data/lib/gsc/cli/analytics.rb +788 -0
- data/lib/gsc/cli/audit.rb +1976 -0
- data/lib/gsc/cli/base.rb +384 -0
- data/lib/gsc/cli/cache.rb +266 -0
- data/lib/gsc/cli/canonical.rb +223 -0
- data/lib/gsc/cli/citation_simulator.rb +152 -0
- data/lib/gsc/cli/dashboard.rb +354 -0
- data/lib/gsc/cli/doctor.rb +129 -0
- data/lib/gsc/cli/eeat.rb +125 -0
- data/lib/gsc/cli/ga4.rb +852 -0
- data/lib/gsc/cli/growth.rb +650 -0
- data/lib/gsc/cli/hreflang.rb +164 -0
- data/lib/gsc/cli/image_seo.rb +162 -0
- data/lib/gsc/cli/indexing.rb +458 -0
- data/lib/gsc/cli/intent_shift.rb +125 -0
- data/lib/gsc/cli/keyword_value.rb +134 -0
- data/lib/gsc/cli/keywords.rb +795 -0
- data/lib/gsc/cli/landing_roi.rb +308 -0
- data/lib/gsc/cli/low_ctr.rb +213 -0
- data/lib/gsc/cli/mobile_parity.rb +150 -0
- data/lib/gsc/cli/report.rb +100 -0
- data/lib/gsc/cli/rich_results.rb +172 -0
- data/lib/gsc/cli/schema_generate.rb +149 -0
- data/lib/gsc/cli/seasonal.rb +232 -0
- data/lib/gsc/cli/security.rb +153 -0
- data/lib/gsc/cli/setup.rb +1291 -0
- data/lib/gsc/cli/sitemap_tree.rb +143 -0
- data/lib/gsc/cli/skill_pack.rb +62 -0
- data/lib/gsc/cli/soft_404.rb +199 -0
- data/lib/gsc/cli/sparkline.rb +227 -0
- data/lib/gsc/cli/watchdog.rb +150 -0
- data/lib/gsc/cli/zombie_purger.rb +208 -0
- data/lib/gsc/cli.rb +706 -5111
- data/lib/gsc/cli_advanced.rb +1513 -0
- data/lib/gsc/client.rb +17 -2
- data/lib/gsc/color.rb +16 -1
- data/lib/gsc/command_registry.rb +47 -9
- data/lib/gsc/config.rb +11 -2
- data/lib/gsc/content_gap.rb +112 -0
- data/lib/gsc/ctr_curve.rb +115 -0
- data/lib/gsc/decay_predictor.rb +322 -0
- data/lib/gsc/doctor.rb +434 -0
- data/lib/gsc/eeat_auditor.rb +428 -0
- data/lib/gsc/entity_auditor.rb +229 -0
- data/lib/gsc/firewall_scanner.rb +733 -0
- data/lib/gsc/geo_auditor.rb +368 -0
- data/lib/gsc/google_suggest.rb +109 -0
- data/lib/gsc/google_trends.rb +8 -1
- data/lib/gsc/heading_validator.rb +283 -0
- data/lib/gsc/hreflang_validator.rb +412 -0
- data/lib/gsc/image_seo.rb +286 -0
- data/lib/gsc/indexing_queue.rb +179 -0
- data/lib/gsc/indexnow.rb +93 -0
- data/lib/gsc/intent_shift.rb +188 -0
- data/lib/gsc/internal_links.rb +249 -0
- data/lib/gsc/keyword_value.rb +191 -0
- data/lib/gsc/landing_roi.rb +195 -0
- data/lib/gsc/llms_generator.rb +425 -0
- data/lib/gsc/low_ctr_rewriter.rb +408 -0
- data/lib/gsc/mobile_parity.rb +222 -0
- data/lib/gsc/network_tracer.rb +93 -0
- data/lib/gsc/open_page_rank.rb +72 -0
- data/lib/gsc/page_analyzer.rb +47 -7
- data/lib/gsc/page_comparator.rb +108 -0
- data/lib/gsc/page_speed.rb +110 -0
- data/lib/gsc/prompts.rb +38 -29
- data/lib/gsc/questions_harvester.rb +178 -0
- data/lib/gsc/report_generator.rb +461 -0
- data/lib/gsc/rich_results.rb +388 -0
- data/lib/gsc/robots_checker.rb +114 -0
- data/lib/gsc/schema_generator.rb +788 -0
- data/lib/gsc/schema_validator.rb +120 -0
- data/lib/gsc/seasonal_predictor.rb +381 -0
- data/lib/gsc/security_scanner.rb +496 -0
- data/lib/gsc/serp_feature_detector.rb +359 -0
- data/lib/gsc/serp_preview.rb +152 -0
- data/lib/gsc/site_crawler.rb +113 -21
- data/lib/gsc/sitemap_loader.rb +15 -4
- data/lib/gsc/sitemap_tree.rb +301 -0
- data/lib/gsc/skill_pack.rb +195 -0
- data/lib/gsc/soft_404_analyzer.rb +385 -0
- data/lib/gsc/sparkline.rb +171 -0
- data/lib/gsc/speed_correlator.rb +416 -0
- data/lib/gsc/striking_playbook.rb +190 -0
- data/lib/gsc/title_optimizer.rb +420 -0
- data/lib/gsc/vault.rb +260 -0
- data/lib/gsc/version.rb +1 -1
- data/lib/gsc/watchdog.rb +235 -0
- data/lib/gsc/zombie_purger.rb +366 -0
- data/lib/gsc.rb +144 -0
- metadata +91 -2
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'json'
|
|
5
|
+
require 'date'
|
|
6
|
+
|
|
7
|
+
module GSC
|
|
8
|
+
class IntentShift
|
|
9
|
+
TRANSACTIONAL_MODIFIERS = %w[
|
|
10
|
+
buy order purchase discount coupon price pricing cost cheap deal shop store subscription checkout hire
|
|
11
|
+
].freeze
|
|
12
|
+
|
|
13
|
+
COMMERCIAL_MODIFIERS = %w[
|
|
14
|
+
best top review reviews vs versus compare comparison alternative alternatives recommended software tool platform
|
|
15
|
+
].freeze
|
|
16
|
+
|
|
17
|
+
INFORMATIONAL_MODIFIERS = %w[
|
|
18
|
+
how what why when where who guide tutorial tips steps learn ideas strategy example examples template explain
|
|
19
|
+
].freeze
|
|
20
|
+
|
|
21
|
+
NAVIGATIONAL_MODIFIERS = %w[
|
|
22
|
+
login log-in signin sign-in portal account dashboard support helpdesk download
|
|
23
|
+
].freeze
|
|
24
|
+
|
|
25
|
+
attr_reader :options, :api, :domain
|
|
26
|
+
|
|
27
|
+
def initialize(options = {}, api = nil, domain = nil)
|
|
28
|
+
@options = options
|
|
29
|
+
@api = api
|
|
30
|
+
@domain = domain.to_s.strip
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
def self.analyze(options = {}, api = nil, domain = nil)
|
|
34
|
+
new(options, api, domain).analyze
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
def analyze
|
|
38
|
+
rows = collect_rows
|
|
39
|
+
shifts = detect_intent_shifts(rows)
|
|
40
|
+
summarize_portfolio(shifts)
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
def self.classify_query(query, brand = nil)
|
|
44
|
+
q = query.to_s.downcase.strip
|
|
45
|
+
|
|
46
|
+
if brand && !brand.empty? && q.include?(brand.downcase)
|
|
47
|
+
return :navigational if NAVIGATIONAL_MODIFIERS.any? { |m| q.include?(m) }
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
return :navigational if NAVIGATIONAL_MODIFIERS.any? { |m| q.match?(/\b#{Regexp.escape(m)}\b/) }
|
|
51
|
+
return :transactional if TRANSACTIONAL_MODIFIERS.any? { |m| q.match?(/\b#{Regexp.escape(m)}\b/) }
|
|
52
|
+
return :commercial if COMMERCIAL_MODIFIERS.any? { |m| q.match?(/\b#{Regexp.escape(m)}\b/) }
|
|
53
|
+
return :informational if INFORMATIONAL_MODIFIERS.any? { |m| q.match?(/\b#{Regexp.escape(m)}\b/) }
|
|
54
|
+
|
|
55
|
+
:informational
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
def self.classify_page(url)
|
|
59
|
+
u = url.to_s.downcase
|
|
60
|
+
if u =~ %r{/(?:cart|checkout|pricing|buy|products?|shop|store|orders?)(?:/|$|\?|#)}
|
|
61
|
+
:transactional
|
|
62
|
+
elsif u =~ %r{/(?:blog|guides?|tutorials?|learn|how-to|articles?|docs|knowledge-base)(?:/|$|\?|#)}
|
|
63
|
+
:informational
|
|
64
|
+
elsif u =~ %r{/(?:comparison|compare|vs|alternatives?|best|reviews?)(?:/|$|\?|#)}
|
|
65
|
+
:commercial
|
|
66
|
+
elsif u =~ %r{/(?:login|portal|accounts?|dashboard|signin)(?:/|$|\?|#)}
|
|
67
|
+
:navigational
|
|
68
|
+
else
|
|
69
|
+
:hybrid
|
|
70
|
+
end
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
private
|
|
74
|
+
|
|
75
|
+
def collect_rows
|
|
76
|
+
if @api
|
|
77
|
+
begin
|
|
78
|
+
days = (@options[:days] || 28).to_i
|
|
79
|
+
end_date = (Date.today - 2).strftime('%Y-%m-%d')
|
|
80
|
+
start_date = (Date.today - 2 - days).strftime('%Y-%m-%d')
|
|
81
|
+
res = @api.search_analytics(@domain, start_date: start_date, end_date: end_date, dimensions: %w[query page])
|
|
82
|
+
raw_rows = res['rows'] || []
|
|
83
|
+
if raw_rows.any?
|
|
84
|
+
return raw_rows.map do |r|
|
|
85
|
+
{
|
|
86
|
+
query: r['keys'][0],
|
|
87
|
+
page: r['keys'][1],
|
|
88
|
+
clicks: r['clicks'] || 0,
|
|
89
|
+
impressions: r['impressions'] || 0,
|
|
90
|
+
ctr: ((r['ctr'] || 0) * 100.0).round(2),
|
|
91
|
+
position: (r['position'] || 0).round(1)
|
|
92
|
+
}
|
|
93
|
+
end
|
|
94
|
+
end
|
|
95
|
+
rescue StandardError
|
|
96
|
+
# GSC query failed
|
|
97
|
+
end
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
[]
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
def detect_intent_shifts(rows)
|
|
104
|
+
brand = @options[:brand] || @domain.split('.').first
|
|
105
|
+
|
|
106
|
+
shifts = []
|
|
107
|
+
|
|
108
|
+
rows.each do |row|
|
|
109
|
+
q_intent = self.class.classify_query(row[:query], brand)
|
|
110
|
+
p_intent = self.class.classify_page(row[:page])
|
|
111
|
+
|
|
112
|
+
mismatch = false
|
|
113
|
+
risk_level = :none
|
|
114
|
+
diagnosis = nil
|
|
115
|
+
prescription = nil
|
|
116
|
+
|
|
117
|
+
# Scenario 1: Informational/Commercial query ranking on a purely Transactional page
|
|
118
|
+
if (q_intent == :informational || q_intent == :commercial) && p_intent == :transactional
|
|
119
|
+
mismatch = true
|
|
120
|
+
risk_level = row[:position] > 10.0 ? :high : :medium
|
|
121
|
+
diagnosis = "Google expects #{q_intent.to_s.upcase} research content, but ranking URL is a TRANSACTIONAL #{File.basename(row[:page])} page."
|
|
122
|
+
prescription = "Publish a dedicated #{q_intent} guide/comparison landing page to capture Top 3 SERP intent instead of sending users to checkout/product page."
|
|
123
|
+
|
|
124
|
+
# Scenario 2: Transactional query ranking on an Informational blog post
|
|
125
|
+
elsif q_intent == :transactional && p_intent == :informational
|
|
126
|
+
mismatch = true
|
|
127
|
+
risk_level = :medium
|
|
128
|
+
diagnosis = "Users have HIGH BUYING INTENT (#{row[:query]}), but are landing on an INFORMATIONAL blog article."
|
|
129
|
+
prescription = "Embed prominent 1-click checkout widgets, pricing tables, and product CTAs directly above the fold in this blog post."
|
|
130
|
+
|
|
131
|
+
# Scenario 3: Commercial comparison query landing on generic pricing page
|
|
132
|
+
elsif q_intent == :commercial && p_intent == :hybrid
|
|
133
|
+
mismatch = true
|
|
134
|
+
risk_level = :low
|
|
135
|
+
diagnosis = "Comparison query landing on generic page."
|
|
136
|
+
prescription = "Deploy a structured vs/comparison matrix table."
|
|
137
|
+
end
|
|
138
|
+
|
|
139
|
+
shifts << {
|
|
140
|
+
query: row[:query],
|
|
141
|
+
page: row[:page],
|
|
142
|
+
clicks: row[:clicks],
|
|
143
|
+
impressions: row[:impressions],
|
|
144
|
+
position: row[:position],
|
|
145
|
+
ctr: row[:ctr],
|
|
146
|
+
query_intent: q_intent,
|
|
147
|
+
page_intent: p_intent,
|
|
148
|
+
has_mismatch: mismatch,
|
|
149
|
+
risk_level: risk_level,
|
|
150
|
+
diagnosis: diagnosis,
|
|
151
|
+
prescription: prescription
|
|
152
|
+
}
|
|
153
|
+
end
|
|
154
|
+
|
|
155
|
+
shifts
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
def summarize_portfolio(shifts)
|
|
159
|
+
total = shifts.size
|
|
160
|
+
mismatched = shifts.select { |s| s[:has_mismatch] }
|
|
161
|
+
high_risk = shifts.select { |s| s[:risk_level] == :high }
|
|
162
|
+
|
|
163
|
+
intent_counts = {
|
|
164
|
+
informational: shifts.count { |s| s[:query_intent] == :informational },
|
|
165
|
+
transactional: shifts.count { |s| s[:query_intent] == :transactional },
|
|
166
|
+
commercial: shifts.count { |s| s[:query_intent] == :commercial },
|
|
167
|
+
navigational: shifts.count { |s| s[:query_intent] == :navigational }
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
volatility_score = ((mismatched.size.to_f / [total, 1].max) * 100.0).round(1)
|
|
171
|
+
|
|
172
|
+
prescriptions = mismatched.map { |m| m[:prescription] }.compact.uniq
|
|
173
|
+
grade = total.zero? ? 'N/A' : (volatility_score > 40.0 ? 'HIGH RISK' : (volatility_score > 20.0 ? 'MODERATE' : 'OPTIMAL'))
|
|
174
|
+
|
|
175
|
+
{
|
|
176
|
+
domain: @domain,
|
|
177
|
+
total_queries_analyzed: total,
|
|
178
|
+
intent_distribution: intent_counts,
|
|
179
|
+
mismatched_queries_count: mismatched.size,
|
|
180
|
+
high_risk_shifts_count: high_risk.size,
|
|
181
|
+
portfolio_volatility_pct: volatility_score,
|
|
182
|
+
risk_grade: grade,
|
|
183
|
+
prescriptions: prescriptions,
|
|
184
|
+
shifts: shifts
|
|
185
|
+
}
|
|
186
|
+
end
|
|
187
|
+
end
|
|
188
|
+
end
|
|
@@ -0,0 +1,249 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'uri'
|
|
5
|
+
require 'set'
|
|
6
|
+
require 'thread'
|
|
7
|
+
require 'time'
|
|
8
|
+
|
|
9
|
+
module GSC
|
|
10
|
+
class InternalLinks
|
|
11
|
+
attr_reader :base_url, :pages, :graph, :orphans, :depths, :concurrency
|
|
12
|
+
|
|
13
|
+
def initialize(base_url, limit: 50, concurrency: 5)
|
|
14
|
+
@base_url = base_url.to_s.strip
|
|
15
|
+
@base_url = "https://#{@base_url}" unless @base_url =~ %r{^https?://}
|
|
16
|
+
@base_uri = URI.parse(@base_url)
|
|
17
|
+
@limit = limit.to_i > 0 ? limit.to_i : 50
|
|
18
|
+
@concurrency = [[concurrency.to_i, 1].max, 20].min
|
|
19
|
+
@graph = Hash.new { |h, k| h[k] = Set.new } # target_url => Set of source_urls
|
|
20
|
+
@out_links = Hash.new { |h, k| h[k] = Set.new } # source_url => Set of target_urls
|
|
21
|
+
@page_titles = {} # url => String
|
|
22
|
+
@page_anchors = Hash.new { |h, k| h[k] = {} } # target_url => { source_url => anchor_text }
|
|
23
|
+
@all_discovered = Set.new
|
|
24
|
+
@mutex = Mutex.new
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def audit(sitemap_urls = nil, &progress_block)
|
|
28
|
+
start_time = Time.now
|
|
29
|
+
|
|
30
|
+
urls_to_crawl = if sitemap_urls && !sitemap_urls.empty?
|
|
31
|
+
sitemap_urls.first(@limit)
|
|
32
|
+
else
|
|
33
|
+
discover_urls
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
urls_to_crawl.each do |url|
|
|
37
|
+
@all_discovered << normalize_url(url)
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
# Multi-threaded concurrent crawling
|
|
41
|
+
crawl_urls_concurrently(urls_to_crawl, &progress_block)
|
|
42
|
+
|
|
43
|
+
crawl_duration = (Time.now - start_time).round(2)
|
|
44
|
+
|
|
45
|
+
# Calculate click depths via BFS from root
|
|
46
|
+
normalized_root = normalize_url(@base_url)
|
|
47
|
+
depths = calculate_depths(normalized_root)
|
|
48
|
+
|
|
49
|
+
# Top hub pages (highest out-links and in-links)
|
|
50
|
+
top_hubs = top_linked_pages(10)
|
|
51
|
+
hub_urls = top_hubs.map { |h| h[:url] }.reject { |u| u == normalized_root }
|
|
52
|
+
|
|
53
|
+
# Calculate Orphans and Weak Pages
|
|
54
|
+
orphans = []
|
|
55
|
+
orphan_details = []
|
|
56
|
+
weak_pages = []
|
|
57
|
+
deep_pages = []
|
|
58
|
+
|
|
59
|
+
@all_discovered.each do |url|
|
|
60
|
+
next if url == normalized_root
|
|
61
|
+
|
|
62
|
+
in_degree = @graph[url].size
|
|
63
|
+
depth = depths[url]
|
|
64
|
+
|
|
65
|
+
if depth && depth >= 4
|
|
66
|
+
deep_pages << { url: url, depth: depth, in_degree: in_degree }
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
if in_degree == 0
|
|
70
|
+
orphans << url
|
|
71
|
+
orphan_details << {
|
|
72
|
+
url: url,
|
|
73
|
+
depth: depth || 'Unreachable via internal links (∞)',
|
|
74
|
+
suggested_rescues: generate_rescue_suggestions(url, hub_urls, normalized_root)
|
|
75
|
+
}
|
|
76
|
+
elsif in_degree == 1
|
|
77
|
+
source = @graph[url].first
|
|
78
|
+
anchor = @page_anchors.dig(url, source) || 'View page'
|
|
79
|
+
weak_pages << {
|
|
80
|
+
url: url,
|
|
81
|
+
source: source,
|
|
82
|
+
anchor: anchor,
|
|
83
|
+
depth: depth,
|
|
84
|
+
suggested_rescues: generate_rescue_suggestions(url, hub_urls, normalized_root)
|
|
85
|
+
}
|
|
86
|
+
end
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
# Link Equity Health Score (0–100)
|
|
90
|
+
health_score = [100 - (orphans.size * 12) - (weak_pages.size * 3) - (deep_pages.size * 4), 0].max
|
|
91
|
+
health_grade = case health_score
|
|
92
|
+
when 90..100 then 'A'
|
|
93
|
+
when 80..89 then 'B'
|
|
94
|
+
when 70..79 then 'C'
|
|
95
|
+
when 60..69 then 'D'
|
|
96
|
+
else 'F'
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
# Depth distribution
|
|
100
|
+
depth_dist = Hash.new(0)
|
|
101
|
+
depths.each_value { |d| depth_dist[d] += 1 }
|
|
102
|
+
|
|
103
|
+
{
|
|
104
|
+
base_url: @base_url,
|
|
105
|
+
total_pages: @all_discovered.size,
|
|
106
|
+
crawl_duration_s: crawl_duration,
|
|
107
|
+
health_score: health_score,
|
|
108
|
+
health_grade: health_grade,
|
|
109
|
+
orphans: orphans,
|
|
110
|
+
orphan_details: orphan_details,
|
|
111
|
+
weak_pages: weak_pages,
|
|
112
|
+
deep_pages: deep_pages,
|
|
113
|
+
top_linked: top_hubs,
|
|
114
|
+
depths: depths,
|
|
115
|
+
depth_distribution: depth_dist.sort.to_h
|
|
116
|
+
}
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
private
|
|
120
|
+
|
|
121
|
+
def crawl_urls_concurrently(urls, &_block)
|
|
122
|
+
queue = Queue.new
|
|
123
|
+
urls.each { |u| queue << u }
|
|
124
|
+
|
|
125
|
+
workers = (1..@concurrency).map do
|
|
126
|
+
Thread.new do
|
|
127
|
+
while !queue.empty? && (url = queue.pop(true) rescue nil)
|
|
128
|
+
crawl_single_page(url)
|
|
129
|
+
end
|
|
130
|
+
end
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
workers.each(&:join)
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
def crawl_single_page(url)
|
|
137
|
+
pa = GSC::PageAnalyzer.new(url)
|
|
138
|
+
pa.load_content!
|
|
139
|
+
dom = pa.analyze_dom
|
|
140
|
+
return unless dom
|
|
141
|
+
|
|
142
|
+
norm_source = normalize_url(url)
|
|
143
|
+
title = dom.dig(:title, :text).to_s.strip
|
|
144
|
+
links = dom.dig(:links, :all) || []
|
|
145
|
+
|
|
146
|
+
@mutex.synchronize do
|
|
147
|
+
@page_titles[norm_source] = title unless title.empty?
|
|
148
|
+
end
|
|
149
|
+
|
|
150
|
+
links.each do |link_obj|
|
|
151
|
+
href = link_obj[:href]
|
|
152
|
+
target_url = resolve_internal_url(href)
|
|
153
|
+
next unless target_url
|
|
154
|
+
|
|
155
|
+
norm_target = normalize_url(target_url)
|
|
156
|
+
next if norm_target == norm_source
|
|
157
|
+
|
|
158
|
+
anchor = link_obj[:anchor].to_s.strip
|
|
159
|
+
|
|
160
|
+
@mutex.synchronize do
|
|
161
|
+
@graph[norm_target] << norm_source
|
|
162
|
+
@out_links[norm_source] << norm_target
|
|
163
|
+
@page_anchors[norm_target][norm_source] = anchor unless anchor.empty?
|
|
164
|
+
end
|
|
165
|
+
end
|
|
166
|
+
rescue StandardError
|
|
167
|
+
# resilient worker execution
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
def discover_urls
|
|
171
|
+
urls = GSC::SitemapLoader.resolve_urls(@base_url, @base_url, quiet: true)
|
|
172
|
+
urls.empty? ? [@base_url] : urls.first(@limit)
|
|
173
|
+
rescue StandardError
|
|
174
|
+
[@base_url]
|
|
175
|
+
end
|
|
176
|
+
|
|
177
|
+
def resolve_internal_url(href)
|
|
178
|
+
return nil if href.nil? || href.strip.empty?
|
|
179
|
+
return nil if href =~ /^(mailto|tel|javascript|#):/i
|
|
180
|
+
|
|
181
|
+
uri = URI.join(@base_url, href) rescue nil
|
|
182
|
+
return nil unless uri && uri.scheme =~ /^https?$/i
|
|
183
|
+
return nil unless uri.host.downcase == @base_uri.host.downcase
|
|
184
|
+
|
|
185
|
+
uri.fragment = nil
|
|
186
|
+
uri.to_s
|
|
187
|
+
end
|
|
188
|
+
|
|
189
|
+
def normalize_url(url)
|
|
190
|
+
u = url.to_s.strip.sub(%r{/$}, '')
|
|
191
|
+
u.empty? ? @base_url : u
|
|
192
|
+
end
|
|
193
|
+
|
|
194
|
+
def calculate_depths(root_url)
|
|
195
|
+
depths = { root_url => 0 }
|
|
196
|
+
queue = [root_url]
|
|
197
|
+
|
|
198
|
+
until queue.empty?
|
|
199
|
+
curr = queue.shift
|
|
200
|
+
curr_depth = depths[curr]
|
|
201
|
+
|
|
202
|
+
(@out_links[curr] || []).each do |neighbor|
|
|
203
|
+
next if depths.key?(neighbor)
|
|
204
|
+
|
|
205
|
+
depths[neighbor] = curr_depth + 1
|
|
206
|
+
queue << neighbor
|
|
207
|
+
end
|
|
208
|
+
end
|
|
209
|
+
|
|
210
|
+
depths
|
|
211
|
+
end
|
|
212
|
+
|
|
213
|
+
def top_linked_pages(limit)
|
|
214
|
+
@graph.map do |url, sources|
|
|
215
|
+
{
|
|
216
|
+
url: url,
|
|
217
|
+
title: @page_titles[url] || '',
|
|
218
|
+
incoming_count: sources.size,
|
|
219
|
+
outgoing_count: (@out_links[url] || []).size
|
|
220
|
+
}
|
|
221
|
+
end.sort_by { |item| -item[:incoming_count] }.first(limit)
|
|
222
|
+
end
|
|
223
|
+
|
|
224
|
+
def generate_rescue_suggestions(orphan_url, hub_urls, root_url)
|
|
225
|
+
# Extract keyword tokens from slug
|
|
226
|
+
path = URI.parse(orphan_url).path.to_s rescue ''
|
|
227
|
+
slug = path.split('/').reject(&:empty?).last || ''
|
|
228
|
+
keyword = slug.gsub(/[-_]+/, ' ').strip
|
|
229
|
+
keyword_title = keyword.split.map(&:capitalize).join(' ')
|
|
230
|
+
|
|
231
|
+
# Pick 2 best rescue sources: 1 high-equity hub, 1 root/pillar
|
|
232
|
+
sources = []
|
|
233
|
+
if !hub_urls.empty?
|
|
234
|
+
# Pick first hub that isn't the orphan itself
|
|
235
|
+
hub_source = hub_urls.find { |h| h != orphan_url }
|
|
236
|
+
sources << hub_source if hub_source
|
|
237
|
+
end
|
|
238
|
+
sources << root_url unless sources.include?(root_url)
|
|
239
|
+
|
|
240
|
+
sources.first(2).map do |src|
|
|
241
|
+
{
|
|
242
|
+
source_url: src,
|
|
243
|
+
recommended_anchor: keyword.empty? ? 'Explore this guide' : keyword_title,
|
|
244
|
+
action: "Add in-content link from #{src} pointing to #{orphan_url} with anchor \"#{keyword.empty? ? 'Learn more' : keyword_title}\""
|
|
245
|
+
}
|
|
246
|
+
end
|
|
247
|
+
end
|
|
248
|
+
end
|
|
249
|
+
end
|
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'json'
|
|
5
|
+
require 'date'
|
|
6
|
+
require_relative 'ctr_curve'
|
|
7
|
+
require_relative 'intent_shift'
|
|
8
|
+
|
|
9
|
+
module GSC
|
|
10
|
+
class KeywordValue
|
|
11
|
+
DEFAULT_AOV = 75.0
|
|
12
|
+
DEFAULT_CONV_RATE = 0.025 # 2.5%
|
|
13
|
+
DEFAULT_MARGIN = 0.60 # 60% gross margin
|
|
14
|
+
|
|
15
|
+
attr_reader :options, :api, :domain, :aov, :conv_rate, :margin, :target_pos
|
|
16
|
+
|
|
17
|
+
def initialize(options = {}, api = nil, domain = nil)
|
|
18
|
+
@options = options
|
|
19
|
+
@api = api
|
|
20
|
+
@domain = domain.to_s.strip
|
|
21
|
+
@aov = (options[:aov] || DEFAULT_AOV).to_f
|
|
22
|
+
@conv_rate = (options[:conv_rate] || DEFAULT_CONV_RATE).to_f
|
|
23
|
+
@margin = (options[:margin] || DEFAULT_MARGIN).to_f
|
|
24
|
+
@target_pos = (options[:target_pos] || 3).to_i
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def self.analyze(options = {}, api = nil, domain = nil)
|
|
28
|
+
new(options, api, domain).analyze
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
def analyze
|
|
32
|
+
rows = fetch_rows
|
|
33
|
+
target_ctr = GSC::CtrCurve.benchmark_for(@target_pos)
|
|
34
|
+
|
|
35
|
+
analyzed_keywords = rows.map do |row|
|
|
36
|
+
analyze_row(row, target_ctr)
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
# Sort by monthly revenue upside descending
|
|
40
|
+
analyzed_keywords.sort_by! { |k| -k[:monthly_revenue_upside] }
|
|
41
|
+
|
|
42
|
+
synthesize_portfolio(analyzed_keywords)
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
private
|
|
46
|
+
|
|
47
|
+
def analyze_row(row, target_ctr)
|
|
48
|
+
query = row[:query]
|
|
49
|
+
imp = row[:impressions] || 0
|
|
50
|
+
clicks = row[:clicks] || 0
|
|
51
|
+
pos = (row[:position] || 100.0).round(1)
|
|
52
|
+
ctr = row[:ctr] || (imp > 0 ? ((clicks.to_f / imp) * 100.0).round(2) : 0.0)
|
|
53
|
+
|
|
54
|
+
# Classify intent to apply dynamic conversion multipliers
|
|
55
|
+
intent = GSC::IntentShift.classify_query(query)
|
|
56
|
+
intent_multiplier = case intent
|
|
57
|
+
when :transactional then 2.0
|
|
58
|
+
when :commercial then 1.2
|
|
59
|
+
when :navigational then 1.5
|
|
60
|
+
else 0.6 # informational
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
effective_conv_rate = (@conv_rate * intent_multiplier).round(4)
|
|
64
|
+
|
|
65
|
+
# Current realized monthly revenue
|
|
66
|
+
current_orders = (clicks * effective_conv_rate).round(1)
|
|
67
|
+
current_revenue = (current_orders * @aov).round(2)
|
|
68
|
+
current_gross_profit = (current_revenue * @margin).round(2)
|
|
69
|
+
|
|
70
|
+
# Projected revenue at target position (e.g. Top 3)
|
|
71
|
+
if pos <= @target_pos
|
|
72
|
+
projected_clicks = clicks
|
|
73
|
+
incremental_clicks = 0
|
|
74
|
+
potential_revenue = current_revenue
|
|
75
|
+
monthly_upside = 0.0
|
|
76
|
+
else
|
|
77
|
+
projected_clicks = ((target_ctr / 100.0) * imp).round
|
|
78
|
+
incremental_clicks = [projected_clicks - clicks, 0].max
|
|
79
|
+
potential_orders = (projected_clicks * effective_conv_rate).round(1)
|
|
80
|
+
potential_revenue = (potential_orders * @aov).round(2)
|
|
81
|
+
monthly_upside = [potential_revenue - current_revenue, 0.0].max.round(2)
|
|
82
|
+
end
|
|
83
|
+
annual_upside = (monthly_upside * 12).round(2)
|
|
84
|
+
|
|
85
|
+
# Value Index score (0-100)
|
|
86
|
+
# High score if: in striking distance (pos 4-15), high upside, high intent
|
|
87
|
+
pos_weight = if pos.between?(4.0, 10.0) then 1.0
|
|
88
|
+
elsif pos.between?(10.1, 20.0) then 0.8
|
|
89
|
+
elsif pos <= 3.0 then 0.3 # already top 3
|
|
90
|
+
else 0.4
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
raw_score = (pos_weight * 40) + ([monthly_upside / 50.0, 40].min) + (intent_multiplier * 10)
|
|
94
|
+
value_score = [[raw_score.round, 100].min, 1].max
|
|
95
|
+
|
|
96
|
+
action = if pos <= 3.0
|
|
97
|
+
"👑 Top 3 Defend: Protect ranking with fresh content updates & internal hub equity."
|
|
98
|
+
elsif pos.between?(4.0, 10.0)
|
|
99
|
+
"🚀 Page-1 Striking Distance: +$#{format_currency(monthly_upside)}/mo upside! Optimize title CTR & add FAQ schema."
|
|
100
|
+
elsif pos.between?(10.1, 20.0)
|
|
101
|
+
"🎯 Page-2 Striking Query: +$#{format_currency(monthly_upside)}/mo upside. Expand content depth & earn 2 backlinks."
|
|
102
|
+
else
|
|
103
|
+
"🌱 Deep Discovery: Expand keyword topic cluster to lift organic visibility."
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
{
|
|
107
|
+
query: query,
|
|
108
|
+
intent: intent,
|
|
109
|
+
position: pos,
|
|
110
|
+
impressions: imp,
|
|
111
|
+
clicks: clicks,
|
|
112
|
+
ctr: ctr,
|
|
113
|
+
effective_conv_rate: (effective_conv_rate * 100.0).round(2),
|
|
114
|
+
current_orders: current_orders,
|
|
115
|
+
current_monthly_revenue: current_revenue,
|
|
116
|
+
current_gross_profit: current_gross_profit,
|
|
117
|
+
potential_clicks: projected_clicks,
|
|
118
|
+
incremental_clicks: incremental_clicks,
|
|
119
|
+
potential_monthly_revenue: potential_revenue,
|
|
120
|
+
monthly_revenue_upside: monthly_upside,
|
|
121
|
+
annual_revenue_upside: annual_upside,
|
|
122
|
+
value_score: value_score,
|
|
123
|
+
action: action
|
|
124
|
+
}
|
|
125
|
+
end
|
|
126
|
+
|
|
127
|
+
def synthesize_portfolio(keywords)
|
|
128
|
+
total_current_rev = keywords.sum { |k| k[:current_monthly_revenue] }.round(2)
|
|
129
|
+
total_potential_rev = keywords.sum { |k| k[:potential_monthly_revenue] }.round(2)
|
|
130
|
+
total_monthly_upside = keywords.sum { |k| k[:monthly_revenue_upside] }.round(2)
|
|
131
|
+
total_annual_upside = (total_monthly_upside * 12).round(2)
|
|
132
|
+
|
|
133
|
+
striking_count = keywords.count { |k| k[:position].between?(4.0, 20.0) }
|
|
134
|
+
high_intent_count = keywords.count { |k| k[:intent] == :transactional || k[:intent] == :commercial }
|
|
135
|
+
|
|
136
|
+
{
|
|
137
|
+
domain: @domain,
|
|
138
|
+
parameters: {
|
|
139
|
+
aov: @aov,
|
|
140
|
+
conversion_rate_pct: (@conv_rate * 100.0).round(2),
|
|
141
|
+
profit_margin_pct: (@margin * 100.0).round(2),
|
|
142
|
+
target_position: @target_pos
|
|
143
|
+
},
|
|
144
|
+
total_keywords_analyzed: keywords.size,
|
|
145
|
+
striking_distance_keywords: striking_count,
|
|
146
|
+
high_commercial_intent_keywords: high_intent_count,
|
|
147
|
+
financials: {
|
|
148
|
+
current_monthly_revenue: total_current_rev,
|
|
149
|
+
current_annual_run_rate: (total_current_rev * 12).round(2),
|
|
150
|
+
potential_monthly_revenue: total_potential_rev,
|
|
151
|
+
unlocked_monthly_upside: total_monthly_upside,
|
|
152
|
+
unlocked_annual_pipeline_upside: total_annual_upside
|
|
153
|
+
},
|
|
154
|
+
keywords: keywords
|
|
155
|
+
}
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
def fetch_rows
|
|
159
|
+
if @api
|
|
160
|
+
begin
|
|
161
|
+
days = (@options[:days] || 28).to_i
|
|
162
|
+
end_date = (Date.today - 2).strftime('%Y-%m-%d')
|
|
163
|
+
start_date = (Date.today - 2 - days).strftime('%Y-%m-%d')
|
|
164
|
+
res = @api.search_analytics(@domain, start_date: start_date, end_date: end_date, dimensions: %w[query])
|
|
165
|
+
rows = res['rows'] || []
|
|
166
|
+
if rows.any?
|
|
167
|
+
return rows.first(25).map do |r|
|
|
168
|
+
{
|
|
169
|
+
query: r['keys'][0],
|
|
170
|
+
clicks: r['clicks'] || 0,
|
|
171
|
+
impressions: r['impressions'] || 0,
|
|
172
|
+
ctr: ((r['ctr'] || 0) * 100.0).round(2),
|
|
173
|
+
position: (r['position'] || 0).round(1)
|
|
174
|
+
}
|
|
175
|
+
end
|
|
176
|
+
end
|
|
177
|
+
rescue StandardError
|
|
178
|
+
# GSC query failed
|
|
179
|
+
end
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
[]
|
|
183
|
+
end
|
|
184
|
+
|
|
185
|
+
def format_currency(val)
|
|
186
|
+
parts = sprintf('%.2f', val.to_f).split('.')
|
|
187
|
+
parts[0] = parts[0].reverse.gsub(/(\d{3})(?=\d)/, '\\1,').reverse
|
|
188
|
+
parts.join('.')
|
|
189
|
+
end
|
|
190
|
+
end
|
|
191
|
+
end
|