gsc-cli 2.1.0 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +410 -414
  3. data/bin/gsc +28067 -5661
  4. data/dist/gsc +29121 -5046
  5. data/lib/gsc/aio_hunter.rb +343 -0
  6. data/lib/gsc/answer_synthesizer.rb +157 -0
  7. data/lib/gsc/api.rb +53 -1
  8. data/lib/gsc/auth.rb +26 -0
  9. data/lib/gsc/brand_segmenter.rb +140 -0
  10. data/lib/gsc/cache_manager.rb +806 -0
  11. data/lib/gsc/cannibalization_analyzer.rb +141 -0
  12. data/lib/gsc/canonical_chains.rb +367 -0
  13. data/lib/gsc/citation_simulator.rb +339 -0
  14. data/lib/gsc/cli/aio_hunter.rb +154 -0
  15. data/lib/gsc/cli/analytics.rb +788 -0
  16. data/lib/gsc/cli/audit.rb +1976 -0
  17. data/lib/gsc/cli/base.rb +384 -0
  18. data/lib/gsc/cli/cache.rb +266 -0
  19. data/lib/gsc/cli/canonical.rb +223 -0
  20. data/lib/gsc/cli/citation_simulator.rb +152 -0
  21. data/lib/gsc/cli/dashboard.rb +354 -0
  22. data/lib/gsc/cli/doctor.rb +129 -0
  23. data/lib/gsc/cli/eeat.rb +125 -0
  24. data/lib/gsc/cli/ga4.rb +852 -0
  25. data/lib/gsc/cli/growth.rb +650 -0
  26. data/lib/gsc/cli/hreflang.rb +164 -0
  27. data/lib/gsc/cli/image_seo.rb +162 -0
  28. data/lib/gsc/cli/indexing.rb +458 -0
  29. data/lib/gsc/cli/intent_shift.rb +125 -0
  30. data/lib/gsc/cli/keyword_value.rb +134 -0
  31. data/lib/gsc/cli/keywords.rb +795 -0
  32. data/lib/gsc/cli/landing_roi.rb +308 -0
  33. data/lib/gsc/cli/low_ctr.rb +213 -0
  34. data/lib/gsc/cli/mobile_parity.rb +150 -0
  35. data/lib/gsc/cli/report.rb +100 -0
  36. data/lib/gsc/cli/rich_results.rb +172 -0
  37. data/lib/gsc/cli/schema_generate.rb +149 -0
  38. data/lib/gsc/cli/seasonal.rb +232 -0
  39. data/lib/gsc/cli/security.rb +153 -0
  40. data/lib/gsc/cli/setup.rb +1291 -0
  41. data/lib/gsc/cli/sitemap_tree.rb +143 -0
  42. data/lib/gsc/cli/skill_pack.rb +62 -0
  43. data/lib/gsc/cli/soft_404.rb +199 -0
  44. data/lib/gsc/cli/sparkline.rb +227 -0
  45. data/lib/gsc/cli/watchdog.rb +150 -0
  46. data/lib/gsc/cli/zombie_purger.rb +208 -0
  47. data/lib/gsc/cli.rb +707 -5265
  48. data/lib/gsc/cli_advanced.rb +987 -44
  49. data/lib/gsc/client.rb +17 -2
  50. data/lib/gsc/color.rb +16 -1
  51. data/lib/gsc/command_registry.rb +47 -9
  52. data/lib/gsc/config.rb +2 -2
  53. data/lib/gsc/ctr_curve.rb +115 -0
  54. data/lib/gsc/decay_predictor.rb +322 -0
  55. data/lib/gsc/doctor.rb +434 -0
  56. data/lib/gsc/eeat_auditor.rb +428 -0
  57. data/lib/gsc/entity_auditor.rb +229 -0
  58. data/lib/gsc/firewall_scanner.rb +733 -0
  59. data/lib/gsc/geo_auditor.rb +368 -0
  60. data/lib/gsc/google_trends.rb +8 -1
  61. data/lib/gsc/heading_validator.rb +283 -0
  62. data/lib/gsc/hreflang_validator.rb +412 -0
  63. data/lib/gsc/image_seo.rb +286 -0
  64. data/lib/gsc/indexing_queue.rb +179 -0
  65. data/lib/gsc/indexnow.rb +93 -0
  66. data/lib/gsc/intent_shift.rb +188 -0
  67. data/lib/gsc/internal_links.rb +153 -36
  68. data/lib/gsc/keyword_value.rb +191 -0
  69. data/lib/gsc/landing_roi.rb +195 -0
  70. data/lib/gsc/llms_generator.rb +343 -22
  71. data/lib/gsc/low_ctr_rewriter.rb +408 -0
  72. data/lib/gsc/mobile_parity.rb +222 -0
  73. data/lib/gsc/network_tracer.rb +8 -1
  74. data/lib/gsc/page_analyzer.rb +47 -7
  75. data/lib/gsc/prompts.rb +38 -29
  76. data/lib/gsc/questions_harvester.rb +178 -0
  77. data/lib/gsc/report_generator.rb +461 -0
  78. data/lib/gsc/rich_results.rb +388 -0
  79. data/lib/gsc/robots_checker.rb +46 -15
  80. data/lib/gsc/schema_generator.rb +788 -0
  81. data/lib/gsc/schema_validator.rb +36 -38
  82. data/lib/gsc/seasonal_predictor.rb +381 -0
  83. data/lib/gsc/security_scanner.rb +496 -0
  84. data/lib/gsc/serp_feature_detector.rb +359 -0
  85. data/lib/gsc/serp_preview.rb +108 -22
  86. data/lib/gsc/site_crawler.rb +113 -21
  87. data/lib/gsc/sitemap_loader.rb +15 -4
  88. data/lib/gsc/sitemap_tree.rb +301 -0
  89. data/lib/gsc/skill_pack.rb +195 -0
  90. data/lib/gsc/soft_404_analyzer.rb +385 -0
  91. data/lib/gsc/sparkline.rb +171 -0
  92. data/lib/gsc/speed_correlator.rb +416 -0
  93. data/lib/gsc/striking_playbook.rb +190 -0
  94. data/lib/gsc/title_optimizer.rb +420 -0
  95. data/lib/gsc/vault.rb +260 -0
  96. data/lib/gsc/version.rb +1 -1
  97. data/lib/gsc/watchdog.rb +235 -0
  98. data/lib/gsc/zombie_purger.rb +366 -0
  99. data/lib/gsc.rb +118 -0
  100. metadata +75 -1
@@ -3,22 +3,30 @@
3
3
 
4
4
  require 'uri'
5
5
  require 'set'
6
+ require 'thread'
7
+ require 'time'
6
8
 
7
9
  module GSC
8
10
  class InternalLinks
9
- attr_reader :base_url, :pages, :graph, :orphans, :depths
11
+ attr_reader :base_url, :pages, :graph, :orphans, :depths, :concurrency
10
12
 
11
- def initialize(base_url, limit: 50)
13
+ def initialize(base_url, limit: 50, concurrency: 5)
12
14
  @base_url = base_url.to_s.strip
13
15
  @base_url = "https://#{@base_url}" unless @base_url =~ %r{^https?://}
14
16
  @base_uri = URI.parse(@base_url)
15
- @limit = limit
16
- @graph = Hash.new { |h, k| h[k] = Set.new } # target_url => Set of source_urls
17
- @out_links = Hash.new { |h, k| h[k] = Set.new } # source_url => Set of target_urls
17
+ @limit = limit.to_i > 0 ? limit.to_i : 50
18
+ @concurrency = [[concurrency.to_i, 1].max, 20].min
19
+ @graph = Hash.new { |h, k| h[k] = Set.new } # target_url => Set of source_urls
20
+ @out_links = Hash.new { |h, k| h[k] = Set.new } # source_url => Set of target_urls
21
+ @page_titles = {} # url => String
22
+ @page_anchors = Hash.new { |h, k| h[k] = {} } # target_url => { source_url => anchor_text }
18
23
  @all_discovered = Set.new
24
+ @mutex = Mutex.new
19
25
  end
20
26
 
21
- def audit(sitemap_urls = nil)
27
+ def audit(sitemap_urls = nil, &progress_block)
28
+ start_time = Time.now
29
+
22
30
  urls_to_crawl = if sitemap_urls && !sitemap_urls.empty?
23
31
  sitemap_urls.first(@limit)
24
32
  else
@@ -29,59 +37,138 @@ module GSC
29
37
  @all_discovered << normalize_url(url)
30
38
  end
31
39
 
32
- # Crawl each URL and extract internal links
33
- urls_to_crawl.each do |url|
34
- pa = GSC::PageAnalyzer.new(url)
35
- pa.load_content! rescue next
36
- dom = pa.analyze_dom rescue next
37
-
38
- norm_source = normalize_url(url)
39
- links = dom.dig(:links, :all) || []
40
+ # Multi-threaded concurrent crawling
41
+ crawl_urls_concurrently(urls_to_crawl, &progress_block)
40
42
 
41
- links.each do |link_obj|
42
- href = link_obj[:href]
43
- target_url = resolve_internal_url(href)
44
- next unless target_url
43
+ crawl_duration = (Time.now - start_time).round(2)
45
44
 
46
- norm_target = normalize_url(target_url)
47
- next if norm_target == norm_source
45
+ # Calculate click depths via BFS from root
46
+ normalized_root = normalize_url(@base_url)
47
+ depths = calculate_depths(normalized_root)
48
48
 
49
- @graph[norm_target] << norm_source
50
- @out_links[norm_source] << norm_target
51
- end
52
- end
49
+ # Top hub pages (highest out-links and in-links)
50
+ top_hubs = top_linked_pages(10)
51
+ hub_urls = top_hubs.map { |h| h[:url] }.reject { |u| u == normalized_root }
53
52
 
54
- # Calculate Orphans (pages in sitemap/discovered with 0 incoming internal links)
53
+ # Calculate Orphans and Weak Pages
55
54
  orphans = []
56
- weak_pages = [] # only 1 internal link
55
+ orphan_details = []
56
+ weak_pages = []
57
+ deep_pages = []
57
58
 
58
59
  @all_discovered.each do |url|
60
+ next if url == normalized_root
61
+
59
62
  in_degree = @graph[url].size
60
- if in_degree == 0 && url != normalize_url(@base_url)
63
+ depth = depths[url]
64
+
65
+ if depth && depth >= 4
66
+ deep_pages << { url: url, depth: depth, in_degree: in_degree }
67
+ end
68
+
69
+ if in_degree == 0
61
70
  orphans << url
71
+ orphan_details << {
72
+ url: url,
73
+ depth: depth || 'Unreachable via internal links (∞)',
74
+ suggested_rescues: generate_rescue_suggestions(url, hub_urls, normalized_root)
75
+ }
62
76
  elsif in_degree == 1
63
- weak_pages << { url: url, source: @graph[url].first }
77
+ source = @graph[url].first
78
+ anchor = @page_anchors.dig(url, source) || 'View page'
79
+ weak_pages << {
80
+ url: url,
81
+ source: source,
82
+ anchor: anchor,
83
+ depth: depth,
84
+ suggested_rescues: generate_rescue_suggestions(url, hub_urls, normalized_root)
85
+ }
64
86
  end
65
87
  end
66
88
 
67
- # Calculate click depths via BFS from root
68
- depths = calculate_depths(normalize_url(@base_url))
89
+ # Link Equity Health Score (0–100)
90
+ health_score = [100 - (orphans.size * 12) - (weak_pages.size * 3) - (deep_pages.size * 4), 0].max
91
+ health_grade = case health_score
92
+ when 90..100 then 'A'
93
+ when 80..89 then 'B'
94
+ when 70..79 then 'C'
95
+ when 60..69 then 'D'
96
+ else 'F'
97
+ end
98
+
99
+ # Depth distribution
100
+ depth_dist = Hash.new(0)
101
+ depths.each_value { |d| depth_dist[d] += 1 }
69
102
 
70
103
  {
71
104
  base_url: @base_url,
72
105
  total_pages: @all_discovered.size,
106
+ crawl_duration_s: crawl_duration,
107
+ health_score: health_score,
108
+ health_grade: health_grade,
73
109
  orphans: orphans,
110
+ orphan_details: orphan_details,
74
111
  weak_pages: weak_pages,
75
- top_linked: top_linked_pages(10),
76
- depths: depths
112
+ deep_pages: deep_pages,
113
+ top_linked: top_hubs,
114
+ depths: depths,
115
+ depth_distribution: depth_dist.sort.to_h
77
116
  }
78
117
  end
79
118
 
80
119
  private
81
120
 
121
+ def crawl_urls_concurrently(urls, &_block)
122
+ queue = Queue.new
123
+ urls.each { |u| queue << u }
124
+
125
+ workers = (1..@concurrency).map do
126
+ Thread.new do
127
+ while !queue.empty? && (url = queue.pop(true) rescue nil)
128
+ crawl_single_page(url)
129
+ end
130
+ end
131
+ end
132
+
133
+ workers.each(&:join)
134
+ end
135
+
136
+ def crawl_single_page(url)
137
+ pa = GSC::PageAnalyzer.new(url)
138
+ pa.load_content!
139
+ dom = pa.analyze_dom
140
+ return unless dom
141
+
142
+ norm_source = normalize_url(url)
143
+ title = dom.dig(:title, :text).to_s.strip
144
+ links = dom.dig(:links, :all) || []
145
+
146
+ @mutex.synchronize do
147
+ @page_titles[norm_source] = title unless title.empty?
148
+ end
149
+
150
+ links.each do |link_obj|
151
+ href = link_obj[:href]
152
+ target_url = resolve_internal_url(href)
153
+ next unless target_url
154
+
155
+ norm_target = normalize_url(target_url)
156
+ next if norm_target == norm_source
157
+
158
+ anchor = link_obj[:anchor].to_s.strip
159
+
160
+ @mutex.synchronize do
161
+ @graph[norm_target] << norm_source
162
+ @out_links[norm_source] << norm_target
163
+ @page_anchors[norm_target][norm_source] = anchor unless anchor.empty?
164
+ end
165
+ end
166
+ rescue StandardError
167
+ # resilient worker execution
168
+ end
169
+
82
170
  def discover_urls
83
- loader = GSC::SitemapLoader.new(@base_url)
84
- urls = loader.load
171
+ urls = GSC::SitemapLoader.resolve_urls(@base_url, @base_url, quiet: true)
85
172
  urls.empty? ? [@base_url] : urls.first(@limit)
86
173
  rescue StandardError
87
174
  [@base_url]
@@ -101,7 +188,7 @@ module GSC
101
188
 
102
189
  def normalize_url(url)
103
190
  u = url.to_s.strip.sub(%r{/$}, '')
104
- u
191
+ u.empty? ? @base_url : u
105
192
  end
106
193
 
107
194
  def calculate_depths(root_url)
@@ -125,8 +212,38 @@ module GSC
125
212
 
126
213
  def top_linked_pages(limit)
127
214
  @graph.map do |url, sources|
128
- { url: url, incoming_count: sources.size }
215
+ {
216
+ url: url,
217
+ title: @page_titles[url] || '',
218
+ incoming_count: sources.size,
219
+ outgoing_count: (@out_links[url] || []).size
220
+ }
129
221
  end.sort_by { |item| -item[:incoming_count] }.first(limit)
130
222
  end
223
+
224
+ def generate_rescue_suggestions(orphan_url, hub_urls, root_url)
225
+ # Extract keyword tokens from slug
226
+ path = URI.parse(orphan_url).path.to_s rescue ''
227
+ slug = path.split('/').reject(&:empty?).last || ''
228
+ keyword = slug.gsub(/[-_]+/, ' ').strip
229
+ keyword_title = keyword.split.map(&:capitalize).join(' ')
230
+
231
+ # Pick 2 best rescue sources: 1 high-equity hub, 1 root/pillar
232
+ sources = []
233
+ if !hub_urls.empty?
234
+ # Pick first hub that isn't the orphan itself
235
+ hub_source = hub_urls.find { |h| h != orphan_url }
236
+ sources << hub_source if hub_source
237
+ end
238
+ sources << root_url unless sources.include?(root_url)
239
+
240
+ sources.first(2).map do |src|
241
+ {
242
+ source_url: src,
243
+ recommended_anchor: keyword.empty? ? 'Explore this guide' : keyword_title,
244
+ action: "Add in-content link from #{src} pointing to #{orphan_url} with anchor \"#{keyword.empty? ? 'Learn more' : keyword_title}\""
245
+ }
246
+ end
247
+ end
131
248
  end
132
249
  end
@@ -0,0 +1,191 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'json'
5
+ require 'date'
6
+ require_relative 'ctr_curve'
7
+ require_relative 'intent_shift'
8
+
9
+ module GSC
10
+ class KeywordValue
11
+ DEFAULT_AOV = 75.0
12
+ DEFAULT_CONV_RATE = 0.025 # 2.5%
13
+ DEFAULT_MARGIN = 0.60 # 60% gross margin
14
+
15
+ attr_reader :options, :api, :domain, :aov, :conv_rate, :margin, :target_pos
16
+
17
+ def initialize(options = {}, api = nil, domain = nil)
18
+ @options = options
19
+ @api = api
20
+ @domain = domain.to_s.strip
21
+ @aov = (options[:aov] || DEFAULT_AOV).to_f
22
+ @conv_rate = (options[:conv_rate] || DEFAULT_CONV_RATE).to_f
23
+ @margin = (options[:margin] || DEFAULT_MARGIN).to_f
24
+ @target_pos = (options[:target_pos] || 3).to_i
25
+ end
26
+
27
+ def self.analyze(options = {}, api = nil, domain = nil)
28
+ new(options, api, domain).analyze
29
+ end
30
+
31
+ def analyze
32
+ rows = fetch_rows
33
+ target_ctr = GSC::CtrCurve.benchmark_for(@target_pos)
34
+
35
+ analyzed_keywords = rows.map do |row|
36
+ analyze_row(row, target_ctr)
37
+ end
38
+
39
+ # Sort by monthly revenue upside descending
40
+ analyzed_keywords.sort_by! { |k| -k[:monthly_revenue_upside] }
41
+
42
+ synthesize_portfolio(analyzed_keywords)
43
+ end
44
+
45
+ private
46
+
47
+ def analyze_row(row, target_ctr)
48
+ query = row[:query]
49
+ imp = row[:impressions] || 0
50
+ clicks = row[:clicks] || 0
51
+ pos = (row[:position] || 100.0).round(1)
52
+ ctr = row[:ctr] || (imp > 0 ? ((clicks.to_f / imp) * 100.0).round(2) : 0.0)
53
+
54
+ # Classify intent to apply dynamic conversion multipliers
55
+ intent = GSC::IntentShift.classify_query(query)
56
+ intent_multiplier = case intent
57
+ when :transactional then 2.0
58
+ when :commercial then 1.2
59
+ when :navigational then 1.5
60
+ else 0.6 # informational
61
+ end
62
+
63
+ effective_conv_rate = (@conv_rate * intent_multiplier).round(4)
64
+
65
+ # Current realized monthly revenue
66
+ current_orders = (clicks * effective_conv_rate).round(1)
67
+ current_revenue = (current_orders * @aov).round(2)
68
+ current_gross_profit = (current_revenue * @margin).round(2)
69
+
70
+ # Projected revenue at target position (e.g. Top 3)
71
+ if pos <= @target_pos
72
+ projected_clicks = clicks
73
+ incremental_clicks = 0
74
+ potential_revenue = current_revenue
75
+ monthly_upside = 0.0
76
+ else
77
+ projected_clicks = ((target_ctr / 100.0) * imp).round
78
+ incremental_clicks = [projected_clicks - clicks, 0].max
79
+ potential_orders = (projected_clicks * effective_conv_rate).round(1)
80
+ potential_revenue = (potential_orders * @aov).round(2)
81
+ monthly_upside = [potential_revenue - current_revenue, 0.0].max.round(2)
82
+ end
83
+ annual_upside = (monthly_upside * 12).round(2)
84
+
85
+ # Value Index score (0-100)
86
+ # High score if: in striking distance (pos 4-15), high upside, high intent
87
+ pos_weight = if pos.between?(4.0, 10.0) then 1.0
88
+ elsif pos.between?(10.1, 20.0) then 0.8
89
+ elsif pos <= 3.0 then 0.3 # already top 3
90
+ else 0.4
91
+ end
92
+
93
+ raw_score = (pos_weight * 40) + ([monthly_upside / 50.0, 40].min) + (intent_multiplier * 10)
94
+ value_score = [[raw_score.round, 100].min, 1].max
95
+
96
+ action = if pos <= 3.0
97
+ "👑 Top 3 Defend: Protect ranking with fresh content updates & internal hub equity."
98
+ elsif pos.between?(4.0, 10.0)
99
+ "🚀 Page-1 Striking Distance: +$#{format_currency(monthly_upside)}/mo upside! Optimize title CTR & add FAQ schema."
100
+ elsif pos.between?(10.1, 20.0)
101
+ "🎯 Page-2 Striking Query: +$#{format_currency(monthly_upside)}/mo upside. Expand content depth & earn 2 backlinks."
102
+ else
103
+ "🌱 Deep Discovery: Expand keyword topic cluster to lift organic visibility."
104
+ end
105
+
106
+ {
107
+ query: query,
108
+ intent: intent,
109
+ position: pos,
110
+ impressions: imp,
111
+ clicks: clicks,
112
+ ctr: ctr,
113
+ effective_conv_rate: (effective_conv_rate * 100.0).round(2),
114
+ current_orders: current_orders,
115
+ current_monthly_revenue: current_revenue,
116
+ current_gross_profit: current_gross_profit,
117
+ potential_clicks: projected_clicks,
118
+ incremental_clicks: incremental_clicks,
119
+ potential_monthly_revenue: potential_revenue,
120
+ monthly_revenue_upside: monthly_upside,
121
+ annual_revenue_upside: annual_upside,
122
+ value_score: value_score,
123
+ action: action
124
+ }
125
+ end
126
+
127
+ def synthesize_portfolio(keywords)
128
+ total_current_rev = keywords.sum { |k| k[:current_monthly_revenue] }.round(2)
129
+ total_potential_rev = keywords.sum { |k| k[:potential_monthly_revenue] }.round(2)
130
+ total_monthly_upside = keywords.sum { |k| k[:monthly_revenue_upside] }.round(2)
131
+ total_annual_upside = (total_monthly_upside * 12).round(2)
132
+
133
+ striking_count = keywords.count { |k| k[:position].between?(4.0, 20.0) }
134
+ high_intent_count = keywords.count { |k| k[:intent] == :transactional || k[:intent] == :commercial }
135
+
136
+ {
137
+ domain: @domain,
138
+ parameters: {
139
+ aov: @aov,
140
+ conversion_rate_pct: (@conv_rate * 100.0).round(2),
141
+ profit_margin_pct: (@margin * 100.0).round(2),
142
+ target_position: @target_pos
143
+ },
144
+ total_keywords_analyzed: keywords.size,
145
+ striking_distance_keywords: striking_count,
146
+ high_commercial_intent_keywords: high_intent_count,
147
+ financials: {
148
+ current_monthly_revenue: total_current_rev,
149
+ current_annual_run_rate: (total_current_rev * 12).round(2),
150
+ potential_monthly_revenue: total_potential_rev,
151
+ unlocked_monthly_upside: total_monthly_upside,
152
+ unlocked_annual_pipeline_upside: total_annual_upside
153
+ },
154
+ keywords: keywords
155
+ }
156
+ end
157
+
158
+ def fetch_rows
159
+ if @api
160
+ begin
161
+ days = (@options[:days] || 28).to_i
162
+ end_date = (Date.today - 2).strftime('%Y-%m-%d')
163
+ start_date = (Date.today - 2 - days).strftime('%Y-%m-%d')
164
+ res = @api.search_analytics(@domain, start_date: start_date, end_date: end_date, dimensions: %w[query])
165
+ rows = res['rows'] || []
166
+ if rows.any?
167
+ return rows.first(25).map do |r|
168
+ {
169
+ query: r['keys'][0],
170
+ clicks: r['clicks'] || 0,
171
+ impressions: r['impressions'] || 0,
172
+ ctr: ((r['ctr'] || 0) * 100.0).round(2),
173
+ position: (r['position'] || 0).round(1)
174
+ }
175
+ end
176
+ end
177
+ rescue StandardError
178
+ # GSC query failed
179
+ end
180
+ end
181
+
182
+ []
183
+ end
184
+
185
+ def format_currency(val)
186
+ parts = sprintf('%.2f', val.to_f).split('.')
187
+ parts[0] = parts[0].reverse.gsub(/(\d{3})(?=\d)/, '\\1,').reverse
188
+ parts.join('.')
189
+ end
190
+ end
191
+ end
@@ -0,0 +1,195 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'json'
5
+ require 'uri'
6
+
7
+ module GSC
8
+ class LandingRoi
9
+ DEFAULT_AOV = 75.0 # Average Order Value / Customer LTV ($)
10
+ DEFAULT_CONV_RATE = 0.025 # 2.5% Conversion rate of engaged visitors
11
+ DEFAULT_BENCHMARK_BOUNCE = 45.0 # 45% standard healthy bounce rate
12
+ DEFAULT_CPC = 1.50 # Replacement PPC cost per click
13
+
14
+ attr_reader :options, :aov, :conv_rate, :benchmark_bounce, :cpc
15
+
16
+ def initialize(options = {})
17
+ @options = options
18
+ @aov = (options[:aov] || DEFAULT_AOV).to_f
19
+ @conv_rate = (options[:conv_rate] || DEFAULT_CONV_RATE).to_f
20
+ @benchmark_bounce = (options[:benchmark_bounce] || DEFAULT_BENCHMARK_BOUNCE).to_f
21
+ @cpc = (options[:cpc] || DEFAULT_CPC).to_f
22
+ end
23
+
24
+ # Analyzes combined GSC pre-click and GA4 post-click data
25
+ def analyze_pages(pages_data)
26
+ analyzed = pages_data.map do |page|
27
+ analyze_single_page(page)
28
+ end
29
+
30
+ # Sort by highest monthly revenue leak by default
31
+ analyzed.sort_by! { |p| -(p[:monthly_revenue_leak] || 0.0) }
32
+
33
+ total_clicks = analyzed.sum { |p| p[:clicks] || 0 }
34
+ total_sessions = analyzed.sum { |p| p[:sessions] || 0 }
35
+ total_leak = analyzed.sum { |p| p[:monthly_revenue_leak] || 0.0 }
36
+ total_actual_rev = analyzed.sum { |p| p[:actual_revenue_est] || 0.0 }
37
+ total_lost_visitors = analyzed.sum { |p| p[:lost_visitors] || 0 }
38
+
39
+ quadrant_summary = {
40
+ cash_cows: analyzed.count { |p| p[:quadrant] == :cash_cow },
41
+ revenue_leakers: analyzed.count { |p| p[:quadrant] == :revenue_leaker },
42
+ hidden_gems: analyzed.count { |p| p[:quadrant] == :hidden_gem },
43
+ zombies: analyzed.count { |p| p[:quadrant] == :zombie }
44
+ }
45
+
46
+ {
47
+ total_pages: analyzed.size,
48
+ total_clicks: total_clicks,
49
+ total_sessions: total_sessions,
50
+ total_monthly_leak: total_leak.round(2),
51
+ total_annual_leak: (total_leak * 12.0).round(2),
52
+ total_actual_revenue_est: total_actual_rev.round(2),
53
+ total_lost_visitors: total_lost_visitors,
54
+ quadrants: quadrant_summary,
55
+ assumptions: {
56
+ aov: @aov,
57
+ conv_rate_pct: (@conv_rate * 100.0).round(2),
58
+ benchmark_bounce_pct: @benchmark_bounce,
59
+ estimated_cpc: @cpc
60
+ },
61
+ pages: analyzed
62
+ }
63
+ end
64
+
65
+ def analyze_single_page(page)
66
+ url = page[:url] || page['url'] || page[:path] || page['path'] || '/'
67
+ clicks = (page[:clicks] || page['clicks'] || 0).to_i
68
+ impressions = (page[:impressions] || page['impressions'] || 0).to_i
69
+ position = (page[:position] || page['position'] || 0.0).to_f.round(1)
70
+ ctr = (page[:ctr] || page['ctr'] || 0.0).to_f.round(2)
71
+ sessions = (page[:sessions] || page['sessions'] || clicks).to_i
72
+ bounce_rate = (page[:bounce_rate] || page['bounce_rate'] || estimate_bounce_rate(clicks, position)).to_f.round(1)
73
+ duration = (page[:duration] || page['duration'] || page[:duration_seconds] || 60.0).to_f.round(1)
74
+
75
+ # Economic Calculations
76
+ excess_bounce = [0.0, bounce_rate - @benchmark_bounce].max
77
+ lost_visitors = (clicks * (excess_bounce / 100.0)).round
78
+ monthly_revenue_leak = (lost_visitors * @conv_rate * @aov).round(2)
79
+ annual_revenue_leak = (monthly_revenue_leak * 12.0).round(2)
80
+
81
+ engaged_clicks = [0, clicks - lost_visitors].max
82
+ actual_revenue_est = (engaged_clicks * @conv_rate * @aov).round(2)
83
+ traffic_asset_value = (clicks * @cpc).round(2)
84
+
85
+ # Page Economic Health Index (PEHI 0-100)
86
+ # High clicks + low bounce + high duration = 100
87
+ pehi_score = calculate_pehi(clicks, bounce_rate, duration)
88
+
89
+ # Strategic Quadrant
90
+ quadrant = classify_quadrant(clicks, bounce_rate, duration, position)
91
+
92
+ # Conversion Prescriptions & Fixes
93
+ prescriptions = generate_cro_prescriptions(bounce_rate, duration, clicks, position)
94
+
95
+ {
96
+ url: url,
97
+ clicks: clicks,
98
+ impressions: impressions,
99
+ position: position,
100
+ ctr: ctr,
101
+ sessions: sessions,
102
+ bounce_rate: bounce_rate,
103
+ duration_seconds: duration,
104
+ pehi_score: pehi_score,
105
+ quadrant: quadrant,
106
+ lost_visitors: lost_visitors,
107
+ monthly_revenue_leak: monthly_revenue_leak,
108
+ annual_revenue_leak: annual_revenue_leak,
109
+ actual_revenue_est: actual_revenue_est,
110
+ traffic_asset_value: traffic_asset_value,
111
+ prescriptions: prescriptions
112
+ }
113
+ end
114
+
115
+ private
116
+
117
+ def calculate_pehi(clicks, bounce_rate, duration)
118
+ score = 50.0
119
+
120
+ # Bounce Rate Component (max +/- 30pts)
121
+ if bounce_rate <= 35.0
122
+ score += 30.0
123
+ elsif bounce_rate <= 45.0
124
+ score += 15.0
125
+ elsif bounce_rate >= 75.0
126
+ score -= 30.0
127
+ elsif bounce_rate >= 60.0
128
+ score -= 15.0
129
+ end
130
+
131
+ # Duration Component (max +/- 15pts)
132
+ if duration >= 120.0
133
+ score += 15.0
134
+ elsif duration >= 60.0
135
+ score += 8.0
136
+ elsif duration <= 25.0
137
+ score -= 15.0
138
+ elsif duration <= 45.0
139
+ score -= 8.0
140
+ end
141
+
142
+ # Volume Bonus (max +5pts)
143
+ score += 5.0 if clicks >= 100
144
+
145
+ score.clamp(0.0, 100.0).round
146
+ end
147
+
148
+ def classify_quadrant(clicks, bounce_rate, duration, position)
149
+ if clicks >= 25 && bounce_rate <= @benchmark_bounce && duration >= 45.0
150
+ :cash_cow
151
+ elsif clicks >= 25 && bounce_rate > @benchmark_bounce
152
+ :revenue_leaker
153
+ elsif clicks < 25 && bounce_rate <= @benchmark_bounce && duration >= 60.0
154
+ :hidden_gem
155
+ else
156
+ :zombie
157
+ end
158
+ end
159
+
160
+ def generate_cro_prescriptions(bounce_rate, duration, clicks, position)
161
+ fixes = []
162
+
163
+ if bounce_rate >= 70.0
164
+ fixes << "🚨 HIGH BOUNCE TRAP: Place sticky, high-contrast CTA button above 450px fold."
165
+ fixes << "🛡️ ADD SOCIAL PROOF: Insert customer rating stars, logos, or guarantee badge in the top viewport."
166
+ elsif bounce_rate > 55.0
167
+ fixes << "⚡ REDUCE BOUNCE: Add interactive jump-links / Table of Contents to lower drop-off."
168
+ end
169
+
170
+ if duration < 30.0
171
+ fixes << "⏱️ LOW TIME ON PAGE: Searchers are not finding answers instantly; add a 2-sentence executive summary under H1."
172
+ end
173
+
174
+ if clicks >= 50 && bounce_rate > 60.0
175
+ fixes << "💰 HIGH TRAFFIC BLEED: Priority CRO target. Run A/B test on hero headline and primary action."
176
+ end
177
+
178
+ if position >= 7.0 && bounce_rate <= 40.0
179
+ fixes << "⭐ HIDDEN HIGH CONVERTER: Searchers love this page (low bounce). Build 3 internal links to push from pos #{position} to Top 3."
180
+ end
181
+
182
+ fixes << "✅ HEALTHY ENGAGEMENT: Maintain current content structure and monitor core queries." if fixes.empty?
183
+ fixes
184
+ end
185
+
186
+ def estimate_bounce_rate(clicks, position)
187
+ # Fallback heuristic when GA4 is not linked:
188
+ # Informational / deeper SERP positions generally experience higher bounce rates
189
+ base = 50.0
190
+ base += 5.0 if position > 5.0
191
+ base += 8.0 if position > 10.0
192
+ base.clamp(35.0, 80.0)
193
+ end
194
+ end
195
+ end