gsc-cli 2.0.2 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/AUTH.md +205 -0
- data/FUNDING.md +120 -0
- data/README.md +463 -299
- data/bin/gsc +29158 -4921
- data/dist/gsc +29158 -4921
- data/lib/gsc/aio_hunter.rb +343 -0
- data/lib/gsc/answer_synthesizer.rb +157 -0
- data/lib/gsc/api.rb +53 -1
- data/lib/gsc/auth.rb +26 -0
- data/lib/gsc/backlinks_manager.rb +96 -0
- data/lib/gsc/brand_segmenter.rb +140 -0
- data/lib/gsc/cache_manager.rb +806 -0
- data/lib/gsc/cannibalization_analyzer.rb +141 -0
- data/lib/gsc/canonical_chains.rb +367 -0
- data/lib/gsc/citation_simulator.rb +339 -0
- data/lib/gsc/cli/aio_hunter.rb +154 -0
- data/lib/gsc/cli/analytics.rb +788 -0
- data/lib/gsc/cli/audit.rb +1976 -0
- data/lib/gsc/cli/base.rb +384 -0
- data/lib/gsc/cli/cache.rb +266 -0
- data/lib/gsc/cli/canonical.rb +223 -0
- data/lib/gsc/cli/citation_simulator.rb +152 -0
- data/lib/gsc/cli/dashboard.rb +354 -0
- data/lib/gsc/cli/doctor.rb +129 -0
- data/lib/gsc/cli/eeat.rb +125 -0
- data/lib/gsc/cli/ga4.rb +852 -0
- data/lib/gsc/cli/growth.rb +650 -0
- data/lib/gsc/cli/hreflang.rb +164 -0
- data/lib/gsc/cli/image_seo.rb +162 -0
- data/lib/gsc/cli/indexing.rb +458 -0
- data/lib/gsc/cli/intent_shift.rb +125 -0
- data/lib/gsc/cli/keyword_value.rb +134 -0
- data/lib/gsc/cli/keywords.rb +795 -0
- data/lib/gsc/cli/landing_roi.rb +308 -0
- data/lib/gsc/cli/low_ctr.rb +213 -0
- data/lib/gsc/cli/mobile_parity.rb +150 -0
- data/lib/gsc/cli/report.rb +100 -0
- data/lib/gsc/cli/rich_results.rb +172 -0
- data/lib/gsc/cli/schema_generate.rb +149 -0
- data/lib/gsc/cli/seasonal.rb +232 -0
- data/lib/gsc/cli/security.rb +153 -0
- data/lib/gsc/cli/setup.rb +1291 -0
- data/lib/gsc/cli/sitemap_tree.rb +143 -0
- data/lib/gsc/cli/skill_pack.rb +62 -0
- data/lib/gsc/cli/soft_404.rb +199 -0
- data/lib/gsc/cli/sparkline.rb +227 -0
- data/lib/gsc/cli/watchdog.rb +150 -0
- data/lib/gsc/cli/zombie_purger.rb +208 -0
- data/lib/gsc/cli.rb +706 -5111
- data/lib/gsc/cli_advanced.rb +1513 -0
- data/lib/gsc/client.rb +17 -2
- data/lib/gsc/color.rb +16 -1
- data/lib/gsc/command_registry.rb +47 -9
- data/lib/gsc/config.rb +11 -2
- data/lib/gsc/content_gap.rb +112 -0
- data/lib/gsc/ctr_curve.rb +115 -0
- data/lib/gsc/decay_predictor.rb +322 -0
- data/lib/gsc/doctor.rb +434 -0
- data/lib/gsc/eeat_auditor.rb +428 -0
- data/lib/gsc/entity_auditor.rb +229 -0
- data/lib/gsc/firewall_scanner.rb +733 -0
- data/lib/gsc/geo_auditor.rb +368 -0
- data/lib/gsc/google_suggest.rb +109 -0
- data/lib/gsc/google_trends.rb +8 -1
- data/lib/gsc/heading_validator.rb +283 -0
- data/lib/gsc/hreflang_validator.rb +412 -0
- data/lib/gsc/image_seo.rb +286 -0
- data/lib/gsc/indexing_queue.rb +179 -0
- data/lib/gsc/indexnow.rb +93 -0
- data/lib/gsc/intent_shift.rb +188 -0
- data/lib/gsc/internal_links.rb +249 -0
- data/lib/gsc/keyword_value.rb +191 -0
- data/lib/gsc/landing_roi.rb +195 -0
- data/lib/gsc/llms_generator.rb +425 -0
- data/lib/gsc/low_ctr_rewriter.rb +408 -0
- data/lib/gsc/mobile_parity.rb +222 -0
- data/lib/gsc/network_tracer.rb +93 -0
- data/lib/gsc/open_page_rank.rb +72 -0
- data/lib/gsc/page_analyzer.rb +47 -7
- data/lib/gsc/page_comparator.rb +108 -0
- data/lib/gsc/page_speed.rb +110 -0
- data/lib/gsc/prompts.rb +38 -29
- data/lib/gsc/questions_harvester.rb +178 -0
- data/lib/gsc/report_generator.rb +461 -0
- data/lib/gsc/rich_results.rb +388 -0
- data/lib/gsc/robots_checker.rb +114 -0
- data/lib/gsc/schema_generator.rb +788 -0
- data/lib/gsc/schema_validator.rb +120 -0
- data/lib/gsc/seasonal_predictor.rb +381 -0
- data/lib/gsc/security_scanner.rb +496 -0
- data/lib/gsc/serp_feature_detector.rb +359 -0
- data/lib/gsc/serp_preview.rb +152 -0
- data/lib/gsc/site_crawler.rb +113 -21
- data/lib/gsc/sitemap_loader.rb +15 -4
- data/lib/gsc/sitemap_tree.rb +301 -0
- data/lib/gsc/skill_pack.rb +195 -0
- data/lib/gsc/soft_404_analyzer.rb +385 -0
- data/lib/gsc/sparkline.rb +171 -0
- data/lib/gsc/speed_correlator.rb +416 -0
- data/lib/gsc/striking_playbook.rb +190 -0
- data/lib/gsc/title_optimizer.rb +420 -0
- data/lib/gsc/vault.rb +260 -0
- data/lib/gsc/version.rb +1 -1
- data/lib/gsc/watchdog.rb +235 -0
- data/lib/gsc/zombie_purger.rb +366 -0
- data/lib/gsc.rb +144 -0
- metadata +91 -2
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module GSC
|
|
4
|
+
class CannibalizationAnalyzer
|
|
5
|
+
def self.analyze(rows, min_imp: 10, min_secondary_ratio: 0.15)
|
|
6
|
+
grouped = Hash.new { |h, k| h[k] = [] }
|
|
7
|
+
|
|
8
|
+
rows.each do |r|
|
|
9
|
+
query = r[:query] || r['query'] || r.dig('keys', 0)
|
|
10
|
+
page = r[:page] || r['page'] || r.dig('keys', 1)
|
|
11
|
+
clicks = r[:clicks] || r['clicks'] || 0
|
|
12
|
+
imp = r[:impressions] || r['impressions'] || 0
|
|
13
|
+
pos = (r[:position] || r['position'] || 100.0).to_f.round(1)
|
|
14
|
+
|
|
15
|
+
next unless query && page
|
|
16
|
+
|
|
17
|
+
grouped[query] << {
|
|
18
|
+
page: page,
|
|
19
|
+
clicks: clicks,
|
|
20
|
+
impressions: imp,
|
|
21
|
+
position: pos
|
|
22
|
+
}
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
conflicts = []
|
|
26
|
+
total_wasted_imp = 0
|
|
27
|
+
|
|
28
|
+
grouped.each do |query, pages|
|
|
29
|
+
next if pages.size < 2
|
|
30
|
+
|
|
31
|
+
total_imp = pages.sum { |p| p[:impressions] }
|
|
32
|
+
next if total_imp < min_imp
|
|
33
|
+
|
|
34
|
+
sorted_pages = pages.sort_by { |p| -p[:impressions] }
|
|
35
|
+
primary = sorted_pages[0]
|
|
36
|
+
secondary = sorted_pages[1]
|
|
37
|
+
|
|
38
|
+
sec_ratio = secondary[:impressions].to_f / total_imp
|
|
39
|
+
next if sec_ratio < min_secondary_ratio
|
|
40
|
+
|
|
41
|
+
severity = calculate_severity(primary, secondary, total_imp, sec_ratio)
|
|
42
|
+
remedy = determine_remedy(primary, secondary, query)
|
|
43
|
+
|
|
44
|
+
# Wasted / split impressions from secondary+ competing pages
|
|
45
|
+
diluted_imp = pages[1..].sum { |p| p[:impressions] }
|
|
46
|
+
total_wasted_imp += diluted_imp
|
|
47
|
+
|
|
48
|
+
conflicts << {
|
|
49
|
+
query: query,
|
|
50
|
+
severity: severity,
|
|
51
|
+
total_impressions: total_imp,
|
|
52
|
+
total_clicks: pages.sum { |p| p[:clicks] },
|
|
53
|
+
competing_pages_count: pages.size,
|
|
54
|
+
diluted_impressions: diluted_imp,
|
|
55
|
+
remedy: remedy,
|
|
56
|
+
pages: sorted_pages.map do |p|
|
|
57
|
+
share = ((p[:impressions].to_f / total_imp) * 100).round(1)
|
|
58
|
+
p.merge(
|
|
59
|
+
impression_share: share,
|
|
60
|
+
impression_share_str: "#{share}%"
|
|
61
|
+
)
|
|
62
|
+
end
|
|
63
|
+
}
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
# Sort by severity (CRITICAL > HIGH > MODERATE > LOW) then total impressions
|
|
67
|
+
severity_order = { 'CRITICAL' => 4, 'HIGH' => 3, 'MODERATE' => 2, 'LOW' => 1 }
|
|
68
|
+
conflicts.sort_by! { |c| [-severity_order[c[:severity]], -c[:total_impressions]] }
|
|
69
|
+
|
|
70
|
+
critical_count = conflicts.count { |c| c[:severity] == 'CRITICAL' }
|
|
71
|
+
health_score = [100 - (conflicts.size * 10 + critical_count * 15), 0].max
|
|
72
|
+
|
|
73
|
+
grade = case health_score
|
|
74
|
+
when 90..100 then 'A'
|
|
75
|
+
when 80..89 then 'B'
|
|
76
|
+
when 70..79 then 'C'
|
|
77
|
+
when 60..69 then 'D'
|
|
78
|
+
else 'F'
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
{
|
|
82
|
+
total_queries_evaluated: grouped.size,
|
|
83
|
+
conflicts_count: conflicts.size,
|
|
84
|
+
critical_count: critical_count,
|
|
85
|
+
total_diluted_impressions: total_wasted_imp,
|
|
86
|
+
health_score: health_score,
|
|
87
|
+
health_grade: grade,
|
|
88
|
+
conflicts: conflicts
|
|
89
|
+
}
|
|
90
|
+
end
|
|
91
|
+
|
|
92
|
+
def self.calculate_severity(primary, secondary, total_imp, sec_ratio)
|
|
93
|
+
# If both rank on Page 1 or 2 and share is closely split
|
|
94
|
+
if (primary[:position] <= 20.0 || secondary[:position] <= 20.0) && (sec_ratio >= 0.35 || total_imp >= 50)
|
|
95
|
+
'CRITICAL'
|
|
96
|
+
elsif sec_ratio >= 0.25 || total_imp >= 25
|
|
97
|
+
'HIGH'
|
|
98
|
+
elsif sec_ratio >= 0.15
|
|
99
|
+
'MODERATE'
|
|
100
|
+
else
|
|
101
|
+
'LOW'
|
|
102
|
+
end
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
def self.determine_remedy(primary, secondary, _query)
|
|
106
|
+
p_path = primary[:page].sub(%r{^https?://[^/]+}, '')
|
|
107
|
+
p_path = '/' if p_path.empty?
|
|
108
|
+
s_path = secondary[:page].sub(%r{^https?://[^/]+}, '')
|
|
109
|
+
s_path = '/' if s_path.empty?
|
|
110
|
+
|
|
111
|
+
# Case 1: Secondary page actually ranks higher than primary!
|
|
112
|
+
if secondary[:position] < (primary[:position] - 1.5)
|
|
113
|
+
{
|
|
114
|
+
action: 'FLIP_FLOP_CONSOLIDATION',
|
|
115
|
+
recommendation: "Googlebot prefers secondary page (#{s_path} ranks Pos #{secondary[:position]} vs #{primary[:position]}). Set canonical from #{p_path} pointing to #{s_path}, or 301 redirect #{p_path} to #{s_path}."
|
|
116
|
+
}
|
|
117
|
+
# Case 2: Inverted or duplicate slugs in same directory
|
|
118
|
+
elsif slug_similarity(p_path, s_path) > 0.7
|
|
119
|
+
{
|
|
120
|
+
action: '301_REDIRECT',
|
|
121
|
+
recommendation: "Near-identical slug intent. 301 redirect secondary #{s_path} into primary #{p_path} to merge PageRank."
|
|
122
|
+
}
|
|
123
|
+
# Case 3: Distinct topics accidentally colliding
|
|
124
|
+
else
|
|
125
|
+
{
|
|
126
|
+
action: 'CANONICAL_OR_LINK_DE_OPTIMIZE',
|
|
127
|
+
recommendation: "Differentiate content angles: set canonical pointing to #{p_path}, and remove \"#{_query}\" keyword from secondary #{s_path}'s H1/title."
|
|
128
|
+
}
|
|
129
|
+
end
|
|
130
|
+
end
|
|
131
|
+
|
|
132
|
+
def self.slug_similarity(path1, path2)
|
|
133
|
+
tokens1 = path1.downcase.split(/[^a-z0-9]+/).reject { |t| t.length < 2 }
|
|
134
|
+
tokens2 = path2.downcase.split(/[^a-z0-9]+/).reject { |t| t.length < 2 }
|
|
135
|
+
return 0.0 if tokens1.empty? || tokens2.empty?
|
|
136
|
+
|
|
137
|
+
common = (tokens1 & tokens2).size
|
|
138
|
+
common.to_f / [tokens1.size, tokens2.size].max
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
end
|
|
@@ -0,0 +1,367 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'net/http'
|
|
5
|
+
require 'uri'
|
|
6
|
+
require 'set'
|
|
7
|
+
require 'time'
|
|
8
|
+
|
|
9
|
+
module GSC
|
|
10
|
+
class CanonicalChains
|
|
11
|
+
EQUITY_DECAY_PER_HOP = 0.15 # 15% estimated equity loss per redirect hop
|
|
12
|
+
|
|
13
|
+
attr_reader :max_hops, :timeout, :user_agent
|
|
14
|
+
|
|
15
|
+
def initialize(max_hops: 10, timeout: 8, user_agent: nil)
|
|
16
|
+
@max_hops = max_hops
|
|
17
|
+
@timeout = timeout
|
|
18
|
+
@user_agent = user_agent || 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) gsc-cli/2.1'
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
# Analyze a single URL through redirect hops and canonical declarations
|
|
22
|
+
def audit_url(url, gsc_rows: nil)
|
|
23
|
+
start_url = normalize_url(url)
|
|
24
|
+
hops = []
|
|
25
|
+
visited_urls = []
|
|
26
|
+
loop_detected = false
|
|
27
|
+
loop_type = nil
|
|
28
|
+
current_url = start_url
|
|
29
|
+
|
|
30
|
+
@max_hops.times do |hop_index|
|
|
31
|
+
if visited_urls.include?(current_url)
|
|
32
|
+
loop_detected = true
|
|
33
|
+
loop_type = :redirect_loop
|
|
34
|
+
hops << {
|
|
35
|
+
hop: hop_index + 1,
|
|
36
|
+
url: current_url,
|
|
37
|
+
status_code: 0,
|
|
38
|
+
duration_ms: 0,
|
|
39
|
+
error: "Redirect loop detected: '#{current_url}' previously visited in chain"
|
|
40
|
+
}
|
|
41
|
+
break
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
visited_urls << current_url
|
|
45
|
+
hop_data = fetch_hop(current_url, hop_index + 1)
|
|
46
|
+
hops << hop_data
|
|
47
|
+
|
|
48
|
+
break if hop_data[:error]
|
|
49
|
+
|
|
50
|
+
# Check for HTTP redirect
|
|
51
|
+
if hop_data[:is_redirect] && hop_data[:redirect_target]
|
|
52
|
+
next_url = hop_data[:redirect_target]
|
|
53
|
+
current_url = next_url
|
|
54
|
+
else
|
|
55
|
+
# Final destination reached, inspect canonical
|
|
56
|
+
break
|
|
57
|
+
end
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
final_hop = hops.reject { |h| h[:error] && h[:status_code] == 0 }.last || hops.last || {}
|
|
61
|
+
final_url = final_hop[:url] || current_url
|
|
62
|
+
final_canonical = final_hop[:canonical_url]
|
|
63
|
+
|
|
64
|
+
# Check for Canonical Loop: Final page's canonical points back to an earlier hop in the chain
|
|
65
|
+
canonical_loop = false
|
|
66
|
+
if final_canonical && normalize_url(final_canonical) != normalize_url(final_url) && visited_urls.include?(normalize_url(final_canonical))
|
|
67
|
+
canonical_loop = true
|
|
68
|
+
loop_detected = true
|
|
69
|
+
loop_type = :mixed_canonical_loop
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
# Canonical target status check: if canonical points to a different URL, check if that canonical itself redirects
|
|
73
|
+
canonical_status = nil
|
|
74
|
+
canonical_redirects = false
|
|
75
|
+
if final_canonical && normalize_url(final_canonical) != normalize_url(final_url)
|
|
76
|
+
canonical_info = check_canonical_target(final_canonical)
|
|
77
|
+
canonical_status = canonical_info[:status_code]
|
|
78
|
+
canonical_redirects = canonical_info[:is_redirect]
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
redirect_hops_count = hops.count { |h| h[:is_redirect] }
|
|
82
|
+
total_duration_ms = hops.sum { |h| h[:duration_ms] || 0 }.round(1)
|
|
83
|
+
|
|
84
|
+
# Equity calculation
|
|
85
|
+
# Base equity = 1.0 (100%). Each redirect hop loses ~15%. Canonical loops destroy equity completely.
|
|
86
|
+
equity_retention = if canonical_loop || loop_type == :redirect_loop
|
|
87
|
+
0.0
|
|
88
|
+
elsif redirect_hops_count == 0
|
|
89
|
+
1.0
|
|
90
|
+
else
|
|
91
|
+
((1.0 - EQUITY_DECAY_PER_HOP)**redirect_hops_count).round(4)
|
|
92
|
+
end
|
|
93
|
+
equity_loss_pct = ((1.0 - equity_retention) * 100).round(1)
|
|
94
|
+
|
|
95
|
+
# Severity classification
|
|
96
|
+
issues = []
|
|
97
|
+
severity = :healthy
|
|
98
|
+
|
|
99
|
+
if loop_detected
|
|
100
|
+
severity = :critical
|
|
101
|
+
issues << {
|
|
102
|
+
code: :loop,
|
|
103
|
+
type: loop_type,
|
|
104
|
+
message: loop_type == :mixed_canonical_loop ?
|
|
105
|
+
"Mixed Canonical Loop: Final URL '#{final_url}' canonicalizes to '#{final_canonical}', which redirects back into chain" :
|
|
106
|
+
"Infinite Redirect Loop detected at '#{current_url}'"
|
|
107
|
+
}
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
if redirect_hops_count >= 2
|
|
111
|
+
severity = :warning if severity == :healthy
|
|
112
|
+
issues << {
|
|
113
|
+
code: :redirect_chain,
|
|
114
|
+
message: "Multi-hop redirect chain detected: #{redirect_hops_count} hops (#{hops.map { |h| h[:status_code] }.compact.join(' -> ')})"
|
|
115
|
+
}
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
if canonical_redirects
|
|
119
|
+
severity = :critical
|
|
120
|
+
issues << {
|
|
121
|
+
code: :canonical_redirects,
|
|
122
|
+
message: "Target canonical URL '#{final_canonical}' issues a redirect (#{canonical_status}). Google ignores redirecting canonicals."
|
|
123
|
+
}
|
|
124
|
+
end
|
|
125
|
+
|
|
126
|
+
if final_canonical.nil? && final_hop[:status_code] == 200
|
|
127
|
+
issues << {
|
|
128
|
+
code: :missing_canonical,
|
|
129
|
+
message: "Final destination '#{final_url}' is missing a rel=\"canonical\" tag"
|
|
130
|
+
}
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
# Cross-reference GSC traffic if rows provided
|
|
134
|
+
traffic_at_risk = extract_gsc_traffic(visited_urls + [final_canonical].compact, gsc_rows)
|
|
135
|
+
|
|
136
|
+
# Generate server collapse rewrite rules
|
|
137
|
+
server_rules = generate_collapse_rules(start_url, final_canonical || final_url)
|
|
138
|
+
|
|
139
|
+
{
|
|
140
|
+
start_url: start_url,
|
|
141
|
+
final_url: final_url,
|
|
142
|
+
canonical_url: final_canonical,
|
|
143
|
+
redirect_hops: redirect_hops_count,
|
|
144
|
+
total_hops: hops.length,
|
|
145
|
+
total_duration_ms: total_duration_ms,
|
|
146
|
+
equity_retention_pct: (equity_retention * 100).round(1),
|
|
147
|
+
equity_loss_pct: equity_loss_pct,
|
|
148
|
+
loop_detected: loop_detected,
|
|
149
|
+
loop_type: loop_type,
|
|
150
|
+
canonical_loop: canonical_loop,
|
|
151
|
+
canonical_status: canonical_status,
|
|
152
|
+
canonical_redirects: canonical_redirects,
|
|
153
|
+
severity: severity,
|
|
154
|
+
issues: issues,
|
|
155
|
+
traffic_at_risk: traffic_at_risk,
|
|
156
|
+
server_rules: server_rules,
|
|
157
|
+
hops: hops
|
|
158
|
+
}
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
# Batch audit an array of URLs or top pages
|
|
162
|
+
def audit_batch(urls, gsc_rows: nil, concurrency: 4)
|
|
163
|
+
clean_urls = urls.map { |u| normalize_url(u) }.uniq
|
|
164
|
+
results = []
|
|
165
|
+
|
|
166
|
+
# Thread pool for faster batch audits
|
|
167
|
+
queue = Queue.new
|
|
168
|
+
clean_urls.each { |u| queue << u }
|
|
169
|
+
mutex = Mutex.new
|
|
170
|
+
|
|
171
|
+
workers = [concurrency, clean_urls.size].min
|
|
172
|
+
workers = 1 if workers < 1
|
|
173
|
+
|
|
174
|
+
threads = Array.new(workers) do
|
|
175
|
+
Thread.new do
|
|
176
|
+
until queue.empty?
|
|
177
|
+
url = begin
|
|
178
|
+
queue.pop(true)
|
|
179
|
+
rescue ThreadError
|
|
180
|
+
nil
|
|
181
|
+
end
|
|
182
|
+
break unless url
|
|
183
|
+
|
|
184
|
+
res = audit_url(url, gsc_rows: gsc_rows)
|
|
185
|
+
mutex.synchronize { results << res }
|
|
186
|
+
end
|
|
187
|
+
end
|
|
188
|
+
end
|
|
189
|
+
threads.each(&:join)
|
|
190
|
+
|
|
191
|
+
# Summary metrics
|
|
192
|
+
total_audited = results.size
|
|
193
|
+
chains = results.select { |r| r[:redirect_hops] >= 2 }
|
|
194
|
+
loops = results.select { |r| r[:loop_detected] }
|
|
195
|
+
canonical_issues = results.select { |r| r[:canonical_redirects] || r[:canonical_loop] }
|
|
196
|
+
total_equity_lost = results.sum { |r| r[:equity_loss_pct] }
|
|
197
|
+
avg_equity_retention = total_audited > 0 ? (results.sum { |r| r[:equity_retention_pct] } / total_audited).round(1) : 100.0
|
|
198
|
+
total_clicks_at_risk = results.sum { |r| r.dig(:traffic_at_risk, :total_clicks) || 0 }
|
|
199
|
+
total_impressions_at_risk = results.sum { |r| r.dig(:traffic_at_risk, :total_impressions) || 0 }
|
|
200
|
+
|
|
201
|
+
{
|
|
202
|
+
total_audited: total_audited,
|
|
203
|
+
redirect_chains_count: chains.size,
|
|
204
|
+
loops_count: loops.size,
|
|
205
|
+
canonical_issues_count: canonical_issues.size,
|
|
206
|
+
avg_equity_retention_pct: avg_equity_retention,
|
|
207
|
+
total_clicks_at_risk: total_clicks_at_risk,
|
|
208
|
+
total_impressions_at_risk: total_impressions_at_risk,
|
|
209
|
+
results: results.sort_by { |r| [r[:loop_detected] ? 0 : 1, -r[:redirect_hops], -r[:equity_loss_pct]] }
|
|
210
|
+
}
|
|
211
|
+
end
|
|
212
|
+
|
|
213
|
+
# Generate server redirect collapse rules (Nginx, Apache, Cloudflare, Vercel/Netlify)
|
|
214
|
+
def generate_collapse_rules(from_url, to_url)
|
|
215
|
+
from_uri = URI.parse(from_url) rescue nil
|
|
216
|
+
to_uri = URI.parse(to_url) rescue nil
|
|
217
|
+
return {} unless from_uri && to_uri
|
|
218
|
+
|
|
219
|
+
from_path = from_uri.path.empty? ? '/' : from_uri.path
|
|
220
|
+
from_path += "?#{from_uri.query}" if from_uri.query
|
|
221
|
+
|
|
222
|
+
to_target = to_url
|
|
223
|
+
|
|
224
|
+
nginx_rule = "rewrite ^#{Regexp.escape(from_uri.path)}/?$ #{to_target} permanent;"
|
|
225
|
+
apache_rule = "RewriteRule ^#{from_uri.path.sub(%r{^/}, '')}/?$ #{to_target} [R=301,L]"
|
|
226
|
+
cloudflare_expr = "(http.request.full_uri eq \"#{from_url}\") -> 301 to #{to_target}"
|
|
227
|
+
vercel_netlify = "#{from_path} #{to_target} 301!"
|
|
228
|
+
|
|
229
|
+
{
|
|
230
|
+
nginx: nginx_rule,
|
|
231
|
+
apache: apache_rule,
|
|
232
|
+
cloudflare: cloudflare_expr,
|
|
233
|
+
vercel_netlify: vercel_netlify
|
|
234
|
+
}
|
|
235
|
+
end
|
|
236
|
+
|
|
237
|
+
private
|
|
238
|
+
|
|
239
|
+
def normalize_url(url)
|
|
240
|
+
u = url.to_s.strip
|
|
241
|
+
u = "https://#{u}" unless u =~ %r{^https?://}
|
|
242
|
+
u
|
|
243
|
+
end
|
|
244
|
+
|
|
245
|
+
def fetch_hop(url, hop_num)
|
|
246
|
+
uri = URI.parse(url) rescue nil
|
|
247
|
+
return { hop: hop_num, url: url, error: "Invalid URI: #{url}", duration_ms: 0, status_code: 0 } unless uri && uri.host
|
|
248
|
+
|
|
249
|
+
start_t = Time.now
|
|
250
|
+
res = nil
|
|
251
|
+
body_content = ''
|
|
252
|
+
|
|
253
|
+
begin
|
|
254
|
+
Net::HTTP.start(uri.hostname, uri.port, use_ssl: (uri.scheme == 'https'), open_timeout: @timeout, read_timeout: @timeout) do |http|
|
|
255
|
+
req = Net::HTTP::Get.new(uri.request_uri)
|
|
256
|
+
req['User-Agent'] = @user_agent
|
|
257
|
+
req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
|
|
258
|
+
res = http.request(req)
|
|
259
|
+
body_content = res.body.to_s[0..25_000] if res.body # read first 25KB for head tags
|
|
260
|
+
end
|
|
261
|
+
rescue StandardError => e
|
|
262
|
+
return {
|
|
263
|
+
hop: hop_num,
|
|
264
|
+
url: url,
|
|
265
|
+
status_code: 0,
|
|
266
|
+
duration_ms: ((Time.now - start_t) * 1000).round(1),
|
|
267
|
+
error: e.message
|
|
268
|
+
}
|
|
269
|
+
end
|
|
270
|
+
|
|
271
|
+
duration_ms = ((Time.now - start_t) * 1000).round(1)
|
|
272
|
+
status_code = res.code.to_i
|
|
273
|
+
location = res['location']
|
|
274
|
+
redirect_target = nil
|
|
275
|
+
|
|
276
|
+
if [301, 302, 303, 307, 308].include?(status_code) && location
|
|
277
|
+
redirect_target = URI.join(url, location).to_s rescue location
|
|
278
|
+
end
|
|
279
|
+
|
|
280
|
+
# Extract canonical from header and HTML body
|
|
281
|
+
canonical_header = extract_canonical_header(res['link'])
|
|
282
|
+
canonical_html = extract_canonical_html(body_content, url)
|
|
283
|
+
canonical_url = canonical_header || canonical_html
|
|
284
|
+
|
|
285
|
+
{
|
|
286
|
+
hop: hop_num,
|
|
287
|
+
url: url,
|
|
288
|
+
status_code: status_code,
|
|
289
|
+
duration_ms: duration_ms,
|
|
290
|
+
is_redirect: !redirect_target.nil?,
|
|
291
|
+
redirect_target: redirect_target,
|
|
292
|
+
canonical_url: canonical_url,
|
|
293
|
+
canonical_source: canonical_header ? :http_header : (canonical_html ? :html_tag : nil),
|
|
294
|
+
x_robots_tag: res['x-robots-tag'],
|
|
295
|
+
server: res['server'],
|
|
296
|
+
content_type: res['content-type']
|
|
297
|
+
}
|
|
298
|
+
end
|
|
299
|
+
|
|
300
|
+
def extract_canonical_header(link_header)
|
|
301
|
+
return nil unless link_header
|
|
302
|
+
if link_header =~ /<([^>]+)>;\s*rel=["']?canonical["']?/i
|
|
303
|
+
$1.strip
|
|
304
|
+
end
|
|
305
|
+
end
|
|
306
|
+
|
|
307
|
+
def extract_canonical_html(html, base_url)
|
|
308
|
+
return nil unless html && !html.empty?
|
|
309
|
+
# Case-insensitive scan for <link ... rel="canonical" ... href="..." >
|
|
310
|
+
if html =~ /<link\b[^>]*\brel=["']canonical["'][^>]*>/i
|
|
311
|
+
tag = $&
|
|
312
|
+
if tag =~ /href=["']([^"']+)["']/i
|
|
313
|
+
raw_href = $1.strip
|
|
314
|
+
URI.join(base_url, raw_href).to_s rescue raw_href
|
|
315
|
+
end
|
|
316
|
+
elsif html =~ /<link\b[^>]*\bhref=["']([^"']+)["'][^>]*\brel=["']canonical["'][^>]*>/i
|
|
317
|
+
raw_href = $1.strip
|
|
318
|
+
URI.join(base_url, raw_href).to_s rescue raw_href
|
|
319
|
+
end
|
|
320
|
+
end
|
|
321
|
+
|
|
322
|
+
def check_canonical_target(canonical_url)
|
|
323
|
+
uri = URI.parse(canonical_url) rescue nil
|
|
324
|
+
return { status_code: 0, is_redirect: false } unless uri && uri.host
|
|
325
|
+
|
|
326
|
+
begin
|
|
327
|
+
Net::HTTP.start(uri.hostname, uri.port, use_ssl: (uri.scheme == 'https'), open_timeout: 4, read_timeout: 5) do |http|
|
|
328
|
+
req = Net::HTTP::Head.new(uri.request_uri)
|
|
329
|
+
req['User-Agent'] = @user_agent
|
|
330
|
+
res = http.request(req)
|
|
331
|
+
code = res.code.to_i
|
|
332
|
+
return { status_code: code, is_redirect: [301, 302, 303, 307, 308].include?(code) }
|
|
333
|
+
end
|
|
334
|
+
rescue StandardError
|
|
335
|
+
{ status_code: 0, is_redirect: false }
|
|
336
|
+
end
|
|
337
|
+
end
|
|
338
|
+
|
|
339
|
+
def extract_gsc_traffic(urls, gsc_rows)
|
|
340
|
+
return { total_clicks: 0, total_impressions: 0, queries: [] } unless gsc_rows && gsc_rows.is_a?(Array)
|
|
341
|
+
|
|
342
|
+
clean_set = urls.map { |u| normalize_url(u).downcase }.to_set
|
|
343
|
+
matched_clicks = 0
|
|
344
|
+
matched_impressions = 0
|
|
345
|
+
matched_queries = []
|
|
346
|
+
|
|
347
|
+
gsc_rows.each do |r|
|
|
348
|
+
keys = r['keys'] || []
|
|
349
|
+
page = keys.find { |k| k =~ %r{^https?://} }
|
|
350
|
+
next unless page && clean_set.include?(normalize_url(page).downcase)
|
|
351
|
+
|
|
352
|
+
clicks = (r['clicks'] || 0).to_i
|
|
353
|
+
impressions = (r['impressions'] || 0).to_i
|
|
354
|
+
matched_clicks += clicks
|
|
355
|
+
matched_impressions += impressions
|
|
356
|
+
query = keys.find { |k| k !~ %r{^https?://} }
|
|
357
|
+
matched_queries << { query: query, clicks: clicks, impressions: impressions } if query
|
|
358
|
+
end
|
|
359
|
+
|
|
360
|
+
{
|
|
361
|
+
total_clicks: matched_clicks,
|
|
362
|
+
total_impressions: matched_impressions,
|
|
363
|
+
queries: matched_queries.sort_by { |q| -q[:clicks] }.first(5)
|
|
364
|
+
}
|
|
365
|
+
end
|
|
366
|
+
end
|
|
367
|
+
end
|