gsc-cli 2.1.0 → 2.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/AUTH.md +4 -1
- data/README.md +448 -408
- data/bin/gsc +28067 -5661
- data/dist/gsc +29121 -5046
- data/lib/gsc/aio_hunter.rb +343 -0
- data/lib/gsc/answer_synthesizer.rb +157 -0
- data/lib/gsc/api.rb +53 -1
- data/lib/gsc/auth.rb +26 -0
- data/lib/gsc/brand_segmenter.rb +140 -0
- data/lib/gsc/cache_manager.rb +806 -0
- data/lib/gsc/cannibalization_analyzer.rb +141 -0
- data/lib/gsc/canonical_chains.rb +367 -0
- data/lib/gsc/citation_simulator.rb +339 -0
- data/lib/gsc/cli/aio_hunter.rb +154 -0
- data/lib/gsc/cli/analytics.rb +788 -0
- data/lib/gsc/cli/audit.rb +1976 -0
- data/lib/gsc/cli/base.rb +384 -0
- data/lib/gsc/cli/cache.rb +266 -0
- data/lib/gsc/cli/canonical.rb +223 -0
- data/lib/gsc/cli/citation_simulator.rb +152 -0
- data/lib/gsc/cli/dashboard.rb +354 -0
- data/lib/gsc/cli/doctor.rb +129 -0
- data/lib/gsc/cli/eeat.rb +125 -0
- data/lib/gsc/cli/ga4.rb +852 -0
- data/lib/gsc/cli/growth.rb +650 -0
- data/lib/gsc/cli/hreflang.rb +164 -0
- data/lib/gsc/cli/image_seo.rb +162 -0
- data/lib/gsc/cli/indexing.rb +458 -0
- data/lib/gsc/cli/intent_shift.rb +125 -0
- data/lib/gsc/cli/keyword_value.rb +134 -0
- data/lib/gsc/cli/keywords.rb +795 -0
- data/lib/gsc/cli/landing_roi.rb +308 -0
- data/lib/gsc/cli/low_ctr.rb +213 -0
- data/lib/gsc/cli/mobile_parity.rb +150 -0
- data/lib/gsc/cli/report.rb +100 -0
- data/lib/gsc/cli/rich_results.rb +172 -0
- data/lib/gsc/cli/schema_generate.rb +149 -0
- data/lib/gsc/cli/seasonal.rb +232 -0
- data/lib/gsc/cli/security.rb +153 -0
- data/lib/gsc/cli/setup.rb +1291 -0
- data/lib/gsc/cli/sitemap_tree.rb +143 -0
- data/lib/gsc/cli/skill_pack.rb +62 -0
- data/lib/gsc/cli/soft_404.rb +199 -0
- data/lib/gsc/cli/sparkline.rb +227 -0
- data/lib/gsc/cli/watchdog.rb +150 -0
- data/lib/gsc/cli/zombie_purger.rb +208 -0
- data/lib/gsc/cli.rb +707 -5265
- data/lib/gsc/cli_advanced.rb +987 -44
- data/lib/gsc/client.rb +17 -2
- data/lib/gsc/color.rb +16 -1
- data/lib/gsc/command_registry.rb +47 -9
- data/lib/gsc/config.rb +2 -2
- data/lib/gsc/ctr_curve.rb +115 -0
- data/lib/gsc/decay_predictor.rb +322 -0
- data/lib/gsc/doctor.rb +434 -0
- data/lib/gsc/eeat_auditor.rb +428 -0
- data/lib/gsc/entity_auditor.rb +229 -0
- data/lib/gsc/firewall_scanner.rb +733 -0
- data/lib/gsc/geo_auditor.rb +368 -0
- data/lib/gsc/google_trends.rb +8 -1
- data/lib/gsc/heading_validator.rb +283 -0
- data/lib/gsc/hreflang_validator.rb +412 -0
- data/lib/gsc/image_seo.rb +286 -0
- data/lib/gsc/indexing_queue.rb +179 -0
- data/lib/gsc/indexnow.rb +93 -0
- data/lib/gsc/intent_shift.rb +188 -0
- data/lib/gsc/internal_links.rb +153 -36
- data/lib/gsc/keyword_value.rb +191 -0
- data/lib/gsc/landing_roi.rb +195 -0
- data/lib/gsc/llms_generator.rb +343 -22
- data/lib/gsc/low_ctr_rewriter.rb +408 -0
- data/lib/gsc/mobile_parity.rb +222 -0
- data/lib/gsc/network_tracer.rb +8 -1
- data/lib/gsc/page_analyzer.rb +47 -7
- data/lib/gsc/prompts.rb +38 -29
- data/lib/gsc/questions_harvester.rb +178 -0
- data/lib/gsc/report_generator.rb +461 -0
- data/lib/gsc/rich_results.rb +388 -0
- data/lib/gsc/robots_checker.rb +46 -15
- data/lib/gsc/schema_generator.rb +788 -0
- data/lib/gsc/schema_validator.rb +36 -38
- data/lib/gsc/seasonal_predictor.rb +381 -0
- data/lib/gsc/security_scanner.rb +496 -0
- data/lib/gsc/serp_feature_detector.rb +359 -0
- data/lib/gsc/serp_preview.rb +108 -22
- data/lib/gsc/site_crawler.rb +113 -21
- data/lib/gsc/sitemap_loader.rb +15 -4
- data/lib/gsc/sitemap_tree.rb +301 -0
- data/lib/gsc/skill_pack.rb +195 -0
- data/lib/gsc/soft_404_analyzer.rb +385 -0
- data/lib/gsc/sparkline.rb +171 -0
- data/lib/gsc/speed_correlator.rb +416 -0
- data/lib/gsc/striking_playbook.rb +190 -0
- data/lib/gsc/title_optimizer.rb +420 -0
- data/lib/gsc/vault.rb +260 -0
- data/lib/gsc/version.rb +1 -1
- data/lib/gsc/watchdog.rb +235 -0
- data/lib/gsc/zombie_purger.rb +366 -0
- data/lib/gsc.rb +118 -0
- metadata +75 -1
data/lib/gsc/watchdog.rb
ADDED
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'json'
|
|
5
|
+
require 'net/http'
|
|
6
|
+
require 'uri'
|
|
7
|
+
require 'date'
|
|
8
|
+
|
|
9
|
+
module GSC
|
|
10
|
+
class Watchdog
|
|
11
|
+
DEFAULT_DROP_THRESHOLD = 2.0 # 2+ rank position drop
|
|
12
|
+
DEFAULT_CTR_THRESHOLD = 20.0 # 20%+ CTR decrease
|
|
13
|
+
|
|
14
|
+
attr_reader :options, :api, :domain
|
|
15
|
+
|
|
16
|
+
def initialize(options = {}, api = nil, domain = nil)
|
|
17
|
+
@options = options
|
|
18
|
+
@api = api
|
|
19
|
+
@domain = domain.to_s.strip
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
def self.check(options = {}, api = nil, domain = nil)
|
|
23
|
+
new(options, api, domain).check
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
def check
|
|
27
|
+
telemetry = fetch_telemetry
|
|
28
|
+
alerts = detect_anomalies(telemetry)
|
|
29
|
+
summary = synthesize_summary(telemetry, alerts)
|
|
30
|
+
|
|
31
|
+
if @options[:webhook] || @options[:slack]
|
|
32
|
+
webhook_res = dispatch_webhook(summary, @options[:webhook] || @options[:slack])
|
|
33
|
+
summary[:webhook_dispatched] = webhook_res
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
summary
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def generate_crontab_entry
|
|
40
|
+
bin = current_binary_path
|
|
41
|
+
hours = (@options[:hours] || 6).to_i
|
|
42
|
+
webhook_flag = @options[:webhook] ? " --webhook #{@options[:webhook]}" : ""
|
|
43
|
+
"0 */#{hours} * * * #{bin} watch #{@domain} --once#{webhook_flag} >> ~/.config/gsc/watch.log 2>&1"
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
def generate_launchd_plist
|
|
47
|
+
bin = current_binary_path
|
|
48
|
+
interval = (@options[:interval] || 21600).to_i
|
|
49
|
+
label = "com.gsc.watchdog.#{@domain.gsub(/[^a-zA-Z0-9]/, '_')}"
|
|
50
|
+
|
|
51
|
+
<<~XML
|
|
52
|
+
<?xml version="1.0" encoding="UTF-8"?>
|
|
53
|
+
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
|
|
54
|
+
<plist version="1.0">
|
|
55
|
+
<dict>
|
|
56
|
+
<key>Label</key>
|
|
57
|
+
<string>#{label}</string>
|
|
58
|
+
<key>ProgramArguments</key>
|
|
59
|
+
<array>
|
|
60
|
+
<string>#{bin}</string>
|
|
61
|
+
<string>watch</string>
|
|
62
|
+
<string>#{@domain}</string>
|
|
63
|
+
<string>--once</string>
|
|
64
|
+
</array>
|
|
65
|
+
<key>StartInterval</key>
|
|
66
|
+
<integer>#{interval}</integer>
|
|
67
|
+
<key>StandardOutPath</key>
|
|
68
|
+
<string>/tmp/gsc_watch_#{@domain}.log</string>
|
|
69
|
+
<key>StandardErrorPath</key>
|
|
70
|
+
<string>/tmp/gsc_watch_#{@domain}_err.log</string>
|
|
71
|
+
</dict>
|
|
72
|
+
</plist>
|
|
73
|
+
XML
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
def generate_systemd_unit
|
|
77
|
+
bin = current_binary_path
|
|
78
|
+
<<~INI
|
|
79
|
+
[Unit]
|
|
80
|
+
Description=GSC Continuous SEO Rank & CTR Volatility Watchdog for #{@domain}
|
|
81
|
+
After=network.target
|
|
82
|
+
|
|
83
|
+
[Service]
|
|
84
|
+
Type=oneshot
|
|
85
|
+
ExecStart=#{bin} watch #{@domain} --once
|
|
86
|
+
StandardOutput=journal
|
|
87
|
+
StandardError=journal
|
|
88
|
+
|
|
89
|
+
[Install]
|
|
90
|
+
WantedBy=multi-user.target
|
|
91
|
+
INI
|
|
92
|
+
end
|
|
93
|
+
|
|
94
|
+
private
|
|
95
|
+
|
|
96
|
+
def current_binary_path
|
|
97
|
+
ENV['GSC_BIN_PATH'] || File.expand_path('~/.local/bin/gsc')
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
def fetch_telemetry
|
|
101
|
+
if @api
|
|
102
|
+
begin
|
|
103
|
+
curr_end = (Date.today - 2).strftime('%Y-%m-%d')
|
|
104
|
+
curr_start = (Date.today - 9).strftime('%Y-%m-%d')
|
|
105
|
+
prev_end = (Date.today - 10).strftime('%Y-%m-%d')
|
|
106
|
+
prev_start = (Date.today - 17).strftime('%Y-%m-%d')
|
|
107
|
+
|
|
108
|
+
curr_res = @api.search_analytics(@domain, start_date: curr_start, end_date: curr_end, dimensions: %w[query])
|
|
109
|
+
prev_res = @api.search_analytics(@domain, start_date: prev_start, end_date: prev_end, dimensions: %w[query])
|
|
110
|
+
|
|
111
|
+
curr_rows = (curr_res['rows'] || []).each_with_object({}) { |r, h| h[r['keys'][0]] = r }
|
|
112
|
+
prev_rows = (prev_res['rows'] || []).each_with_object({}) { |r, h| h[r['keys'][0]] = r }
|
|
113
|
+
|
|
114
|
+
all_queries = (curr_rows.keys + prev_rows.keys).uniq
|
|
115
|
+
|
|
116
|
+
if all_queries.any?
|
|
117
|
+
return all_queries.first(25).map do |q|
|
|
118
|
+
cr = curr_rows[q]
|
|
119
|
+
pr = prev_rows[q]
|
|
120
|
+
|
|
121
|
+
curr_pos = cr ? (cr['position'] || 0).round(1) : 100.0
|
|
122
|
+
prev_pos = pr ? (pr['position'] || 0).round(1) : 100.0
|
|
123
|
+
curr_clicks = cr ? (cr['clicks'] || 0) : 0
|
|
124
|
+
prev_clicks = pr ? (pr['clicks'] || 0) : 0
|
|
125
|
+
curr_ctr = cr ? ((cr['ctr'] || 0) * 100.0).round(2) : 0.0
|
|
126
|
+
prev_ctr = pr ? ((pr['ctr'] || 0) * 100.0).round(2) : 0.0
|
|
127
|
+
|
|
128
|
+
{
|
|
129
|
+
query: q,
|
|
130
|
+
current_position: curr_pos,
|
|
131
|
+
previous_position: prev_pos,
|
|
132
|
+
current_clicks: curr_clicks,
|
|
133
|
+
previous_clicks: prev_clicks,
|
|
134
|
+
current_ctr: curr_ctr,
|
|
135
|
+
previous_ctr: prev_ctr
|
|
136
|
+
}
|
|
137
|
+
end
|
|
138
|
+
end
|
|
139
|
+
rescue StandardError
|
|
140
|
+
# return empty
|
|
141
|
+
end
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
[]
|
|
145
|
+
end
|
|
146
|
+
|
|
147
|
+
def detect_anomalies(telemetry)
|
|
148
|
+
alerts = []
|
|
149
|
+
pos_threshold = (@options[:threshold] || DEFAULT_DROP_THRESHOLD).to_f
|
|
150
|
+
ctr_threshold = DEFAULT_CTR_THRESHOLD
|
|
151
|
+
|
|
152
|
+
telemetry.each do |row|
|
|
153
|
+
pos_delta = (row[:current_position] - row[:previous_position]).round(1) # positive means dropped rank
|
|
154
|
+
ctr_drop_pct = row[:previous_ctr] > 0 ? (((row[:previous_ctr] - row[:current_ctr]) / row[:previous_ctr]) * 100.0).round(1) : 0.0
|
|
155
|
+
|
|
156
|
+
if pos_delta >= pos_threshold
|
|
157
|
+
severity = pos_delta >= 4.0 ? :critical : :warning
|
|
158
|
+
alerts << {
|
|
159
|
+
query: row[:query],
|
|
160
|
+
type: :position_drop,
|
|
161
|
+
severity: severity,
|
|
162
|
+
message: "Rank dropped #{pos_delta} positions (Pos #{row[:previous_position]} ➔ Pos #{row[:current_position]})",
|
|
163
|
+
current_position: row[:current_position],
|
|
164
|
+
previous_position: row[:previous_position],
|
|
165
|
+
current_clicks: row[:current_clicks],
|
|
166
|
+
previous_clicks: row[:previous_clicks]
|
|
167
|
+
}
|
|
168
|
+
elsif ctr_drop_pct >= ctr_threshold && row[:current_position] <= 10.0
|
|
169
|
+
alerts << {
|
|
170
|
+
query: row[:query],
|
|
171
|
+
type: :ctr_anomaly,
|
|
172
|
+
severity: :warning,
|
|
173
|
+
message: "CTR dropped #{ctr_drop_pct}% (from #{row[:previous_ctr]}% to #{row[:current_ctr]}%) despite stable rank (Pos #{row[:current_position]}). Likely Zero-Click AI Overview suppression.",
|
|
174
|
+
current_position: row[:current_position],
|
|
175
|
+
previous_position: row[:previous_position],
|
|
176
|
+
current_clicks: row[:current_clicks],
|
|
177
|
+
previous_clicks: row[:previous_clicks]
|
|
178
|
+
}
|
|
179
|
+
end
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
alerts
|
|
183
|
+
end
|
|
184
|
+
|
|
185
|
+
def synthesize_summary(telemetry, alerts)
|
|
186
|
+
critical_count = alerts.count { |a| a[:severity] == :critical }
|
|
187
|
+
warning_count = alerts.count { |a| a[:severity] == :warning }
|
|
188
|
+
|
|
189
|
+
status = if critical_count > 0
|
|
190
|
+
'CRITICAL ANOMALIES'
|
|
191
|
+
elsif warning_count > 0
|
|
192
|
+
'WARNINGS DETECTED'
|
|
193
|
+
else
|
|
194
|
+
'SERP HEALTH OPTIMAL'
|
|
195
|
+
end
|
|
196
|
+
|
|
197
|
+
{
|
|
198
|
+
domain: @domain,
|
|
199
|
+
monitored_queries_count: telemetry.size,
|
|
200
|
+
status: status,
|
|
201
|
+
critical_alerts_count: critical_count,
|
|
202
|
+
warning_alerts_count: warning_count,
|
|
203
|
+
total_alerts_count: alerts.size,
|
|
204
|
+
checked_at: Time.now.strftime('%Y-%m-%d %H:%M:%S %Z'),
|
|
205
|
+
alerts: alerts,
|
|
206
|
+
telemetry: telemetry
|
|
207
|
+
}
|
|
208
|
+
end
|
|
209
|
+
|
|
210
|
+
def dispatch_webhook(summary, webhook_url)
|
|
211
|
+
uri = URI.parse(webhook_url)
|
|
212
|
+
payload = {
|
|
213
|
+
text: "🚨 [GSC Rank Watchdog] #{summary[:total_alerts_count]} Organic SERP Alerts Detected for #{summary[:domain]} (#{summary[:status]})",
|
|
214
|
+
domain: summary[:domain],
|
|
215
|
+
status: summary[:status],
|
|
216
|
+
critical_count: summary[:critical_alerts_count],
|
|
217
|
+
warnings_count: summary[:warning_alerts_count],
|
|
218
|
+
checked_at: summary[:checked_at],
|
|
219
|
+
alerts: summary[:alerts].map { |a| { query: a[:query], message: a[:message], severity: a[:severity] } }
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
req = Net::HTTP::Post.new(uri)
|
|
223
|
+
req['Content-Type'] = 'application/json'
|
|
224
|
+
req.body = JSON.generate(payload)
|
|
225
|
+
|
|
226
|
+
res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 5, read_timeout: 5) do |http|
|
|
227
|
+
http.request(req)
|
|
228
|
+
end
|
|
229
|
+
|
|
230
|
+
res.code.to_i >= 200 && res.code.to_i < 300
|
|
231
|
+
rescue StandardError => e
|
|
232
|
+
false
|
|
233
|
+
end
|
|
234
|
+
end
|
|
235
|
+
end
|
|
@@ -0,0 +1,366 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'json'
|
|
5
|
+
require 'uri'
|
|
6
|
+
|
|
7
|
+
module GSC
|
|
8
|
+
class ZombiePurger
|
|
9
|
+
DEFAULT_DAYS = 90
|
|
10
|
+
DEFAULT_THRESHOLD = 0
|
|
11
|
+
EST_MONTHLY_CRAWLS_PER_ZOMBIE = 6
|
|
12
|
+
|
|
13
|
+
# Common utility/legal URL path fragments that should remain accessible to users but noindexed
|
|
14
|
+
UTILITY_PATHS = %w[
|
|
15
|
+
/privacy /terms /disclaimer /legal /contact /about-us /about
|
|
16
|
+
/account /login /signin /signup /register /cart /checkout
|
|
17
|
+
/order /thank-you /confirmation /search /sitemap
|
|
18
|
+
].freeze
|
|
19
|
+
|
|
20
|
+
# Common taxonomy, archive, or thin content fragments suited for 410 purge
|
|
21
|
+
THIN_PATTERNS = [
|
|
22
|
+
%r{/page/\d+},
|
|
23
|
+
%r{/tag/},
|
|
24
|
+
%r{/tags/},
|
|
25
|
+
%r{/category/},
|
|
26
|
+
%r{/archive/},
|
|
27
|
+
%r{/author/},
|
|
28
|
+
%r{/attachment/},
|
|
29
|
+
%r{\?replytocom=},
|
|
30
|
+
%r{\?amp=1},
|
|
31
|
+
%r{/feed/?$}
|
|
32
|
+
].freeze
|
|
33
|
+
|
|
34
|
+
STOPWORDS = %w[the a an and of to in for with on at by from is are was were this that].freeze
|
|
35
|
+
|
|
36
|
+
attr_reader :options, :days, :threshold
|
|
37
|
+
|
|
38
|
+
def initialize(options = {})
|
|
39
|
+
@options = options
|
|
40
|
+
@days = (options[:days] || DEFAULT_DAYS).to_i
|
|
41
|
+
@threshold = (options[:threshold] || DEFAULT_THRESHOLD).to_i
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
# Convenience class method for auditing inventory against GSC pages
|
|
45
|
+
def self.audit(inventory_urls, gsc_pages, options = {})
|
|
46
|
+
new(options).audit(inventory_urls, gsc_pages)
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
# inventory_urls: Array of URL strings
|
|
50
|
+
# gsc_pages: Hash of { clean_url => { impressions:, clicks:, position: } } OR Array of row hashes
|
|
51
|
+
def audit(inventory_urls, gsc_pages)
|
|
52
|
+
urls = normalize_url_list(inventory_urls)
|
|
53
|
+
active_map = normalize_gsc_pages(gsc_pages)
|
|
54
|
+
|
|
55
|
+
active_items = []
|
|
56
|
+
zombie_items = []
|
|
57
|
+
|
|
58
|
+
# List of all active clean URLs for candidate pillar matching
|
|
59
|
+
active_candidates = active_map.keys
|
|
60
|
+
|
|
61
|
+
urls.each do |raw_url|
|
|
62
|
+
clean_key = clean_url_key(raw_url)
|
|
63
|
+
page_stat = active_map[clean_key]
|
|
64
|
+
impressions = page_stat ? (page_stat[:impressions] || 0) : 0
|
|
65
|
+
clicks = page_stat ? (page_stat[:clicks] || 0) : 0
|
|
66
|
+
position = page_stat ? (page_stat[:position] || 0.0) : 0.0
|
|
67
|
+
|
|
68
|
+
if impressions > @threshold
|
|
69
|
+
active_items << {
|
|
70
|
+
url: raw_url,
|
|
71
|
+
clean_url: clean_key,
|
|
72
|
+
impressions: impressions,
|
|
73
|
+
clicks: clicks,
|
|
74
|
+
position: position
|
|
75
|
+
}
|
|
76
|
+
else
|
|
77
|
+
# Triage this zombie URL
|
|
78
|
+
triage = classify_zombie(raw_url, active_candidates)
|
|
79
|
+
zombie_items << {
|
|
80
|
+
url: raw_url,
|
|
81
|
+
clean_url: clean_key,
|
|
82
|
+
impressions: impressions,
|
|
83
|
+
clicks: clicks,
|
|
84
|
+
action: triage[:action],
|
|
85
|
+
action_label: format_action_label(triage[:action]),
|
|
86
|
+
target_url: triage[:target_url],
|
|
87
|
+
rationale: triage[:rationale],
|
|
88
|
+
path: uri_path(raw_url)
|
|
89
|
+
}
|
|
90
|
+
end
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
total_inventory = urls.size
|
|
94
|
+
zombie_count = zombie_items.size
|
|
95
|
+
active_count = active_items.size
|
|
96
|
+
|
|
97
|
+
zombie_ratio = total_inventory > 0 ? ((zombie_count.to_f / total_inventory) * 100.0).round(1) : 0.0
|
|
98
|
+
cedi = zombie_ratio # Crawl Equity Dilution Index (0-100, lower is better)
|
|
99
|
+
health_score = [0.0, (100.0 - cedi)].max.round(1)
|
|
100
|
+
grade = compute_grade(cedi)
|
|
101
|
+
|
|
102
|
+
wasted_crawls_monthly = zombie_count * EST_MONTHLY_CRAWLS_PER_ZOMBIE
|
|
103
|
+
wasted_crawls_annually = wasted_crawls_monthly * 12
|
|
104
|
+
|
|
105
|
+
action_breakdown = {
|
|
106
|
+
purge_410: zombie_items.count { |z| z[:action] == :purge_410 },
|
|
107
|
+
redirect_301: zombie_items.count { |z| z[:action] == :redirect_301 },
|
|
108
|
+
consolidate: zombie_items.count { |z| z[:action] == :consolidate },
|
|
109
|
+
noindex: zombie_items.count { |z| z[:action] == :noindex }
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
# Filter if specific action requested
|
|
113
|
+
filtered_zombies = if @options[:action]
|
|
114
|
+
target_act = @options[:action].to_s.downcase.sub('301', '').sub('410', '').to_sym
|
|
115
|
+
zombie_items.select { |z| z[:action].to_s.include?(target_act.to_s) }
|
|
116
|
+
else
|
|
117
|
+
zombie_items
|
|
118
|
+
end
|
|
119
|
+
|
|
120
|
+
limit = (@options[:limit] || 25).to_i
|
|
121
|
+
displayed_zombies = filtered_zombies.first(limit)
|
|
122
|
+
|
|
123
|
+
{
|
|
124
|
+
total_inventory: total_inventory,
|
|
125
|
+
active_count: active_count,
|
|
126
|
+
zombie_count: zombie_count,
|
|
127
|
+
zombie_percentage: zombie_ratio,
|
|
128
|
+
cedi: cedi,
|
|
129
|
+
health_score: health_score,
|
|
130
|
+
grade: grade,
|
|
131
|
+
days: @days,
|
|
132
|
+
threshold: @threshold,
|
|
133
|
+
wasted_crawls_monthly: wasted_crawls_monthly,
|
|
134
|
+
wasted_crawls_annually: wasted_crawls_annually,
|
|
135
|
+
action_breakdown: action_breakdown,
|
|
136
|
+
zombies: displayed_zombies,
|
|
137
|
+
all_zombies_count: filtered_zombies.size,
|
|
138
|
+
server_rules: {
|
|
139
|
+
nginx: generate_nginx(displayed_zombies),
|
|
140
|
+
htaccess: generate_htaccess(displayed_zombies),
|
|
141
|
+
redirects: generate_redirects(displayed_zombies),
|
|
142
|
+
meta_robots: generate_meta_robots
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
end
|
|
146
|
+
|
|
147
|
+
# Classifies a single zombie URL into a strategic action
|
|
148
|
+
def classify_zombie(url, active_candidates)
|
|
149
|
+
path = uri_path(url).downcase
|
|
150
|
+
|
|
151
|
+
# 1. Check if it's a utility or legal page that should be noindexed
|
|
152
|
+
if UTILITY_PATHS.any? { |up| path.start_with?(up) || path == up }
|
|
153
|
+
return {
|
|
154
|
+
action: :noindex,
|
|
155
|
+
target_url: nil,
|
|
156
|
+
rationale: "Utility/legal page necessary for user experience; noindex to preserve crawl equity."
|
|
157
|
+
}
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
# 2. Check for thin taxonomy, pagination, or feed loops
|
|
161
|
+
if THIN_PATTERNS.any? { |pat| path =~ pat }
|
|
162
|
+
return {
|
|
163
|
+
action: :purge_410,
|
|
164
|
+
target_url: nil,
|
|
165
|
+
rationale: "Thin taxonomy/archive/pagination loop; return 410 Gone to immediately drop from Googlebot index."
|
|
166
|
+
}
|
|
167
|
+
end
|
|
168
|
+
|
|
169
|
+
# 3. Check for candidate active pillar target to redirect equity
|
|
170
|
+
candidate = find_candidate_target(url, active_candidates)
|
|
171
|
+
if candidate
|
|
172
|
+
return {
|
|
173
|
+
action: :redirect_301,
|
|
174
|
+
target_url: candidate,
|
|
175
|
+
rationale: "Topical slug overlap with active pillar '#{uri_path(candidate)}'; 301 redirect to consolidate equity."
|
|
176
|
+
}
|
|
177
|
+
end
|
|
178
|
+
|
|
179
|
+
# 4. Check for dated content or parameter duplicates
|
|
180
|
+
if path.match?(%r{/\d{4}/\d{2}/}) || path.include?('?')
|
|
181
|
+
return {
|
|
182
|
+
action: :consolidate,
|
|
183
|
+
target_url: nil,
|
|
184
|
+
rationale: "Dated archive or parameter URL; consolidate or rewrite into evergreen canonical."
|
|
185
|
+
}
|
|
186
|
+
end
|
|
187
|
+
|
|
188
|
+
# 5. Default dead zombie: Purge 410
|
|
189
|
+
{
|
|
190
|
+
action: :purge_410,
|
|
191
|
+
target_url: nil,
|
|
192
|
+
rationale: "Zero impressions over #{@days} days with no active pillar match; return 410 Gone to reclaim crawl budget."
|
|
193
|
+
}
|
|
194
|
+
end
|
|
195
|
+
|
|
196
|
+
# Finds an active page with substantial slug keyword overlap
|
|
197
|
+
def find_candidate_target(zombie_url, active_candidates)
|
|
198
|
+
zombie_tokens = extract_slug_tokens(zombie_url)
|
|
199
|
+
return nil if zombie_tokens.empty?
|
|
200
|
+
|
|
201
|
+
best_candidate = nil
|
|
202
|
+
best_score = 0.0
|
|
203
|
+
|
|
204
|
+
active_candidates.each do |act_url|
|
|
205
|
+
act_tokens = extract_slug_tokens(act_url)
|
|
206
|
+
next if act_tokens.empty?
|
|
207
|
+
|
|
208
|
+
common = zombie_tokens & act_tokens
|
|
209
|
+
next if common.empty?
|
|
210
|
+
|
|
211
|
+
# Score based on token overlap
|
|
212
|
+
overlap_ratio = common.size.to_f / [zombie_tokens.size, act_tokens.size].max
|
|
213
|
+
if overlap_ratio > best_score && common.size >= 2
|
|
214
|
+
best_score = overlap_ratio
|
|
215
|
+
best_candidate = act_url
|
|
216
|
+
end
|
|
217
|
+
end
|
|
218
|
+
|
|
219
|
+
best_score >= 0.4 ? best_candidate : nil
|
|
220
|
+
end
|
|
221
|
+
|
|
222
|
+
# Server directive generators
|
|
223
|
+
def generate_nginx(zombies)
|
|
224
|
+
lines = [
|
|
225
|
+
"# ====================================================================",
|
|
226
|
+
"# Nginx Zombie Content Purge Directives (gsc-cli)",
|
|
227
|
+
"# Reclaims Googlebot crawl budget and stops 404 crawl waste",
|
|
228
|
+
"# ===================================================================="
|
|
229
|
+
]
|
|
230
|
+
|
|
231
|
+
zombies.each do |z|
|
|
232
|
+
path = uri_path(z[:url])
|
|
233
|
+
case z[:action]
|
|
234
|
+
when :purge_410
|
|
235
|
+
lines << "location = #{path} { return 410; }"
|
|
236
|
+
when :redirect_301
|
|
237
|
+
target_path = uri_path(z[:target_url] || '/')
|
|
238
|
+
lines << "location = #{path} { return 301 #{target_path}; }"
|
|
239
|
+
end
|
|
240
|
+
end
|
|
241
|
+
lines.join("\n")
|
|
242
|
+
end
|
|
243
|
+
|
|
244
|
+
def generate_htaccess(zombies)
|
|
245
|
+
lines = [
|
|
246
|
+
"# ====================================================================",
|
|
247
|
+
"# Apache .htaccess Zombie Content Purge Directives (gsc-cli)",
|
|
248
|
+
"# ===================================================================="
|
|
249
|
+
]
|
|
250
|
+
|
|
251
|
+
zombies.each do |z|
|
|
252
|
+
path = uri_path(z[:url])
|
|
253
|
+
case z[:action]
|
|
254
|
+
when :purge_410
|
|
255
|
+
lines << "RedirectGone #{path}"
|
|
256
|
+
when :redirect_301
|
|
257
|
+
target_path = uri_path(z[:target_url] || '/')
|
|
258
|
+
lines << "Redirect 301 #{path} #{target_path}"
|
|
259
|
+
end
|
|
260
|
+
end
|
|
261
|
+
lines.join("\n")
|
|
262
|
+
end
|
|
263
|
+
|
|
264
|
+
def generate_redirects(zombies)
|
|
265
|
+
lines = [
|
|
266
|
+
"# ====================================================================",
|
|
267
|
+
"# Netlify / Cloudflare Pages _redirects Rules (gsc-cli)",
|
|
268
|
+
"# ===================================================================="
|
|
269
|
+
]
|
|
270
|
+
|
|
271
|
+
zombies.each do |z|
|
|
272
|
+
path = uri_path(z[:url])
|
|
273
|
+
case z[:action]
|
|
274
|
+
when :purge_410
|
|
275
|
+
lines << "#{path} 410!"
|
|
276
|
+
when :redirect_301
|
|
277
|
+
target_path = uri_path(z[:target_url] || '/')
|
|
278
|
+
lines << "#{path} #{target_path} 301"
|
|
279
|
+
end
|
|
280
|
+
end
|
|
281
|
+
lines.join("\n")
|
|
282
|
+
end
|
|
283
|
+
|
|
284
|
+
def generate_meta_robots
|
|
285
|
+
'<meta name="robots" content="noindex, follow">'
|
|
286
|
+
end
|
|
287
|
+
|
|
288
|
+
private
|
|
289
|
+
|
|
290
|
+
def compute_grade(cedi)
|
|
291
|
+
case cedi
|
|
292
|
+
when 0..10.0 then 'A'
|
|
293
|
+
when 10.1..25.0 then 'B'
|
|
294
|
+
when 25.1..45.0 then 'C'
|
|
295
|
+
when 45.1..70.0 then 'D'
|
|
296
|
+
else 'F'
|
|
297
|
+
end
|
|
298
|
+
end
|
|
299
|
+
|
|
300
|
+
def format_action_label(action)
|
|
301
|
+
case action
|
|
302
|
+
when :purge_410 then 'PURGE (410 GONE)'
|
|
303
|
+
when :redirect_301 then 'REDIRECT (301)'
|
|
304
|
+
when :consolidate then 'CONSOLIDATE'
|
|
305
|
+
when :noindex then 'NOINDEX'
|
|
306
|
+
else action.to_s.upcase
|
|
307
|
+
end
|
|
308
|
+
end
|
|
309
|
+
|
|
310
|
+
def extract_slug_tokens(url)
|
|
311
|
+
path = uri_path(url)
|
|
312
|
+
clean = path.sub(/\.(html?|php|aspx?)$/i, '')
|
|
313
|
+
tokens = clean.split(/[\/\-_]/).map(&:downcase).reject do |t|
|
|
314
|
+
t.empty? || t.match?(/^\d+$/) || STOPWORDS.include?(t)
|
|
315
|
+
end
|
|
316
|
+
tokens.uniq
|
|
317
|
+
end
|
|
318
|
+
|
|
319
|
+
def uri_path(url_str)
|
|
320
|
+
return '/' if url_str.nil? || url_str.empty?
|
|
321
|
+
u = URI.parse(url_str)
|
|
322
|
+
u.path.nil? || u.path.empty? ? '/' : u.path
|
|
323
|
+
rescue URI::InvalidURIError
|
|
324
|
+
url_str.sub(%r{^https?://[^/]+}, '')
|
|
325
|
+
end
|
|
326
|
+
|
|
327
|
+
def clean_url_key(url_str)
|
|
328
|
+
url_str.to_s.strip.downcase.sub(%r{/$}, '')
|
|
329
|
+
end
|
|
330
|
+
|
|
331
|
+
def normalize_url_list(urls)
|
|
332
|
+
Array(urls).map(&:to_s).map(&:strip).reject(&:empty?).uniq
|
|
333
|
+
end
|
|
334
|
+
|
|
335
|
+
def normalize_gsc_pages(gsc_pages)
|
|
336
|
+
map = {}
|
|
337
|
+
if gsc_pages.is_a?(Hash)
|
|
338
|
+
gsc_pages.each do |k, v|
|
|
339
|
+
clean_k = clean_url_key(k)
|
|
340
|
+
if v.is_a?(Hash)
|
|
341
|
+
map[clean_k] = {
|
|
342
|
+
clicks: (v[:clicks] || v['clicks'] || 0).to_i,
|
|
343
|
+
impressions: (v[:impressions] || v['impressions'] || 0).to_i,
|
|
344
|
+
position: (v[:position] || v['position'] || 0.0).to_f
|
|
345
|
+
}
|
|
346
|
+
else
|
|
347
|
+
map[clean_k] = { clicks: 0, impressions: v.to_i, position: 0.0 }
|
|
348
|
+
end
|
|
349
|
+
end
|
|
350
|
+
elsif gsc_pages.is_a?(Array)
|
|
351
|
+
gsc_pages.each do |row|
|
|
352
|
+
next unless row.is_a?(Hash)
|
|
353
|
+
keys = row['keys'] || row[:keys]
|
|
354
|
+
next unless keys && keys.first
|
|
355
|
+
clean_k = clean_url_key(keys.first)
|
|
356
|
+
map[clean_k] = {
|
|
357
|
+
clicks: (row['clicks'] || row[:clicks] || 0).to_i,
|
|
358
|
+
impressions: (row['impressions'] || row[:impressions] || 0).to_i,
|
|
359
|
+
position: (row['position'] || row[:position] || 0.0).to_f
|
|
360
|
+
}
|
|
361
|
+
end
|
|
362
|
+
end
|
|
363
|
+
map
|
|
364
|
+
end
|
|
365
|
+
end
|
|
366
|
+
end
|