gsc-cli 2.1.0 → 2.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. checksums.yaml +4 -4
  2. data/AUTH.md +4 -1
  3. data/README.md +448 -408
  4. data/bin/gsc +28067 -5661
  5. data/dist/gsc +29121 -5046
  6. data/lib/gsc/aio_hunter.rb +343 -0
  7. data/lib/gsc/answer_synthesizer.rb +157 -0
  8. data/lib/gsc/api.rb +53 -1
  9. data/lib/gsc/auth.rb +26 -0
  10. data/lib/gsc/brand_segmenter.rb +140 -0
  11. data/lib/gsc/cache_manager.rb +806 -0
  12. data/lib/gsc/cannibalization_analyzer.rb +141 -0
  13. data/lib/gsc/canonical_chains.rb +367 -0
  14. data/lib/gsc/citation_simulator.rb +339 -0
  15. data/lib/gsc/cli/aio_hunter.rb +154 -0
  16. data/lib/gsc/cli/analytics.rb +788 -0
  17. data/lib/gsc/cli/audit.rb +1976 -0
  18. data/lib/gsc/cli/base.rb +384 -0
  19. data/lib/gsc/cli/cache.rb +266 -0
  20. data/lib/gsc/cli/canonical.rb +223 -0
  21. data/lib/gsc/cli/citation_simulator.rb +152 -0
  22. data/lib/gsc/cli/dashboard.rb +354 -0
  23. data/lib/gsc/cli/doctor.rb +129 -0
  24. data/lib/gsc/cli/eeat.rb +125 -0
  25. data/lib/gsc/cli/ga4.rb +852 -0
  26. data/lib/gsc/cli/growth.rb +650 -0
  27. data/lib/gsc/cli/hreflang.rb +164 -0
  28. data/lib/gsc/cli/image_seo.rb +162 -0
  29. data/lib/gsc/cli/indexing.rb +458 -0
  30. data/lib/gsc/cli/intent_shift.rb +125 -0
  31. data/lib/gsc/cli/keyword_value.rb +134 -0
  32. data/lib/gsc/cli/keywords.rb +795 -0
  33. data/lib/gsc/cli/landing_roi.rb +308 -0
  34. data/lib/gsc/cli/low_ctr.rb +213 -0
  35. data/lib/gsc/cli/mobile_parity.rb +150 -0
  36. data/lib/gsc/cli/report.rb +100 -0
  37. data/lib/gsc/cli/rich_results.rb +172 -0
  38. data/lib/gsc/cli/schema_generate.rb +149 -0
  39. data/lib/gsc/cli/seasonal.rb +232 -0
  40. data/lib/gsc/cli/security.rb +153 -0
  41. data/lib/gsc/cli/setup.rb +1291 -0
  42. data/lib/gsc/cli/sitemap_tree.rb +143 -0
  43. data/lib/gsc/cli/skill_pack.rb +62 -0
  44. data/lib/gsc/cli/soft_404.rb +199 -0
  45. data/lib/gsc/cli/sparkline.rb +227 -0
  46. data/lib/gsc/cli/watchdog.rb +150 -0
  47. data/lib/gsc/cli/zombie_purger.rb +208 -0
  48. data/lib/gsc/cli.rb +707 -5265
  49. data/lib/gsc/cli_advanced.rb +987 -44
  50. data/lib/gsc/client.rb +17 -2
  51. data/lib/gsc/color.rb +16 -1
  52. data/lib/gsc/command_registry.rb +47 -9
  53. data/lib/gsc/config.rb +2 -2
  54. data/lib/gsc/ctr_curve.rb +115 -0
  55. data/lib/gsc/decay_predictor.rb +322 -0
  56. data/lib/gsc/doctor.rb +434 -0
  57. data/lib/gsc/eeat_auditor.rb +428 -0
  58. data/lib/gsc/entity_auditor.rb +229 -0
  59. data/lib/gsc/firewall_scanner.rb +733 -0
  60. data/lib/gsc/geo_auditor.rb +368 -0
  61. data/lib/gsc/google_trends.rb +8 -1
  62. data/lib/gsc/heading_validator.rb +283 -0
  63. data/lib/gsc/hreflang_validator.rb +412 -0
  64. data/lib/gsc/image_seo.rb +286 -0
  65. data/lib/gsc/indexing_queue.rb +179 -0
  66. data/lib/gsc/indexnow.rb +93 -0
  67. data/lib/gsc/intent_shift.rb +188 -0
  68. data/lib/gsc/internal_links.rb +153 -36
  69. data/lib/gsc/keyword_value.rb +191 -0
  70. data/lib/gsc/landing_roi.rb +195 -0
  71. data/lib/gsc/llms_generator.rb +343 -22
  72. data/lib/gsc/low_ctr_rewriter.rb +408 -0
  73. data/lib/gsc/mobile_parity.rb +222 -0
  74. data/lib/gsc/network_tracer.rb +8 -1
  75. data/lib/gsc/page_analyzer.rb +47 -7
  76. data/lib/gsc/prompts.rb +38 -29
  77. data/lib/gsc/questions_harvester.rb +178 -0
  78. data/lib/gsc/report_generator.rb +461 -0
  79. data/lib/gsc/rich_results.rb +388 -0
  80. data/lib/gsc/robots_checker.rb +46 -15
  81. data/lib/gsc/schema_generator.rb +788 -0
  82. data/lib/gsc/schema_validator.rb +36 -38
  83. data/lib/gsc/seasonal_predictor.rb +381 -0
  84. data/lib/gsc/security_scanner.rb +496 -0
  85. data/lib/gsc/serp_feature_detector.rb +359 -0
  86. data/lib/gsc/serp_preview.rb +108 -22
  87. data/lib/gsc/site_crawler.rb +113 -21
  88. data/lib/gsc/sitemap_loader.rb +15 -4
  89. data/lib/gsc/sitemap_tree.rb +301 -0
  90. data/lib/gsc/skill_pack.rb +195 -0
  91. data/lib/gsc/soft_404_analyzer.rb +385 -0
  92. data/lib/gsc/sparkline.rb +171 -0
  93. data/lib/gsc/speed_correlator.rb +416 -0
  94. data/lib/gsc/striking_playbook.rb +190 -0
  95. data/lib/gsc/title_optimizer.rb +420 -0
  96. data/lib/gsc/vault.rb +260 -0
  97. data/lib/gsc/version.rb +1 -1
  98. data/lib/gsc/watchdog.rb +235 -0
  99. data/lib/gsc/zombie_purger.rb +366 -0
  100. data/lib/gsc.rb +118 -0
  101. metadata +75 -1
@@ -0,0 +1,235 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'json'
5
+ require 'net/http'
6
+ require 'uri'
7
+ require 'date'
8
+
9
+ module GSC
10
+ class Watchdog
11
+ DEFAULT_DROP_THRESHOLD = 2.0 # 2+ rank position drop
12
+ DEFAULT_CTR_THRESHOLD = 20.0 # 20%+ CTR decrease
13
+
14
+ attr_reader :options, :api, :domain
15
+
16
+ def initialize(options = {}, api = nil, domain = nil)
17
+ @options = options
18
+ @api = api
19
+ @domain = domain.to_s.strip
20
+ end
21
+
22
+ def self.check(options = {}, api = nil, domain = nil)
23
+ new(options, api, domain).check
24
+ end
25
+
26
+ def check
27
+ telemetry = fetch_telemetry
28
+ alerts = detect_anomalies(telemetry)
29
+ summary = synthesize_summary(telemetry, alerts)
30
+
31
+ if @options[:webhook] || @options[:slack]
32
+ webhook_res = dispatch_webhook(summary, @options[:webhook] || @options[:slack])
33
+ summary[:webhook_dispatched] = webhook_res
34
+ end
35
+
36
+ summary
37
+ end
38
+
39
+ def generate_crontab_entry
40
+ bin = current_binary_path
41
+ hours = (@options[:hours] || 6).to_i
42
+ webhook_flag = @options[:webhook] ? " --webhook #{@options[:webhook]}" : ""
43
+ "0 */#{hours} * * * #{bin} watch #{@domain} --once#{webhook_flag} >> ~/.config/gsc/watch.log 2>&1"
44
+ end
45
+
46
+ def generate_launchd_plist
47
+ bin = current_binary_path
48
+ interval = (@options[:interval] || 21600).to_i
49
+ label = "com.gsc.watchdog.#{@domain.gsub(/[^a-zA-Z0-9]/, '_')}"
50
+
51
+ <<~XML
52
+ <?xml version="1.0" encoding="UTF-8"?>
53
+ <!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
54
+ <plist version="1.0">
55
+ <dict>
56
+ <key>Label</key>
57
+ <string>#{label}</string>
58
+ <key>ProgramArguments</key>
59
+ <array>
60
+ <string>#{bin}</string>
61
+ <string>watch</string>
62
+ <string>#{@domain}</string>
63
+ <string>--once</string>
64
+ </array>
65
+ <key>StartInterval</key>
66
+ <integer>#{interval}</integer>
67
+ <key>StandardOutPath</key>
68
+ <string>/tmp/gsc_watch_#{@domain}.log</string>
69
+ <key>StandardErrorPath</key>
70
+ <string>/tmp/gsc_watch_#{@domain}_err.log</string>
71
+ </dict>
72
+ </plist>
73
+ XML
74
+ end
75
+
76
+ def generate_systemd_unit
77
+ bin = current_binary_path
78
+ <<~INI
79
+ [Unit]
80
+ Description=GSC Continuous SEO Rank & CTR Volatility Watchdog for #{@domain}
81
+ After=network.target
82
+
83
+ [Service]
84
+ Type=oneshot
85
+ ExecStart=#{bin} watch #{@domain} --once
86
+ StandardOutput=journal
87
+ StandardError=journal
88
+
89
+ [Install]
90
+ WantedBy=multi-user.target
91
+ INI
92
+ end
93
+
94
+ private
95
+
96
+ def current_binary_path
97
+ ENV['GSC_BIN_PATH'] || File.expand_path('~/.local/bin/gsc')
98
+ end
99
+
100
+ def fetch_telemetry
101
+ if @api
102
+ begin
103
+ curr_end = (Date.today - 2).strftime('%Y-%m-%d')
104
+ curr_start = (Date.today - 9).strftime('%Y-%m-%d')
105
+ prev_end = (Date.today - 10).strftime('%Y-%m-%d')
106
+ prev_start = (Date.today - 17).strftime('%Y-%m-%d')
107
+
108
+ curr_res = @api.search_analytics(@domain, start_date: curr_start, end_date: curr_end, dimensions: %w[query])
109
+ prev_res = @api.search_analytics(@domain, start_date: prev_start, end_date: prev_end, dimensions: %w[query])
110
+
111
+ curr_rows = (curr_res['rows'] || []).each_with_object({}) { |r, h| h[r['keys'][0]] = r }
112
+ prev_rows = (prev_res['rows'] || []).each_with_object({}) { |r, h| h[r['keys'][0]] = r }
113
+
114
+ all_queries = (curr_rows.keys + prev_rows.keys).uniq
115
+
116
+ if all_queries.any?
117
+ return all_queries.first(25).map do |q|
118
+ cr = curr_rows[q]
119
+ pr = prev_rows[q]
120
+
121
+ curr_pos = cr ? (cr['position'] || 0).round(1) : 100.0
122
+ prev_pos = pr ? (pr['position'] || 0).round(1) : 100.0
123
+ curr_clicks = cr ? (cr['clicks'] || 0) : 0
124
+ prev_clicks = pr ? (pr['clicks'] || 0) : 0
125
+ curr_ctr = cr ? ((cr['ctr'] || 0) * 100.0).round(2) : 0.0
126
+ prev_ctr = pr ? ((pr['ctr'] || 0) * 100.0).round(2) : 0.0
127
+
128
+ {
129
+ query: q,
130
+ current_position: curr_pos,
131
+ previous_position: prev_pos,
132
+ current_clicks: curr_clicks,
133
+ previous_clicks: prev_clicks,
134
+ current_ctr: curr_ctr,
135
+ previous_ctr: prev_ctr
136
+ }
137
+ end
138
+ end
139
+ rescue StandardError
140
+ # return empty
141
+ end
142
+ end
143
+
144
+ []
145
+ end
146
+
147
+ def detect_anomalies(telemetry)
148
+ alerts = []
149
+ pos_threshold = (@options[:threshold] || DEFAULT_DROP_THRESHOLD).to_f
150
+ ctr_threshold = DEFAULT_CTR_THRESHOLD
151
+
152
+ telemetry.each do |row|
153
+ pos_delta = (row[:current_position] - row[:previous_position]).round(1) # positive means dropped rank
154
+ ctr_drop_pct = row[:previous_ctr] > 0 ? (((row[:previous_ctr] - row[:current_ctr]) / row[:previous_ctr]) * 100.0).round(1) : 0.0
155
+
156
+ if pos_delta >= pos_threshold
157
+ severity = pos_delta >= 4.0 ? :critical : :warning
158
+ alerts << {
159
+ query: row[:query],
160
+ type: :position_drop,
161
+ severity: severity,
162
+ message: "Rank dropped #{pos_delta} positions (Pos #{row[:previous_position]} ➔ Pos #{row[:current_position]})",
163
+ current_position: row[:current_position],
164
+ previous_position: row[:previous_position],
165
+ current_clicks: row[:current_clicks],
166
+ previous_clicks: row[:previous_clicks]
167
+ }
168
+ elsif ctr_drop_pct >= ctr_threshold && row[:current_position] <= 10.0
169
+ alerts << {
170
+ query: row[:query],
171
+ type: :ctr_anomaly,
172
+ severity: :warning,
173
+ message: "CTR dropped #{ctr_drop_pct}% (from #{row[:previous_ctr]}% to #{row[:current_ctr]}%) despite stable rank (Pos #{row[:current_position]}). Likely Zero-Click AI Overview suppression.",
174
+ current_position: row[:current_position],
175
+ previous_position: row[:previous_position],
176
+ current_clicks: row[:current_clicks],
177
+ previous_clicks: row[:previous_clicks]
178
+ }
179
+ end
180
+ end
181
+
182
+ alerts
183
+ end
184
+
185
+ def synthesize_summary(telemetry, alerts)
186
+ critical_count = alerts.count { |a| a[:severity] == :critical }
187
+ warning_count = alerts.count { |a| a[:severity] == :warning }
188
+
189
+ status = if critical_count > 0
190
+ 'CRITICAL ANOMALIES'
191
+ elsif warning_count > 0
192
+ 'WARNINGS DETECTED'
193
+ else
194
+ 'SERP HEALTH OPTIMAL'
195
+ end
196
+
197
+ {
198
+ domain: @domain,
199
+ monitored_queries_count: telemetry.size,
200
+ status: status,
201
+ critical_alerts_count: critical_count,
202
+ warning_alerts_count: warning_count,
203
+ total_alerts_count: alerts.size,
204
+ checked_at: Time.now.strftime('%Y-%m-%d %H:%M:%S %Z'),
205
+ alerts: alerts,
206
+ telemetry: telemetry
207
+ }
208
+ end
209
+
210
+ def dispatch_webhook(summary, webhook_url)
211
+ uri = URI.parse(webhook_url)
212
+ payload = {
213
+ text: "🚨 [GSC Rank Watchdog] #{summary[:total_alerts_count]} Organic SERP Alerts Detected for #{summary[:domain]} (#{summary[:status]})",
214
+ domain: summary[:domain],
215
+ status: summary[:status],
216
+ critical_count: summary[:critical_alerts_count],
217
+ warnings_count: summary[:warning_alerts_count],
218
+ checked_at: summary[:checked_at],
219
+ alerts: summary[:alerts].map { |a| { query: a[:query], message: a[:message], severity: a[:severity] } }
220
+ }
221
+
222
+ req = Net::HTTP::Post.new(uri)
223
+ req['Content-Type'] = 'application/json'
224
+ req.body = JSON.generate(payload)
225
+
226
+ res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 5, read_timeout: 5) do |http|
227
+ http.request(req)
228
+ end
229
+
230
+ res.code.to_i >= 200 && res.code.to_i < 300
231
+ rescue StandardError => e
232
+ false
233
+ end
234
+ end
235
+ end
@@ -0,0 +1,366 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'json'
5
+ require 'uri'
6
+
7
+ module GSC
8
+ class ZombiePurger
9
+ DEFAULT_DAYS = 90
10
+ DEFAULT_THRESHOLD = 0
11
+ EST_MONTHLY_CRAWLS_PER_ZOMBIE = 6
12
+
13
+ # Common utility/legal URL path fragments that should remain accessible to users but noindexed
14
+ UTILITY_PATHS = %w[
15
+ /privacy /terms /disclaimer /legal /contact /about-us /about
16
+ /account /login /signin /signup /register /cart /checkout
17
+ /order /thank-you /confirmation /search /sitemap
18
+ ].freeze
19
+
20
+ # Common taxonomy, archive, or thin content fragments suited for 410 purge
21
+ THIN_PATTERNS = [
22
+ %r{/page/\d+},
23
+ %r{/tag/},
24
+ %r{/tags/},
25
+ %r{/category/},
26
+ %r{/archive/},
27
+ %r{/author/},
28
+ %r{/attachment/},
29
+ %r{\?replytocom=},
30
+ %r{\?amp=1},
31
+ %r{/feed/?$}
32
+ ].freeze
33
+
34
+ STOPWORDS = %w[the a an and of to in for with on at by from is are was were this that].freeze
35
+
36
+ attr_reader :options, :days, :threshold
37
+
38
+ def initialize(options = {})
39
+ @options = options
40
+ @days = (options[:days] || DEFAULT_DAYS).to_i
41
+ @threshold = (options[:threshold] || DEFAULT_THRESHOLD).to_i
42
+ end
43
+
44
+ # Convenience class method for auditing inventory against GSC pages
45
+ def self.audit(inventory_urls, gsc_pages, options = {})
46
+ new(options).audit(inventory_urls, gsc_pages)
47
+ end
48
+
49
+ # inventory_urls: Array of URL strings
50
+ # gsc_pages: Hash of { clean_url => { impressions:, clicks:, position: } } OR Array of row hashes
51
+ def audit(inventory_urls, gsc_pages)
52
+ urls = normalize_url_list(inventory_urls)
53
+ active_map = normalize_gsc_pages(gsc_pages)
54
+
55
+ active_items = []
56
+ zombie_items = []
57
+
58
+ # List of all active clean URLs for candidate pillar matching
59
+ active_candidates = active_map.keys
60
+
61
+ urls.each do |raw_url|
62
+ clean_key = clean_url_key(raw_url)
63
+ page_stat = active_map[clean_key]
64
+ impressions = page_stat ? (page_stat[:impressions] || 0) : 0
65
+ clicks = page_stat ? (page_stat[:clicks] || 0) : 0
66
+ position = page_stat ? (page_stat[:position] || 0.0) : 0.0
67
+
68
+ if impressions > @threshold
69
+ active_items << {
70
+ url: raw_url,
71
+ clean_url: clean_key,
72
+ impressions: impressions,
73
+ clicks: clicks,
74
+ position: position
75
+ }
76
+ else
77
+ # Triage this zombie URL
78
+ triage = classify_zombie(raw_url, active_candidates)
79
+ zombie_items << {
80
+ url: raw_url,
81
+ clean_url: clean_key,
82
+ impressions: impressions,
83
+ clicks: clicks,
84
+ action: triage[:action],
85
+ action_label: format_action_label(triage[:action]),
86
+ target_url: triage[:target_url],
87
+ rationale: triage[:rationale],
88
+ path: uri_path(raw_url)
89
+ }
90
+ end
91
+ end
92
+
93
+ total_inventory = urls.size
94
+ zombie_count = zombie_items.size
95
+ active_count = active_items.size
96
+
97
+ zombie_ratio = total_inventory > 0 ? ((zombie_count.to_f / total_inventory) * 100.0).round(1) : 0.0
98
+ cedi = zombie_ratio # Crawl Equity Dilution Index (0-100, lower is better)
99
+ health_score = [0.0, (100.0 - cedi)].max.round(1)
100
+ grade = compute_grade(cedi)
101
+
102
+ wasted_crawls_monthly = zombie_count * EST_MONTHLY_CRAWLS_PER_ZOMBIE
103
+ wasted_crawls_annually = wasted_crawls_monthly * 12
104
+
105
+ action_breakdown = {
106
+ purge_410: zombie_items.count { |z| z[:action] == :purge_410 },
107
+ redirect_301: zombie_items.count { |z| z[:action] == :redirect_301 },
108
+ consolidate: zombie_items.count { |z| z[:action] == :consolidate },
109
+ noindex: zombie_items.count { |z| z[:action] == :noindex }
110
+ }
111
+
112
+ # Filter if specific action requested
113
+ filtered_zombies = if @options[:action]
114
+ target_act = @options[:action].to_s.downcase.sub('301', '').sub('410', '').to_sym
115
+ zombie_items.select { |z| z[:action].to_s.include?(target_act.to_s) }
116
+ else
117
+ zombie_items
118
+ end
119
+
120
+ limit = (@options[:limit] || 25).to_i
121
+ displayed_zombies = filtered_zombies.first(limit)
122
+
123
+ {
124
+ total_inventory: total_inventory,
125
+ active_count: active_count,
126
+ zombie_count: zombie_count,
127
+ zombie_percentage: zombie_ratio,
128
+ cedi: cedi,
129
+ health_score: health_score,
130
+ grade: grade,
131
+ days: @days,
132
+ threshold: @threshold,
133
+ wasted_crawls_monthly: wasted_crawls_monthly,
134
+ wasted_crawls_annually: wasted_crawls_annually,
135
+ action_breakdown: action_breakdown,
136
+ zombies: displayed_zombies,
137
+ all_zombies_count: filtered_zombies.size,
138
+ server_rules: {
139
+ nginx: generate_nginx(displayed_zombies),
140
+ htaccess: generate_htaccess(displayed_zombies),
141
+ redirects: generate_redirects(displayed_zombies),
142
+ meta_robots: generate_meta_robots
143
+ }
144
+ }
145
+ end
146
+
147
+ # Classifies a single zombie URL into a strategic action
148
+ def classify_zombie(url, active_candidates)
149
+ path = uri_path(url).downcase
150
+
151
+ # 1. Check if it's a utility or legal page that should be noindexed
152
+ if UTILITY_PATHS.any? { |up| path.start_with?(up) || path == up }
153
+ return {
154
+ action: :noindex,
155
+ target_url: nil,
156
+ rationale: "Utility/legal page necessary for user experience; noindex to preserve crawl equity."
157
+ }
158
+ end
159
+
160
+ # 2. Check for thin taxonomy, pagination, or feed loops
161
+ if THIN_PATTERNS.any? { |pat| path =~ pat }
162
+ return {
163
+ action: :purge_410,
164
+ target_url: nil,
165
+ rationale: "Thin taxonomy/archive/pagination loop; return 410 Gone to immediately drop from Googlebot index."
166
+ }
167
+ end
168
+
169
+ # 3. Check for candidate active pillar target to redirect equity
170
+ candidate = find_candidate_target(url, active_candidates)
171
+ if candidate
172
+ return {
173
+ action: :redirect_301,
174
+ target_url: candidate,
175
+ rationale: "Topical slug overlap with active pillar '#{uri_path(candidate)}'; 301 redirect to consolidate equity."
176
+ }
177
+ end
178
+
179
+ # 4. Check for dated content or parameter duplicates
180
+ if path.match?(%r{/\d{4}/\d{2}/}) || path.include?('?')
181
+ return {
182
+ action: :consolidate,
183
+ target_url: nil,
184
+ rationale: "Dated archive or parameter URL; consolidate or rewrite into evergreen canonical."
185
+ }
186
+ end
187
+
188
+ # 5. Default dead zombie: Purge 410
189
+ {
190
+ action: :purge_410,
191
+ target_url: nil,
192
+ rationale: "Zero impressions over #{@days} days with no active pillar match; return 410 Gone to reclaim crawl budget."
193
+ }
194
+ end
195
+
196
+ # Finds an active page with substantial slug keyword overlap
197
+ def find_candidate_target(zombie_url, active_candidates)
198
+ zombie_tokens = extract_slug_tokens(zombie_url)
199
+ return nil if zombie_tokens.empty?
200
+
201
+ best_candidate = nil
202
+ best_score = 0.0
203
+
204
+ active_candidates.each do |act_url|
205
+ act_tokens = extract_slug_tokens(act_url)
206
+ next if act_tokens.empty?
207
+
208
+ common = zombie_tokens & act_tokens
209
+ next if common.empty?
210
+
211
+ # Score based on token overlap
212
+ overlap_ratio = common.size.to_f / [zombie_tokens.size, act_tokens.size].max
213
+ if overlap_ratio > best_score && common.size >= 2
214
+ best_score = overlap_ratio
215
+ best_candidate = act_url
216
+ end
217
+ end
218
+
219
+ best_score >= 0.4 ? best_candidate : nil
220
+ end
221
+
222
+ # Server directive generators
223
+ def generate_nginx(zombies)
224
+ lines = [
225
+ "# ====================================================================",
226
+ "# Nginx Zombie Content Purge Directives (gsc-cli)",
227
+ "# Reclaims Googlebot crawl budget and stops 404 crawl waste",
228
+ "# ===================================================================="
229
+ ]
230
+
231
+ zombies.each do |z|
232
+ path = uri_path(z[:url])
233
+ case z[:action]
234
+ when :purge_410
235
+ lines << "location = #{path} { return 410; }"
236
+ when :redirect_301
237
+ target_path = uri_path(z[:target_url] || '/')
238
+ lines << "location = #{path} { return 301 #{target_path}; }"
239
+ end
240
+ end
241
+ lines.join("\n")
242
+ end
243
+
244
+ def generate_htaccess(zombies)
245
+ lines = [
246
+ "# ====================================================================",
247
+ "# Apache .htaccess Zombie Content Purge Directives (gsc-cli)",
248
+ "# ===================================================================="
249
+ ]
250
+
251
+ zombies.each do |z|
252
+ path = uri_path(z[:url])
253
+ case z[:action]
254
+ when :purge_410
255
+ lines << "RedirectGone #{path}"
256
+ when :redirect_301
257
+ target_path = uri_path(z[:target_url] || '/')
258
+ lines << "Redirect 301 #{path} #{target_path}"
259
+ end
260
+ end
261
+ lines.join("\n")
262
+ end
263
+
264
+ def generate_redirects(zombies)
265
+ lines = [
266
+ "# ====================================================================",
267
+ "# Netlify / Cloudflare Pages _redirects Rules (gsc-cli)",
268
+ "# ===================================================================="
269
+ ]
270
+
271
+ zombies.each do |z|
272
+ path = uri_path(z[:url])
273
+ case z[:action]
274
+ when :purge_410
275
+ lines << "#{path} 410!"
276
+ when :redirect_301
277
+ target_path = uri_path(z[:target_url] || '/')
278
+ lines << "#{path} #{target_path} 301"
279
+ end
280
+ end
281
+ lines.join("\n")
282
+ end
283
+
284
+ def generate_meta_robots
285
+ '<meta name="robots" content="noindex, follow">'
286
+ end
287
+
288
+ private
289
+
290
+ def compute_grade(cedi)
291
+ case cedi
292
+ when 0..10.0 then 'A'
293
+ when 10.1..25.0 then 'B'
294
+ when 25.1..45.0 then 'C'
295
+ when 45.1..70.0 then 'D'
296
+ else 'F'
297
+ end
298
+ end
299
+
300
+ def format_action_label(action)
301
+ case action
302
+ when :purge_410 then 'PURGE (410 GONE)'
303
+ when :redirect_301 then 'REDIRECT (301)'
304
+ when :consolidate then 'CONSOLIDATE'
305
+ when :noindex then 'NOINDEX'
306
+ else action.to_s.upcase
307
+ end
308
+ end
309
+
310
+ def extract_slug_tokens(url)
311
+ path = uri_path(url)
312
+ clean = path.sub(/\.(html?|php|aspx?)$/i, '')
313
+ tokens = clean.split(/[\/\-_]/).map(&:downcase).reject do |t|
314
+ t.empty? || t.match?(/^\d+$/) || STOPWORDS.include?(t)
315
+ end
316
+ tokens.uniq
317
+ end
318
+
319
+ def uri_path(url_str)
320
+ return '/' if url_str.nil? || url_str.empty?
321
+ u = URI.parse(url_str)
322
+ u.path.nil? || u.path.empty? ? '/' : u.path
323
+ rescue URI::InvalidURIError
324
+ url_str.sub(%r{^https?://[^/]+}, '')
325
+ end
326
+
327
+ def clean_url_key(url_str)
328
+ url_str.to_s.strip.downcase.sub(%r{/$}, '')
329
+ end
330
+
331
+ def normalize_url_list(urls)
332
+ Array(urls).map(&:to_s).map(&:strip).reject(&:empty?).uniq
333
+ end
334
+
335
+ def normalize_gsc_pages(gsc_pages)
336
+ map = {}
337
+ if gsc_pages.is_a?(Hash)
338
+ gsc_pages.each do |k, v|
339
+ clean_k = clean_url_key(k)
340
+ if v.is_a?(Hash)
341
+ map[clean_k] = {
342
+ clicks: (v[:clicks] || v['clicks'] || 0).to_i,
343
+ impressions: (v[:impressions] || v['impressions'] || 0).to_i,
344
+ position: (v[:position] || v['position'] || 0.0).to_f
345
+ }
346
+ else
347
+ map[clean_k] = { clicks: 0, impressions: v.to_i, position: 0.0 }
348
+ end
349
+ end
350
+ elsif gsc_pages.is_a?(Array)
351
+ gsc_pages.each do |row|
352
+ next unless row.is_a?(Hash)
353
+ keys = row['keys'] || row[:keys]
354
+ next unless keys && keys.first
355
+ clean_k = clean_url_key(keys.first)
356
+ map[clean_k] = {
357
+ clicks: (row['clicks'] || row[:clicks] || 0).to_i,
358
+ impressions: (row['impressions'] || row[:impressions] || 0).to_i,
359
+ position: (row['position'] || row[:position] || 0.0).to_f
360
+ }
361
+ end
362
+ end
363
+ map
364
+ end
365
+ end
366
+ end