gsc-cli 2.1.0 → 2.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. checksums.yaml +4 -4
  2. data/AUTH.md +4 -1
  3. data/README.md +448 -408
  4. data/bin/gsc +28067 -5661
  5. data/dist/gsc +29121 -5046
  6. data/lib/gsc/aio_hunter.rb +343 -0
  7. data/lib/gsc/answer_synthesizer.rb +157 -0
  8. data/lib/gsc/api.rb +53 -1
  9. data/lib/gsc/auth.rb +26 -0
  10. data/lib/gsc/brand_segmenter.rb +140 -0
  11. data/lib/gsc/cache_manager.rb +806 -0
  12. data/lib/gsc/cannibalization_analyzer.rb +141 -0
  13. data/lib/gsc/canonical_chains.rb +367 -0
  14. data/lib/gsc/citation_simulator.rb +339 -0
  15. data/lib/gsc/cli/aio_hunter.rb +154 -0
  16. data/lib/gsc/cli/analytics.rb +788 -0
  17. data/lib/gsc/cli/audit.rb +1976 -0
  18. data/lib/gsc/cli/base.rb +384 -0
  19. data/lib/gsc/cli/cache.rb +266 -0
  20. data/lib/gsc/cli/canonical.rb +223 -0
  21. data/lib/gsc/cli/citation_simulator.rb +152 -0
  22. data/lib/gsc/cli/dashboard.rb +354 -0
  23. data/lib/gsc/cli/doctor.rb +129 -0
  24. data/lib/gsc/cli/eeat.rb +125 -0
  25. data/lib/gsc/cli/ga4.rb +852 -0
  26. data/lib/gsc/cli/growth.rb +650 -0
  27. data/lib/gsc/cli/hreflang.rb +164 -0
  28. data/lib/gsc/cli/image_seo.rb +162 -0
  29. data/lib/gsc/cli/indexing.rb +458 -0
  30. data/lib/gsc/cli/intent_shift.rb +125 -0
  31. data/lib/gsc/cli/keyword_value.rb +134 -0
  32. data/lib/gsc/cli/keywords.rb +795 -0
  33. data/lib/gsc/cli/landing_roi.rb +308 -0
  34. data/lib/gsc/cli/low_ctr.rb +213 -0
  35. data/lib/gsc/cli/mobile_parity.rb +150 -0
  36. data/lib/gsc/cli/report.rb +100 -0
  37. data/lib/gsc/cli/rich_results.rb +172 -0
  38. data/lib/gsc/cli/schema_generate.rb +149 -0
  39. data/lib/gsc/cli/seasonal.rb +232 -0
  40. data/lib/gsc/cli/security.rb +153 -0
  41. data/lib/gsc/cli/setup.rb +1291 -0
  42. data/lib/gsc/cli/sitemap_tree.rb +143 -0
  43. data/lib/gsc/cli/skill_pack.rb +62 -0
  44. data/lib/gsc/cli/soft_404.rb +199 -0
  45. data/lib/gsc/cli/sparkline.rb +227 -0
  46. data/lib/gsc/cli/watchdog.rb +150 -0
  47. data/lib/gsc/cli/zombie_purger.rb +208 -0
  48. data/lib/gsc/cli.rb +707 -5265
  49. data/lib/gsc/cli_advanced.rb +987 -44
  50. data/lib/gsc/client.rb +17 -2
  51. data/lib/gsc/color.rb +16 -1
  52. data/lib/gsc/command_registry.rb +47 -9
  53. data/lib/gsc/config.rb +2 -2
  54. data/lib/gsc/ctr_curve.rb +115 -0
  55. data/lib/gsc/decay_predictor.rb +322 -0
  56. data/lib/gsc/doctor.rb +434 -0
  57. data/lib/gsc/eeat_auditor.rb +428 -0
  58. data/lib/gsc/entity_auditor.rb +229 -0
  59. data/lib/gsc/firewall_scanner.rb +733 -0
  60. data/lib/gsc/geo_auditor.rb +368 -0
  61. data/lib/gsc/google_trends.rb +8 -1
  62. data/lib/gsc/heading_validator.rb +283 -0
  63. data/lib/gsc/hreflang_validator.rb +412 -0
  64. data/lib/gsc/image_seo.rb +286 -0
  65. data/lib/gsc/indexing_queue.rb +179 -0
  66. data/lib/gsc/indexnow.rb +93 -0
  67. data/lib/gsc/intent_shift.rb +188 -0
  68. data/lib/gsc/internal_links.rb +153 -36
  69. data/lib/gsc/keyword_value.rb +191 -0
  70. data/lib/gsc/landing_roi.rb +195 -0
  71. data/lib/gsc/llms_generator.rb +343 -22
  72. data/lib/gsc/low_ctr_rewriter.rb +408 -0
  73. data/lib/gsc/mobile_parity.rb +222 -0
  74. data/lib/gsc/network_tracer.rb +8 -1
  75. data/lib/gsc/page_analyzer.rb +47 -7
  76. data/lib/gsc/prompts.rb +38 -29
  77. data/lib/gsc/questions_harvester.rb +178 -0
  78. data/lib/gsc/report_generator.rb +461 -0
  79. data/lib/gsc/rich_results.rb +388 -0
  80. data/lib/gsc/robots_checker.rb +46 -15
  81. data/lib/gsc/schema_generator.rb +788 -0
  82. data/lib/gsc/schema_validator.rb +36 -38
  83. data/lib/gsc/seasonal_predictor.rb +381 -0
  84. data/lib/gsc/security_scanner.rb +496 -0
  85. data/lib/gsc/serp_feature_detector.rb +359 -0
  86. data/lib/gsc/serp_preview.rb +108 -22
  87. data/lib/gsc/site_crawler.rb +113 -21
  88. data/lib/gsc/sitemap_loader.rb +15 -4
  89. data/lib/gsc/sitemap_tree.rb +301 -0
  90. data/lib/gsc/skill_pack.rb +195 -0
  91. data/lib/gsc/soft_404_analyzer.rb +385 -0
  92. data/lib/gsc/sparkline.rb +171 -0
  93. data/lib/gsc/speed_correlator.rb +416 -0
  94. data/lib/gsc/striking_playbook.rb +190 -0
  95. data/lib/gsc/title_optimizer.rb +420 -0
  96. data/lib/gsc/vault.rb +260 -0
  97. data/lib/gsc/version.rb +1 -1
  98. data/lib/gsc/watchdog.rb +235 -0
  99. data/lib/gsc/zombie_purger.rb +366 -0
  100. data/lib/gsc.rb +118 -0
  101. metadata +75 -1
@@ -0,0 +1,141 @@
1
+ # frozen_string_literal: true
2
+
3
+ module GSC
4
+ class CannibalizationAnalyzer
5
+ def self.analyze(rows, min_imp: 10, min_secondary_ratio: 0.15)
6
+ grouped = Hash.new { |h, k| h[k] = [] }
7
+
8
+ rows.each do |r|
9
+ query = r[:query] || r['query'] || r.dig('keys', 0)
10
+ page = r[:page] || r['page'] || r.dig('keys', 1)
11
+ clicks = r[:clicks] || r['clicks'] || 0
12
+ imp = r[:impressions] || r['impressions'] || 0
13
+ pos = (r[:position] || r['position'] || 100.0).to_f.round(1)
14
+
15
+ next unless query && page
16
+
17
+ grouped[query] << {
18
+ page: page,
19
+ clicks: clicks,
20
+ impressions: imp,
21
+ position: pos
22
+ }
23
+ end
24
+
25
+ conflicts = []
26
+ total_wasted_imp = 0
27
+
28
+ grouped.each do |query, pages|
29
+ next if pages.size < 2
30
+
31
+ total_imp = pages.sum { |p| p[:impressions] }
32
+ next if total_imp < min_imp
33
+
34
+ sorted_pages = pages.sort_by { |p| -p[:impressions] }
35
+ primary = sorted_pages[0]
36
+ secondary = sorted_pages[1]
37
+
38
+ sec_ratio = secondary[:impressions].to_f / total_imp
39
+ next if sec_ratio < min_secondary_ratio
40
+
41
+ severity = calculate_severity(primary, secondary, total_imp, sec_ratio)
42
+ remedy = determine_remedy(primary, secondary, query)
43
+
44
+ # Wasted / split impressions from secondary+ competing pages
45
+ diluted_imp = pages[1..].sum { |p| p[:impressions] }
46
+ total_wasted_imp += diluted_imp
47
+
48
+ conflicts << {
49
+ query: query,
50
+ severity: severity,
51
+ total_impressions: total_imp,
52
+ total_clicks: pages.sum { |p| p[:clicks] },
53
+ competing_pages_count: pages.size,
54
+ diluted_impressions: diluted_imp,
55
+ remedy: remedy,
56
+ pages: sorted_pages.map do |p|
57
+ share = ((p[:impressions].to_f / total_imp) * 100).round(1)
58
+ p.merge(
59
+ impression_share: share,
60
+ impression_share_str: "#{share}%"
61
+ )
62
+ end
63
+ }
64
+ end
65
+
66
+ # Sort by severity (CRITICAL > HIGH > MODERATE > LOW) then total impressions
67
+ severity_order = { 'CRITICAL' => 4, 'HIGH' => 3, 'MODERATE' => 2, 'LOW' => 1 }
68
+ conflicts.sort_by! { |c| [-severity_order[c[:severity]], -c[:total_impressions]] }
69
+
70
+ critical_count = conflicts.count { |c| c[:severity] == 'CRITICAL' }
71
+ health_score = [100 - (conflicts.size * 10 + critical_count * 15), 0].max
72
+
73
+ grade = case health_score
74
+ when 90..100 then 'A'
75
+ when 80..89 then 'B'
76
+ when 70..79 then 'C'
77
+ when 60..69 then 'D'
78
+ else 'F'
79
+ end
80
+
81
+ {
82
+ total_queries_evaluated: grouped.size,
83
+ conflicts_count: conflicts.size,
84
+ critical_count: critical_count,
85
+ total_diluted_impressions: total_wasted_imp,
86
+ health_score: health_score,
87
+ health_grade: grade,
88
+ conflicts: conflicts
89
+ }
90
+ end
91
+
92
+ def self.calculate_severity(primary, secondary, total_imp, sec_ratio)
93
+ # If both rank on Page 1 or 2 and share is closely split
94
+ if (primary[:position] <= 20.0 || secondary[:position] <= 20.0) && (sec_ratio >= 0.35 || total_imp >= 50)
95
+ 'CRITICAL'
96
+ elsif sec_ratio >= 0.25 || total_imp >= 25
97
+ 'HIGH'
98
+ elsif sec_ratio >= 0.15
99
+ 'MODERATE'
100
+ else
101
+ 'LOW'
102
+ end
103
+ end
104
+
105
+ def self.determine_remedy(primary, secondary, _query)
106
+ p_path = primary[:page].sub(%r{^https?://[^/]+}, '')
107
+ p_path = '/' if p_path.empty?
108
+ s_path = secondary[:page].sub(%r{^https?://[^/]+}, '')
109
+ s_path = '/' if s_path.empty?
110
+
111
+ # Case 1: Secondary page actually ranks higher than primary!
112
+ if secondary[:position] < (primary[:position] - 1.5)
113
+ {
114
+ action: 'FLIP_FLOP_CONSOLIDATION',
115
+ recommendation: "Googlebot prefers secondary page (#{s_path} ranks Pos #{secondary[:position]} vs #{primary[:position]}). Set canonical from #{p_path} pointing to #{s_path}, or 301 redirect #{p_path} to #{s_path}."
116
+ }
117
+ # Case 2: Inverted or duplicate slugs in same directory
118
+ elsif slug_similarity(p_path, s_path) > 0.7
119
+ {
120
+ action: '301_REDIRECT',
121
+ recommendation: "Near-identical slug intent. 301 redirect secondary #{s_path} into primary #{p_path} to merge PageRank."
122
+ }
123
+ # Case 3: Distinct topics accidentally colliding
124
+ else
125
+ {
126
+ action: 'CANONICAL_OR_LINK_DE_OPTIMIZE',
127
+ recommendation: "Differentiate content angles: set canonical pointing to #{p_path}, and remove \"#{_query}\" keyword from secondary #{s_path}'s H1/title."
128
+ }
129
+ end
130
+ end
131
+
132
+ def self.slug_similarity(path1, path2)
133
+ tokens1 = path1.downcase.split(/[^a-z0-9]+/).reject { |t| t.length < 2 }
134
+ tokens2 = path2.downcase.split(/[^a-z0-9]+/).reject { |t| t.length < 2 }
135
+ return 0.0 if tokens1.empty? || tokens2.empty?
136
+
137
+ common = (tokens1 & tokens2).size
138
+ common.to_f / [tokens1.size, tokens2.size].max
139
+ end
140
+ end
141
+ end
@@ -0,0 +1,367 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'net/http'
5
+ require 'uri'
6
+ require 'set'
7
+ require 'time'
8
+
9
+ module GSC
10
+ class CanonicalChains
11
+ EQUITY_DECAY_PER_HOP = 0.15 # 15% estimated equity loss per redirect hop
12
+
13
+ attr_reader :max_hops, :timeout, :user_agent
14
+
15
+ def initialize(max_hops: 10, timeout: 8, user_agent: nil)
16
+ @max_hops = max_hops
17
+ @timeout = timeout
18
+ @user_agent = user_agent || 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) gsc-cli/2.1'
19
+ end
20
+
21
+ # Analyze a single URL through redirect hops and canonical declarations
22
+ def audit_url(url, gsc_rows: nil)
23
+ start_url = normalize_url(url)
24
+ hops = []
25
+ visited_urls = []
26
+ loop_detected = false
27
+ loop_type = nil
28
+ current_url = start_url
29
+
30
+ @max_hops.times do |hop_index|
31
+ if visited_urls.include?(current_url)
32
+ loop_detected = true
33
+ loop_type = :redirect_loop
34
+ hops << {
35
+ hop: hop_index + 1,
36
+ url: current_url,
37
+ status_code: 0,
38
+ duration_ms: 0,
39
+ error: "Redirect loop detected: '#{current_url}' previously visited in chain"
40
+ }
41
+ break
42
+ end
43
+
44
+ visited_urls << current_url
45
+ hop_data = fetch_hop(current_url, hop_index + 1)
46
+ hops << hop_data
47
+
48
+ break if hop_data[:error]
49
+
50
+ # Check for HTTP redirect
51
+ if hop_data[:is_redirect] && hop_data[:redirect_target]
52
+ next_url = hop_data[:redirect_target]
53
+ current_url = next_url
54
+ else
55
+ # Final destination reached, inspect canonical
56
+ break
57
+ end
58
+ end
59
+
60
+ final_hop = hops.reject { |h| h[:error] && h[:status_code] == 0 }.last || hops.last || {}
61
+ final_url = final_hop[:url] || current_url
62
+ final_canonical = final_hop[:canonical_url]
63
+
64
+ # Check for Canonical Loop: Final page's canonical points back to an earlier hop in the chain
65
+ canonical_loop = false
66
+ if final_canonical && normalize_url(final_canonical) != normalize_url(final_url) && visited_urls.include?(normalize_url(final_canonical))
67
+ canonical_loop = true
68
+ loop_detected = true
69
+ loop_type = :mixed_canonical_loop
70
+ end
71
+
72
+ # Canonical target status check: if canonical points to a different URL, check if that canonical itself redirects
73
+ canonical_status = nil
74
+ canonical_redirects = false
75
+ if final_canonical && normalize_url(final_canonical) != normalize_url(final_url)
76
+ canonical_info = check_canonical_target(final_canonical)
77
+ canonical_status = canonical_info[:status_code]
78
+ canonical_redirects = canonical_info[:is_redirect]
79
+ end
80
+
81
+ redirect_hops_count = hops.count { |h| h[:is_redirect] }
82
+ total_duration_ms = hops.sum { |h| h[:duration_ms] || 0 }.round(1)
83
+
84
+ # Equity calculation
85
+ # Base equity = 1.0 (100%). Each redirect hop loses ~15%. Canonical loops destroy equity completely.
86
+ equity_retention = if canonical_loop || loop_type == :redirect_loop
87
+ 0.0
88
+ elsif redirect_hops_count == 0
89
+ 1.0
90
+ else
91
+ ((1.0 - EQUITY_DECAY_PER_HOP)**redirect_hops_count).round(4)
92
+ end
93
+ equity_loss_pct = ((1.0 - equity_retention) * 100).round(1)
94
+
95
+ # Severity classification
96
+ issues = []
97
+ severity = :healthy
98
+
99
+ if loop_detected
100
+ severity = :critical
101
+ issues << {
102
+ code: :loop,
103
+ type: loop_type,
104
+ message: loop_type == :mixed_canonical_loop ?
105
+ "Mixed Canonical Loop: Final URL '#{final_url}' canonicalizes to '#{final_canonical}', which redirects back into chain" :
106
+ "Infinite Redirect Loop detected at '#{current_url}'"
107
+ }
108
+ end
109
+
110
+ if redirect_hops_count >= 2
111
+ severity = :warning if severity == :healthy
112
+ issues << {
113
+ code: :redirect_chain,
114
+ message: "Multi-hop redirect chain detected: #{redirect_hops_count} hops (#{hops.map { |h| h[:status_code] }.compact.join(' -> ')})"
115
+ }
116
+ end
117
+
118
+ if canonical_redirects
119
+ severity = :critical
120
+ issues << {
121
+ code: :canonical_redirects,
122
+ message: "Target canonical URL '#{final_canonical}' issues a redirect (#{canonical_status}). Google ignores redirecting canonicals."
123
+ }
124
+ end
125
+
126
+ if final_canonical.nil? && final_hop[:status_code] == 200
127
+ issues << {
128
+ code: :missing_canonical,
129
+ message: "Final destination '#{final_url}' is missing a rel=\"canonical\" tag"
130
+ }
131
+ end
132
+
133
+ # Cross-reference GSC traffic if rows provided
134
+ traffic_at_risk = extract_gsc_traffic(visited_urls + [final_canonical].compact, gsc_rows)
135
+
136
+ # Generate server collapse rewrite rules
137
+ server_rules = generate_collapse_rules(start_url, final_canonical || final_url)
138
+
139
+ {
140
+ start_url: start_url,
141
+ final_url: final_url,
142
+ canonical_url: final_canonical,
143
+ redirect_hops: redirect_hops_count,
144
+ total_hops: hops.length,
145
+ total_duration_ms: total_duration_ms,
146
+ equity_retention_pct: (equity_retention * 100).round(1),
147
+ equity_loss_pct: equity_loss_pct,
148
+ loop_detected: loop_detected,
149
+ loop_type: loop_type,
150
+ canonical_loop: canonical_loop,
151
+ canonical_status: canonical_status,
152
+ canonical_redirects: canonical_redirects,
153
+ severity: severity,
154
+ issues: issues,
155
+ traffic_at_risk: traffic_at_risk,
156
+ server_rules: server_rules,
157
+ hops: hops
158
+ }
159
+ end
160
+
161
+ # Batch audit an array of URLs or top pages
162
+ def audit_batch(urls, gsc_rows: nil, concurrency: 4)
163
+ clean_urls = urls.map { |u| normalize_url(u) }.uniq
164
+ results = []
165
+
166
+ # Thread pool for faster batch audits
167
+ queue = Queue.new
168
+ clean_urls.each { |u| queue << u }
169
+ mutex = Mutex.new
170
+
171
+ workers = [concurrency, clean_urls.size].min
172
+ workers = 1 if workers < 1
173
+
174
+ threads = Array.new(workers) do
175
+ Thread.new do
176
+ until queue.empty?
177
+ url = begin
178
+ queue.pop(true)
179
+ rescue ThreadError
180
+ nil
181
+ end
182
+ break unless url
183
+
184
+ res = audit_url(url, gsc_rows: gsc_rows)
185
+ mutex.synchronize { results << res }
186
+ end
187
+ end
188
+ end
189
+ threads.each(&:join)
190
+
191
+ # Summary metrics
192
+ total_audited = results.size
193
+ chains = results.select { |r| r[:redirect_hops] >= 2 }
194
+ loops = results.select { |r| r[:loop_detected] }
195
+ canonical_issues = results.select { |r| r[:canonical_redirects] || r[:canonical_loop] }
196
+ total_equity_lost = results.sum { |r| r[:equity_loss_pct] }
197
+ avg_equity_retention = total_audited > 0 ? (results.sum { |r| r[:equity_retention_pct] } / total_audited).round(1) : 100.0
198
+ total_clicks_at_risk = results.sum { |r| r.dig(:traffic_at_risk, :total_clicks) || 0 }
199
+ total_impressions_at_risk = results.sum { |r| r.dig(:traffic_at_risk, :total_impressions) || 0 }
200
+
201
+ {
202
+ total_audited: total_audited,
203
+ redirect_chains_count: chains.size,
204
+ loops_count: loops.size,
205
+ canonical_issues_count: canonical_issues.size,
206
+ avg_equity_retention_pct: avg_equity_retention,
207
+ total_clicks_at_risk: total_clicks_at_risk,
208
+ total_impressions_at_risk: total_impressions_at_risk,
209
+ results: results.sort_by { |r| [r[:loop_detected] ? 0 : 1, -r[:redirect_hops], -r[:equity_loss_pct]] }
210
+ }
211
+ end
212
+
213
+ # Generate server redirect collapse rules (Nginx, Apache, Cloudflare, Vercel/Netlify)
214
+ def generate_collapse_rules(from_url, to_url)
215
+ from_uri = URI.parse(from_url) rescue nil
216
+ to_uri = URI.parse(to_url) rescue nil
217
+ return {} unless from_uri && to_uri
218
+
219
+ from_path = from_uri.path.empty? ? '/' : from_uri.path
220
+ from_path += "?#{from_uri.query}" if from_uri.query
221
+
222
+ to_target = to_url
223
+
224
+ nginx_rule = "rewrite ^#{Regexp.escape(from_uri.path)}/?$ #{to_target} permanent;"
225
+ apache_rule = "RewriteRule ^#{from_uri.path.sub(%r{^/}, '')}/?$ #{to_target} [R=301,L]"
226
+ cloudflare_expr = "(http.request.full_uri eq \"#{from_url}\") -> 301 to #{to_target}"
227
+ vercel_netlify = "#{from_path} #{to_target} 301!"
228
+
229
+ {
230
+ nginx: nginx_rule,
231
+ apache: apache_rule,
232
+ cloudflare: cloudflare_expr,
233
+ vercel_netlify: vercel_netlify
234
+ }
235
+ end
236
+
237
+ private
238
+
239
+ def normalize_url(url)
240
+ u = url.to_s.strip
241
+ u = "https://#{u}" unless u =~ %r{^https?://}
242
+ u
243
+ end
244
+
245
+ def fetch_hop(url, hop_num)
246
+ uri = URI.parse(url) rescue nil
247
+ return { hop: hop_num, url: url, error: "Invalid URI: #{url}", duration_ms: 0, status_code: 0 } unless uri && uri.host
248
+
249
+ start_t = Time.now
250
+ res = nil
251
+ body_content = ''
252
+
253
+ begin
254
+ Net::HTTP.start(uri.hostname, uri.port, use_ssl: (uri.scheme == 'https'), open_timeout: @timeout, read_timeout: @timeout) do |http|
255
+ req = Net::HTTP::Get.new(uri.request_uri)
256
+ req['User-Agent'] = @user_agent
257
+ req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
258
+ res = http.request(req)
259
+ body_content = res.body.to_s[0..25_000] if res.body # read first 25KB for head tags
260
+ end
261
+ rescue StandardError => e
262
+ return {
263
+ hop: hop_num,
264
+ url: url,
265
+ status_code: 0,
266
+ duration_ms: ((Time.now - start_t) * 1000).round(1),
267
+ error: e.message
268
+ }
269
+ end
270
+
271
+ duration_ms = ((Time.now - start_t) * 1000).round(1)
272
+ status_code = res.code.to_i
273
+ location = res['location']
274
+ redirect_target = nil
275
+
276
+ if [301, 302, 303, 307, 308].include?(status_code) && location
277
+ redirect_target = URI.join(url, location).to_s rescue location
278
+ end
279
+
280
+ # Extract canonical from header and HTML body
281
+ canonical_header = extract_canonical_header(res['link'])
282
+ canonical_html = extract_canonical_html(body_content, url)
283
+ canonical_url = canonical_header || canonical_html
284
+
285
+ {
286
+ hop: hop_num,
287
+ url: url,
288
+ status_code: status_code,
289
+ duration_ms: duration_ms,
290
+ is_redirect: !redirect_target.nil?,
291
+ redirect_target: redirect_target,
292
+ canonical_url: canonical_url,
293
+ canonical_source: canonical_header ? :http_header : (canonical_html ? :html_tag : nil),
294
+ x_robots_tag: res['x-robots-tag'],
295
+ server: res['server'],
296
+ content_type: res['content-type']
297
+ }
298
+ end
299
+
300
+ def extract_canonical_header(link_header)
301
+ return nil unless link_header
302
+ if link_header =~ /<([^>]+)>;\s*rel=["']?canonical["']?/i
303
+ $1.strip
304
+ end
305
+ end
306
+
307
+ def extract_canonical_html(html, base_url)
308
+ return nil unless html && !html.empty?
309
+ # Case-insensitive scan for <link ... rel="canonical" ... href="..." >
310
+ if html =~ /<link\b[^>]*\brel=["']canonical["'][^>]*>/i
311
+ tag = $&
312
+ if tag =~ /href=["']([^"']+)["']/i
313
+ raw_href = $1.strip
314
+ URI.join(base_url, raw_href).to_s rescue raw_href
315
+ end
316
+ elsif html =~ /<link\b[^>]*\bhref=["']([^"']+)["'][^>]*\brel=["']canonical["'][^>]*>/i
317
+ raw_href = $1.strip
318
+ URI.join(base_url, raw_href).to_s rescue raw_href
319
+ end
320
+ end
321
+
322
+ def check_canonical_target(canonical_url)
323
+ uri = URI.parse(canonical_url) rescue nil
324
+ return { status_code: 0, is_redirect: false } unless uri && uri.host
325
+
326
+ begin
327
+ Net::HTTP.start(uri.hostname, uri.port, use_ssl: (uri.scheme == 'https'), open_timeout: 4, read_timeout: 5) do |http|
328
+ req = Net::HTTP::Head.new(uri.request_uri)
329
+ req['User-Agent'] = @user_agent
330
+ res = http.request(req)
331
+ code = res.code.to_i
332
+ return { status_code: code, is_redirect: [301, 302, 303, 307, 308].include?(code) }
333
+ end
334
+ rescue StandardError
335
+ { status_code: 0, is_redirect: false }
336
+ end
337
+ end
338
+
339
+ def extract_gsc_traffic(urls, gsc_rows)
340
+ return { total_clicks: 0, total_impressions: 0, queries: [] } unless gsc_rows && gsc_rows.is_a?(Array)
341
+
342
+ clean_set = urls.map { |u| normalize_url(u).downcase }.to_set
343
+ matched_clicks = 0
344
+ matched_impressions = 0
345
+ matched_queries = []
346
+
347
+ gsc_rows.each do |r|
348
+ keys = r['keys'] || []
349
+ page = keys.find { |k| k =~ %r{^https?://} }
350
+ next unless page && clean_set.include?(normalize_url(page).downcase)
351
+
352
+ clicks = (r['clicks'] || 0).to_i
353
+ impressions = (r['impressions'] || 0).to_i
354
+ matched_clicks += clicks
355
+ matched_impressions += impressions
356
+ query = keys.find { |k| k !~ %r{^https?://} }
357
+ matched_queries << { query: query, clicks: clicks, impressions: impressions } if query
358
+ end
359
+
360
+ {
361
+ total_clicks: matched_clicks,
362
+ total_impressions: matched_impressions,
363
+ queries: matched_queries.sort_by { |q| -q[:clicks] }.first(5)
364
+ }
365
+ end
366
+ end
367
+ end