gsc-cli 2.0.2 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. checksums.yaml +4 -4
  2. data/AUTH.md +205 -0
  3. data/FUNDING.md +120 -0
  4. data/README.md +463 -299
  5. data/bin/gsc +29158 -4921
  6. data/dist/gsc +29158 -4921
  7. data/lib/gsc/aio_hunter.rb +343 -0
  8. data/lib/gsc/answer_synthesizer.rb +157 -0
  9. data/lib/gsc/api.rb +53 -1
  10. data/lib/gsc/auth.rb +26 -0
  11. data/lib/gsc/backlinks_manager.rb +96 -0
  12. data/lib/gsc/brand_segmenter.rb +140 -0
  13. data/lib/gsc/cache_manager.rb +806 -0
  14. data/lib/gsc/cannibalization_analyzer.rb +141 -0
  15. data/lib/gsc/canonical_chains.rb +367 -0
  16. data/lib/gsc/citation_simulator.rb +339 -0
  17. data/lib/gsc/cli/aio_hunter.rb +154 -0
  18. data/lib/gsc/cli/analytics.rb +788 -0
  19. data/lib/gsc/cli/audit.rb +1976 -0
  20. data/lib/gsc/cli/base.rb +384 -0
  21. data/lib/gsc/cli/cache.rb +266 -0
  22. data/lib/gsc/cli/canonical.rb +223 -0
  23. data/lib/gsc/cli/citation_simulator.rb +152 -0
  24. data/lib/gsc/cli/dashboard.rb +354 -0
  25. data/lib/gsc/cli/doctor.rb +129 -0
  26. data/lib/gsc/cli/eeat.rb +125 -0
  27. data/lib/gsc/cli/ga4.rb +852 -0
  28. data/lib/gsc/cli/growth.rb +650 -0
  29. data/lib/gsc/cli/hreflang.rb +164 -0
  30. data/lib/gsc/cli/image_seo.rb +162 -0
  31. data/lib/gsc/cli/indexing.rb +458 -0
  32. data/lib/gsc/cli/intent_shift.rb +125 -0
  33. data/lib/gsc/cli/keyword_value.rb +134 -0
  34. data/lib/gsc/cli/keywords.rb +795 -0
  35. data/lib/gsc/cli/landing_roi.rb +308 -0
  36. data/lib/gsc/cli/low_ctr.rb +213 -0
  37. data/lib/gsc/cli/mobile_parity.rb +150 -0
  38. data/lib/gsc/cli/report.rb +100 -0
  39. data/lib/gsc/cli/rich_results.rb +172 -0
  40. data/lib/gsc/cli/schema_generate.rb +149 -0
  41. data/lib/gsc/cli/seasonal.rb +232 -0
  42. data/lib/gsc/cli/security.rb +153 -0
  43. data/lib/gsc/cli/setup.rb +1291 -0
  44. data/lib/gsc/cli/sitemap_tree.rb +143 -0
  45. data/lib/gsc/cli/skill_pack.rb +62 -0
  46. data/lib/gsc/cli/soft_404.rb +199 -0
  47. data/lib/gsc/cli/sparkline.rb +227 -0
  48. data/lib/gsc/cli/watchdog.rb +150 -0
  49. data/lib/gsc/cli/zombie_purger.rb +208 -0
  50. data/lib/gsc/cli.rb +706 -5111
  51. data/lib/gsc/cli_advanced.rb +1513 -0
  52. data/lib/gsc/client.rb +17 -2
  53. data/lib/gsc/color.rb +16 -1
  54. data/lib/gsc/command_registry.rb +47 -9
  55. data/lib/gsc/config.rb +11 -2
  56. data/lib/gsc/content_gap.rb +112 -0
  57. data/lib/gsc/ctr_curve.rb +115 -0
  58. data/lib/gsc/decay_predictor.rb +322 -0
  59. data/lib/gsc/doctor.rb +434 -0
  60. data/lib/gsc/eeat_auditor.rb +428 -0
  61. data/lib/gsc/entity_auditor.rb +229 -0
  62. data/lib/gsc/firewall_scanner.rb +733 -0
  63. data/lib/gsc/geo_auditor.rb +368 -0
  64. data/lib/gsc/google_suggest.rb +109 -0
  65. data/lib/gsc/google_trends.rb +8 -1
  66. data/lib/gsc/heading_validator.rb +283 -0
  67. data/lib/gsc/hreflang_validator.rb +412 -0
  68. data/lib/gsc/image_seo.rb +286 -0
  69. data/lib/gsc/indexing_queue.rb +179 -0
  70. data/lib/gsc/indexnow.rb +93 -0
  71. data/lib/gsc/intent_shift.rb +188 -0
  72. data/lib/gsc/internal_links.rb +249 -0
  73. data/lib/gsc/keyword_value.rb +191 -0
  74. data/lib/gsc/landing_roi.rb +195 -0
  75. data/lib/gsc/llms_generator.rb +425 -0
  76. data/lib/gsc/low_ctr_rewriter.rb +408 -0
  77. data/lib/gsc/mobile_parity.rb +222 -0
  78. data/lib/gsc/network_tracer.rb +93 -0
  79. data/lib/gsc/open_page_rank.rb +72 -0
  80. data/lib/gsc/page_analyzer.rb +47 -7
  81. data/lib/gsc/page_comparator.rb +108 -0
  82. data/lib/gsc/page_speed.rb +110 -0
  83. data/lib/gsc/prompts.rb +38 -29
  84. data/lib/gsc/questions_harvester.rb +178 -0
  85. data/lib/gsc/report_generator.rb +461 -0
  86. data/lib/gsc/rich_results.rb +388 -0
  87. data/lib/gsc/robots_checker.rb +114 -0
  88. data/lib/gsc/schema_generator.rb +788 -0
  89. data/lib/gsc/schema_validator.rb +120 -0
  90. data/lib/gsc/seasonal_predictor.rb +381 -0
  91. data/lib/gsc/security_scanner.rb +496 -0
  92. data/lib/gsc/serp_feature_detector.rb +359 -0
  93. data/lib/gsc/serp_preview.rb +152 -0
  94. data/lib/gsc/site_crawler.rb +113 -21
  95. data/lib/gsc/sitemap_loader.rb +15 -4
  96. data/lib/gsc/sitemap_tree.rb +301 -0
  97. data/lib/gsc/skill_pack.rb +195 -0
  98. data/lib/gsc/soft_404_analyzer.rb +385 -0
  99. data/lib/gsc/sparkline.rb +171 -0
  100. data/lib/gsc/speed_correlator.rb +416 -0
  101. data/lib/gsc/striking_playbook.rb +190 -0
  102. data/lib/gsc/title_optimizer.rb +420 -0
  103. data/lib/gsc/vault.rb +260 -0
  104. data/lib/gsc/version.rb +1 -1
  105. data/lib/gsc/watchdog.rb +235 -0
  106. data/lib/gsc/zombie_purger.rb +366 -0
  107. data/lib/gsc.rb +144 -0
  108. metadata +91 -2
@@ -0,0 +1,283 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'json'
4
+
5
+ module GSC
6
+ class HeadingValidator
7
+ attr_reader :headings, :counts, :violations, :score, :grade, :depth
8
+
9
+ def self.analyze(html, target_keywords: nil)
10
+ new(html, target_keywords: target_keywords).audit
11
+ end
12
+
13
+ def initialize(html, target_keywords: nil)
14
+ @html = html.to_s.dup.force_encoding('UTF-8').scrub
15
+ @target_keywords = extract_keyword_tokens(target_keywords)
16
+ @headings = []
17
+ @counts = { 'h1' => 0, 'h2' => 0, 'h3' => 0, 'h4' => 0, 'h5' => 0, 'h6' => 0 }
18
+ @violations = []
19
+ @score = 100
20
+ @grade = 'A'
21
+ @depth = 0
22
+ end
23
+
24
+ def audit
25
+ parse_headings!
26
+ validate_hierarchy!
27
+ calculate_score!
28
+ audit_keywords! if @target_keywords && !@target_keywords.empty?
29
+
30
+ {
31
+ score: @score,
32
+ grade: @grade,
33
+ depth: @depth,
34
+ count: @headings.size,
35
+ counts: @counts,
36
+ h1_count: @counts['h1'],
37
+ empty_count: @headings.count { |h| h[:empty] },
38
+ headings: @headings,
39
+ violations: @violations,
40
+ keyword_analysis: @keyword_analysis,
41
+ summary: summary_text,
42
+ ascii_tree: render_ascii_tree(colorize: false),
43
+ color_tree: render_ascii_tree(colorize: true)
44
+ }
45
+ end
46
+
47
+ def render_ascii_tree(colorize: false)
48
+ return ' (No headings found on page)' if @headings.empty?
49
+
50
+ lines = []
51
+ @headings.each_with_index do |h, idx|
52
+ level = h[:level]
53
+ tag_str = "[#{h[:tag].upcase}]"
54
+ prefix = ' ' + (' ' * (level - 1)) + (level == 1 ? '■ ' : '├── ')
55
+
56
+ text_str = h[:empty] ? '(Empty Heading Tag)' : h[:text]
57
+ # Truncate very long heading in display
58
+ display_text = text_str.length > 75 ? "#{text_str[0..72]}..." : text_str
59
+
60
+ # Check if this heading has an attached violation
61
+ v_msgs = h[:violations].map { |v| v[:message] }
62
+ v_suffix = v_msgs.empty? ? '' : " ⚠️ #{v_msgs.join('; ')}"
63
+
64
+ if colorize && defined?(Color)
65
+ tag_color = case level
66
+ when 1 then Color::CYAN
67
+ when 2 then Color::BLUE
68
+ when 3 then Color::MAGENTA
69
+ else Color::GRAY
70
+ end
71
+ colored_tag = Color.c(tag_str, tag_color, Color::BOLD)
72
+ colored_text = h[:empty] ? Color.c(display_text, Color::RED, Color::DIM) : display_text
73
+ colored_suffix = v_suffix.empty? ? '' : Color.c(v_suffix, Color::YELLOW, Color::BOLD)
74
+ lines << "#{prefix}#{colored_tag} #{colored_text}#{colored_suffix}"
75
+ else
76
+ lines << "#{prefix}#{tag_str} #{display_text}#{v_suffix}"
77
+ end
78
+ end
79
+
80
+ lines.join("\n")
81
+ end
82
+
83
+ private
84
+
85
+ def parse_headings!
86
+ idx = 0
87
+ @html.scan(%r{<(h[1-6])(?:\s+[^>]*)?>(.*?)</\1>}im) do |tag, content|
88
+ idx += 1
89
+ clean = clean_text(content)
90
+ level = tag[1].to_i
91
+ is_empty = clean.empty?
92
+
93
+ h_info = {
94
+ index: idx,
95
+ tag: tag.downcase,
96
+ level: level,
97
+ text: clean,
98
+ length: clean.length,
99
+ empty: is_empty,
100
+ violations: []
101
+ }
102
+
103
+ @headings << h_info
104
+ @counts[tag.downcase] += 1
105
+ @depth = level if level > @depth
106
+ end
107
+ end
108
+
109
+ def validate_hierarchy!
110
+ return if @headings.empty?
111
+
112
+ # 1. Check if first heading is H1
113
+ first_h = @headings.first
114
+ if first_h[:level] != 1
115
+ v = {
116
+ type: :first_not_h1,
117
+ severity: :warning,
118
+ tag: first_h[:tag],
119
+ index: 1,
120
+ message: "Page starts with <#{first_h[:tag]}> instead of <h1>"
121
+ }
122
+ @violations << v
123
+ first_h[:violations] << v
124
+ end
125
+
126
+ # 2. Check H1 count
127
+ if @counts['h1'] == 0
128
+ @violations << {
129
+ type: :missing_h1,
130
+ severity: :critical,
131
+ message: 'Missing <h1> tag. Document has 0 top-level headings.'
132
+ }
133
+ elsif @counts['h1'] > 1
134
+ extra_h1s = @headings.select { |h| h[:tag] == 'h1' }[1..]
135
+ extra_h1s.each do |eh|
136
+ v = {
137
+ type: :multiple_h1,
138
+ severity: :warning,
139
+ tag: 'h1',
140
+ index: eh[:index],
141
+ message: "Multiple <h1> tag detected at position ##{eh[:index]} ('#{eh[:text][0..40]}...')"
142
+ }
143
+ @violations << v
144
+ eh[:violations] << v
145
+ end
146
+ end
147
+
148
+ # 3. Check sequential depth skips (e.g. H1 -> H3 or H2 -> H4)
149
+ prev_level = nil
150
+ @headings.each do |h|
151
+ curr_level = h[:level]
152
+
153
+ if prev_level && curr_level > prev_level + 1
154
+ skipped = (prev_level + 1...curr_level).map { |lvl| "<h#{lvl}>" }.join(', ')
155
+ v = {
156
+ type: :skipped_level,
157
+ severity: :warning,
158
+ tag: h[:tag],
159
+ index: h[:index],
160
+ from_level: prev_level,
161
+ to_level: curr_level,
162
+ message: "Skipped heading level: jumped from <h#{prev_level}> to <h#{curr_level}> (skipped #{skipped})"
163
+ }
164
+ @violations << v
165
+ h[:violations] << v
166
+ end
167
+
168
+ # 4. Check empty heading
169
+ if h[:empty]
170
+ v = {
171
+ type: :empty_heading,
172
+ severity: :error,
173
+ tag: h[:tag],
174
+ index: h[:index],
175
+ message: "Empty <#{h[:tag]}> tag with no text content"
176
+ }
177
+ @violations << v
178
+ h[:violations] << v
179
+ end
180
+
181
+ # 5. Check overlong heading (> 80 chars)
182
+ if h[:length] > 80
183
+ v = {
184
+ type: :overlong_heading,
185
+ severity: :info,
186
+ tag: h[:tag],
187
+ index: h[:index],
188
+ message: "Heading length (#{h[:length]} chars) exceeds 80 chars; risk of topical dilution"
189
+ }
190
+ @violations << v
191
+ h[:violations] << v
192
+ end
193
+
194
+ prev_level = curr_level
195
+ end
196
+ end
197
+
198
+ def calculate_score!
199
+ if @headings.empty?
200
+ @score = 0
201
+ @grade = 'F'
202
+ return
203
+ end
204
+
205
+ deductions = 0
206
+ @violations.each do |v|
207
+ deductions += case v[:type]
208
+ when :missing_h1 then 30
209
+ when :multiple_h1 then 15
210
+ when :skipped_level then 10
211
+ when :empty_heading then 10
212
+ when :first_not_h1 then 10
213
+ when :overlong_heading then 5
214
+ else 5
215
+ end
216
+ end
217
+
218
+ @score = [100 - deductions, 0].max
219
+ @grade = case @score
220
+ when 90..100 then 'A'
221
+ when 80..89 then 'B'
222
+ when 70..79 then 'C'
223
+ when 60..69 then 'D'
224
+ else 'F'
225
+ end
226
+ end
227
+
228
+ def audit_keywords!
229
+ h1 = @headings.find { |h| h[:tag] == 'h1' }
230
+ h1_matches = []
231
+ if h1
232
+ h1_text = h1[:text].downcase
233
+ h1_matches = @target_keywords.select { |kw| h1_text.include?(kw) }
234
+ end
235
+
236
+ h2_matches = {}
237
+ @headings.select { |h| h[:tag] == 'h2' }.each do |h2|
238
+ h2_text = h2[:text].downcase
239
+ matched = @target_keywords.select { |kw| h2_text.include?(kw) }
240
+ h2_matches[h2[:text]] = matched unless matched.empty?
241
+ end
242
+
243
+ @keyword_analysis = {
244
+ target_keywords: @target_keywords,
245
+ h1_has_keyword: !h1_matches.empty?,
246
+ h1_matched_keywords: h1_matches,
247
+ h2_matched_count: h2_matches.size,
248
+ h2_matches: h2_matches
249
+ }
250
+ end
251
+
252
+ def extract_keyword_tokens(keywords)
253
+ return [] if keywords.nil? || keywords.to_s.strip.empty?
254
+
255
+ if keywords.is_a?(Array)
256
+ keywords.map { |k| k.to_s.downcase.strip }.reject(&:empty?)
257
+ else
258
+ keywords.to_s.downcase.split(/[,|\s]+/).map(&:strip).reject { |w| w.length < 3 }
259
+ end
260
+ end
261
+
262
+ def clean_text(str)
263
+ str.to_s
264
+ .dup
265
+ .force_encoding('UTF-8')
266
+ .scrub
267
+ .gsub(/<[^>]+>/, ' ')
268
+ .gsub(/&amp;/, '&')
269
+ .gsub(/&lt;/, '<')
270
+ .gsub(/&gt;/, '>')
271
+ .gsub(/&quot;/, '"')
272
+ .gsub(/&#39;/, "'")
273
+ .gsub(/&nbsp;/, ' ')
274
+ .gsub(/\s+/, ' ')
275
+ .strip
276
+ end
277
+
278
+ def summary_text
279
+ counts_str = @counts.select { |_k, v| v > 0 }.map { |k, v| "#{v}x #{k.upcase}" }.join(', ')
280
+ "Heading Health Score: #{@score}/100 (Grade #{@grade}) | Depth: H#{@depth} | Total: #{@headings.size} (#{counts_str}) | #{@violations.size} violations"
281
+ end
282
+ end
283
+ end
@@ -0,0 +1,412 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'net/http'
5
+ require 'uri'
6
+ require 'json'
7
+ require 'set'
8
+
9
+ module GSC
10
+ class HreflangValidator
11
+ # Common ISO 639-1 two-letter language codes
12
+ ISO_639_1_LANGUAGES = Set.new(%w[
13
+ aa ab ae af ak am an ar as av ay az ba be bg bh bi bm bn bo br bs ca ce ch co cr cs cu cv cy da de dv dz ee el
14
+ en eo es et eu fa ff fi fj fo fr fy ga gd gl gn gu gv ha he hi ho hr ht hu hy hz ia id ie ig ii ik io is it iu
15
+ ja jv ka kg ki kj kk kl km kn ko kr ks ku kv kw ky la lb lg li ln lo lt lu lv mg mh mi mk ml mn mr ms mt my na
16
+ nb nd ne ng nl nn no nr nv ny oc oj om or os pa pi pl ps pt qu rm rn ro ru rw sa sc sd se sg si sk sl sm sn so
17
+ sq sr ss st su sv sw ta te tg th ti tk tl tn to tr ts tt tw ty ug uk ur uz ve vi vo wa wo xh yi yo za zh zu
18
+ ]).freeze
19
+
20
+ # Common script codes (4 letters, Titlecase)
21
+ SCRIPT_CODES = Set.new(%w[
22
+ Hans Hant Cyrl Latn Arab Deva
23
+ ]).freeze
24
+
25
+ # ISO 3166-1 alpha-2 two-letter uppercase country/region codes
26
+ ISO_3166_1_REGIONS = Set.new(%w[
27
+ AD AE AF AG AI AL AM AO AQ AR AS AT AU AW AX AZ BA BB BD BE BF BG BH BI BJ BL BM BN BO BQ BR BS BT BV BW BY BZ
28
+ CA CC CD CF CG CH CI CK CL CM CN CO CR CU CV CW CX CY CZ DE DJ DK DM DO DZ EC EE EG EH ER ES ET FI FJ FK FM FO
29
+ FR GA GB GD GE GF GG GH GI GL GM GN GP GQ GR GS GT GU GW GY HK HM HN HR HT HU ID IE IL IM IN IO IQ IR IS IT JE
30
+ JM JO JP KE KG KH KI KM KN KP KR KW KY KZ LA LB LC LI LK LR LS LT LU LV LY MA MC MD ME MF MG MH MK ML MM MN MO
31
+ MP MQ MR MS MT MU MV MW MX MY MZ NA NC NE NF NG NI NL NO NP NR NU NZ OM PA PE PF PG PH PK PL PM PN PR PS PT PW
32
+ PY QA RE RO RS RU RW SA SB SC SD SE SG SH SI SJ SK SL SM SN SO SR SS ST SV SX SY SZ TC TD TF TG TH TJ TK TL TM
33
+ TN TO TR TT TV TW TZ UA UG UM US UY UZ VA VC VE VG VI VN VU WF WS YE YT ZA ZM ZW
34
+ ]).freeze
35
+
36
+ # Known common mistakes & corrections
37
+ CODE_CORRECTIONS = {
38
+ 'en-uk' => 'en-GB (The ISO 3166-1 region code for the UK is "GB", not "UK")',
39
+ 'uk' => 'uk is Ukrainian; for United Kingdom English, use "en-GB"',
40
+ 'jp' => 'jp is country code; for Japanese language, use "ja" or "ja-JP"',
41
+ 'kr' => 'kr is country code; for Korean language, use "ko" or "ko-KR"',
42
+ 'sp' => 'sp is invalid; for Spanish, use "es"',
43
+ 'ge' => 'ge is Georgia; for German, use "de"',
44
+ 'cz' => 'cz is Czech Republic country code; for Czech language, use "cs"',
45
+ 'se' => 'se is Northern Sami; for Swedish, use "sv" or "sv-SE"',
46
+ 'dk' => 'dk is Denmark country code; for Danish language, use "da" or "da-DK"',
47
+ 'gr' => 'gr is Greece country code; for Greek language, use "el" or "el-GR"',
48
+ 'es-la' => 'es-LA is invalid ("LA" is Laos, not Latin America; use "es-419" or country-specific codes)',
49
+ 'es-sa' => 'es-SA is invalid ("SA" is Saudi Arabia; use country-specific codes like es-AR, es-CL)',
50
+ 'cn' => 'cn is country code; for Chinese, use "zh", "zh-Hans", or "zh-Hant"'
51
+ }.freeze
52
+
53
+ attr_reader :options, :url, :html, :headers
54
+
55
+ def initialize(options = {})
56
+ @options = options
57
+ end
58
+
59
+ def self.audit(target, options = {})
60
+ new(options).audit(target)
61
+ end
62
+
63
+ def audit(target)
64
+ @url, @html, @headers = load_target(target)
65
+ tags = extract_hreflang_tags(@html, @headers, @url)
66
+
67
+ # Validate language/region codes
68
+ tags.each { |tag| validate_tag_code(tag) }
69
+
70
+ # Self-referencing check
71
+ self_ref_found = check_self_reference(tags, @url)
72
+
73
+ # x-default check
74
+ x_default_tag = tags.find { |t| t[:hreflang].downcase == 'x-default' }
75
+
76
+ # Graph Reciprocity Check (for remote URLs)
77
+ if @options[:check_reciprocity] != false && @url.match?(%r{^https?://})
78
+ verify_reciprocity(tags, @url)
79
+ end
80
+
81
+ evaluate_health(tags, self_ref_found, x_default_tag)
82
+ end
83
+
84
+ private
85
+
86
+ def load_target(target)
87
+ target_str = target.to_s.strip
88
+ if target_str.match?(%r{^https?://})
89
+ uri = URI.parse(target_str)
90
+ req = Net::HTTP::Get.new(uri)
91
+ req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) gsc-cli/2.1'
92
+ req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
93
+
94
+ res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 8, read_timeout: 10) do |http|
95
+ http.request(req)
96
+ end
97
+ [target_str, res.body.to_s.dup.force_encoding('UTF-8').scrub, res.to_hash]
98
+ elsif File.exist?(target_str)
99
+ [target_str, File.read(target_str, encoding: 'UTF-8'), {}]
100
+ else
101
+ # In-memory HTML string (e.g. test fixtures)
102
+ base = @options[:url] || 'https://example.com/'
103
+ [base, target_str, {}]
104
+ end
105
+ rescue StandardError => e
106
+ [target_str.to_s, "<html><head><!-- Error: #{e.message} --></head><body></body></html>", {}]
107
+ end
108
+
109
+ def extract_hreflang_tags(html_content, resp_headers, base_url)
110
+ tags = []
111
+
112
+ # 1. Extract from HTML <link ...>
113
+ html_content.scan(/<link\b([^>]*?)>/im) do |match|
114
+ attrs_str = match.first
115
+ rel = extract_attr(attrs_str, 'rel')&.downcase
116
+ next unless rel == 'alternate'
117
+
118
+ hreflang = extract_attr(attrs_str, 'hreflang')
119
+ href = extract_attr(attrs_str, 'href')
120
+ next if hreflang.nil? || href.nil? || href.empty?
121
+
122
+ resolved_href = resolve_url(href, base_url)
123
+ tags << {
124
+ source: :html,
125
+ hreflang: hreflang.strip,
126
+ href: resolved_href,
127
+ raw_href: href,
128
+ issues: [],
129
+ status: :pending
130
+ }
131
+ end
132
+
133
+ # 2. Extract from HTTP response headers (Link: <...>; rel="alternate"; hreflang="...")
134
+ link_headers = Array(resp_headers['link']) + Array(resp_headers['Link'])
135
+ link_headers.each do |lh|
136
+ lh.split(',').each do |part|
137
+ if part =~ /<([^>]+)>\s*;\s*rel\s*=\s*(?:["']?alternate["']?)/i
138
+ href = $1
139
+ hreflang = nil
140
+ if part =~ /hreflang\s*=\s*["']?([^;"'\s]+)["']?/i
141
+ hreflang = $1
142
+ end
143
+ if hreflang && href
144
+ resolved_href = resolve_url(href, base_url)
145
+ tags << {
146
+ source: :http_header,
147
+ hreflang: hreflang.strip,
148
+ href: resolved_href,
149
+ raw_href: href,
150
+ issues: [],
151
+ status: :pending
152
+ }
153
+ end
154
+ end
155
+ end
156
+ end
157
+
158
+ tags
159
+ end
160
+
161
+ def validate_tag_code(tag)
162
+ code = tag[:hreflang].strip
163
+ return if code.downcase == 'x-default'
164
+
165
+ # Check for uppercase language code error (e.g., "EN", "EN-us")
166
+ if code =~ /^[A-Z]{2}(?:-|$)/
167
+ tag[:issues] << :uppercase_language
168
+ end
169
+
170
+ # Check for known common misconceptions
171
+ lowered = code.downcase
172
+ if CODE_CORRECTIONS.key?(lowered)
173
+ tag[:issues] << :common_mistake
174
+ tag[:correction] = CODE_CORRECTIONS[lowered]
175
+ end
176
+
177
+ # Split by hyphen
178
+ parts = code.split('-')
179
+ lang = parts[0].downcase
180
+
181
+ unless ISO_639_1_LANGUAGES.include?(lang)
182
+ tag[:issues] << :invalid_language unless tag[:issues].include?(:common_mistake)
183
+ end
184
+
185
+ if parts.size == 2
186
+ subtag = parts[1]
187
+ if subtag.length == 4 # Script (e.g. Hans, Hant)
188
+ unless SCRIPT_CODES.include?(subtag.capitalize)
189
+ tag[:issues] << :invalid_script
190
+ end
191
+ elsif subtag.length == 2 # Region (e.g. US, GB)
192
+ region_upper = subtag.upcase
193
+ unless ISO_3166_1_REGIONS.include?(region_upper)
194
+ tag[:issues] << :invalid_region unless tag[:issues].include?(:common_mistake)
195
+ end
196
+ if subtag =~ /^[a-z]{2}$/
197
+ tag[:issues] << :lowercase_region
198
+ end
199
+ elsif subtag == '419' # UN M.49 code for Latin America & Caribbean (valid in Google)
200
+ # valid
201
+ else
202
+ tag[:issues] << :invalid_format
203
+ end
204
+ elsif parts.size > 2
205
+ # e.g., zh-Hans-CN
206
+ script = parts[1]
207
+ region = parts[2]
208
+ unless SCRIPT_CODES.include?(script.capitalize)
209
+ tag[:issues] << :invalid_script
210
+ end
211
+ unless ISO_3166_1_REGIONS.include?(region.upcase)
212
+ tag[:issues] << :invalid_region
213
+ end
214
+ end
215
+ end
216
+
217
+ def check_self_reference(tags, origin_url)
218
+ norm_origin = normalize_url(origin_url)
219
+ match = tags.find do |t|
220
+ normalize_url(t[:href]) == norm_origin && t[:hreflang].downcase != 'x-default'
221
+ end
222
+
223
+ !match.nil?
224
+ end
225
+
226
+ def verify_reciprocity(tags, origin_url)
227
+ norm_origin = normalize_url(origin_url)
228
+
229
+ tags.each do |tag|
230
+ norm_target = normalize_url(tag[:href])
231
+ if norm_target == norm_origin
232
+ tag[:status] = :self_referential
233
+ next
234
+ end
235
+
236
+ # Fetch the alternate page and check for reciprocal link
237
+ begin
238
+ uri = URI.parse(tag[:href])
239
+ req = Net::HTTP::Get.new(uri)
240
+ req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) gsc-cli/2.1'
241
+
242
+ res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 4, read_timeout: 5) do |http|
243
+ http.request(req)
244
+ end
245
+
246
+ if res.code.to_i >= 400
247
+ tag[:status] = :http_error
248
+ tag[:issues] << "http_#{res.code}".to_sym
249
+ next
250
+ end
251
+
252
+ dest_html = res.body.to_s.dup.force_encoding('UTF-8').scrub
253
+ dest_tags = extract_hreflang_tags(dest_html, res.to_hash, tag[:href])
254
+
255
+ # Check if dest_tags contains a link back to origin_url
256
+ return_tag = dest_tags.find { |dt| normalize_url(dt[:href]) == norm_origin }
257
+
258
+ if return_tag
259
+ tag[:status] = :confirmed
260
+ tag[:return_hreflang] = return_tag[:hreflang]
261
+ else
262
+ tag[:status] = :missing_return
263
+ tag[:issues] << :missing_reciprocal_tag
264
+ end
265
+ rescue StandardError => e
266
+ tag[:status] = :unreachable
267
+ tag[:issues] << :network_error
268
+ end
269
+ end
270
+ end
271
+
272
+ def evaluate_health(tags, self_ref_found, x_default_tag)
273
+ total = tags.size
274
+
275
+ if total.zero?
276
+ return {
277
+ url: @url,
278
+ total_tags: 0,
279
+ health_score: 50.0,
280
+ grade: 'C',
281
+ self_reference: false,
282
+ has_x_default: false,
283
+ reciprocal_count: 0,
284
+ unreciprocated_count: 0,
285
+ invalid_code_count: 0,
286
+ issues_summary: { missing_hreflang: 1 },
287
+ prescriptions: ['Add bidirectional hreflang alternate tags to properly localize international audience.'],
288
+ cluster_snippets: { html: '', sitemap_xml: '' },
289
+ tags: []
290
+ }
291
+ end
292
+
293
+ invalid_codes = tags.count { |t| (t[:issues] & %i[invalid_language invalid_region invalid_script common_mistake uppercase_language]).any? }
294
+ unreciprocated = tags.count { |t| t[:status] == :missing_return }
295
+ confirmed = tags.count { |t| t[:status] == :confirmed || t[:status] == :self_referential }
296
+
297
+ score = 100.0
298
+ score -= 25.0 unless self_ref_found
299
+ score -= 15.0 unless x_default_tag
300
+ score -= [invalid_codes * 15.0, 30.0].min
301
+ score -= [unreciprocated * 20.0, 40.0].min
302
+
303
+ score = [[score.round(1), 100.0].min, 0.0].max
304
+ grade = compute_grade(score)
305
+
306
+ summary = {
307
+ total_tags: total,
308
+ self_referencing: self_ref_found,
309
+ x_default: !x_default_tag.nil?,
310
+ confirmed_reciprocal: confirmed,
311
+ unreciprocated: unreciprocated,
312
+ invalid_codes: invalid_codes
313
+ }
314
+
315
+ prescriptions = generate_prescriptions(self_ref_found, x_default_tag, invalid_codes, unreciprocated, tags)
316
+ snippets = generate_cluster_snippets(tags, @url)
317
+
318
+ {
319
+ url: @url,
320
+ total_tags: total,
321
+ health_score: score,
322
+ grade: grade,
323
+ self_reference: self_ref_found,
324
+ has_x_default: !x_default_tag.nil?,
325
+ reciprocal_count: confirmed,
326
+ unreciprocated_count: unreciprocated,
327
+ invalid_code_count: invalid_codes,
328
+ issues_summary: summary,
329
+ prescriptions: prescriptions,
330
+ cluster_snippets: snippets,
331
+ tags: tags
332
+ }
333
+ end
334
+
335
+ def generate_prescriptions(self_ref, x_def, invalid_count, unrecip_count, tags)
336
+ recs = []
337
+
338
+ unless self_ref
339
+ recs << "Add a self-referencing hreflang tag pointing back to #{@url}. Per Google guidelines, every language variant must include its own URL."
340
+ end
341
+
342
+ unless x_def
343
+ recs << "Include an 'x-default' hreflang tag to handle international users whose language/region is not explicitly matched."
344
+ end
345
+
346
+ if invalid_count > 0
347
+ mistakes = tags.select { |t| t[:correction] }
348
+ if mistakes.any?
349
+ mistakes.each do |m|
350
+ recs << "Fix invalid hreflang code '#{m[:hreflang]}': #{m[:correction]}."
351
+ end
352
+ else
353
+ recs << "Correct #{invalid_count} invalid ISO 639-1 language or ISO 3166-1 region codes."
354
+ end
355
+ end
356
+
357
+ if unrecip_count > 0
358
+ recs << "Resolve #{unrecip_count} missing return tags. Hreflang links must be bidirectional; alternate pages must point back to origin."
359
+ end
360
+
361
+ recs
362
+ end
363
+
364
+ def generate_cluster_snippets(tags, base_url)
365
+ html_lines = []
366
+ xml_lines = []
367
+
368
+ tags.each do |t|
369
+ html_lines << "<link rel=\"alternate\" hreflang=\"#{t[:hreflang]}\" href=\"#{t[:href]}\" />"
370
+ xml_lines << " <xhtml:link rel=\"alternate\" hreflang=\"#{t[:hreflang]}\" href=\"#{t[:href]}\" />"
371
+ end
372
+
373
+ {
374
+ html: html_lines.join("\n"),
375
+ sitemap_xml: xml_lines.join("\n")
376
+ }
377
+ end
378
+
379
+ def compute_grade(score)
380
+ case score
381
+ when 90.0..100.0 then 'A+'
382
+ when 80.0...90.0 then 'A'
383
+ when 70.0...80.0 then 'B'
384
+ when 55.0...70.0 then 'C'
385
+ when 40.0...55.0 then 'D'
386
+ else 'F'
387
+ end
388
+ end
389
+
390
+ def extract_attr(tag_str, attr_name)
391
+ if tag_str =~ /\b#{attr_name}\s*=\s*(['"])(.*?)\1/i
392
+ $2.strip
393
+ elsif tag_str =~ /\b#{attr_name}\s*=\s*([^\s>]+)/i
394
+ $1.strip
395
+ else
396
+ nil
397
+ end
398
+ end
399
+
400
+ def resolve_url(href, base)
401
+ return href if href.match?(%r{^https?://})
402
+ URI.join(base, href).to_s
403
+ rescue StandardError
404
+ href
405
+ end
406
+
407
+ def normalize_url(url)
408
+ u = url.to_s.strip.downcase.sub(%r{/$}, '')
409
+ u.split('#').first
410
+ end
411
+ end
412
+ end