gsc-cli 2.0.2 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/AUTH.md +205 -0
- data/FUNDING.md +120 -0
- data/README.md +463 -299
- data/bin/gsc +29158 -4921
- data/dist/gsc +29158 -4921
- data/lib/gsc/aio_hunter.rb +343 -0
- data/lib/gsc/answer_synthesizer.rb +157 -0
- data/lib/gsc/api.rb +53 -1
- data/lib/gsc/auth.rb +26 -0
- data/lib/gsc/backlinks_manager.rb +96 -0
- data/lib/gsc/brand_segmenter.rb +140 -0
- data/lib/gsc/cache_manager.rb +806 -0
- data/lib/gsc/cannibalization_analyzer.rb +141 -0
- data/lib/gsc/canonical_chains.rb +367 -0
- data/lib/gsc/citation_simulator.rb +339 -0
- data/lib/gsc/cli/aio_hunter.rb +154 -0
- data/lib/gsc/cli/analytics.rb +788 -0
- data/lib/gsc/cli/audit.rb +1976 -0
- data/lib/gsc/cli/base.rb +384 -0
- data/lib/gsc/cli/cache.rb +266 -0
- data/lib/gsc/cli/canonical.rb +223 -0
- data/lib/gsc/cli/citation_simulator.rb +152 -0
- data/lib/gsc/cli/dashboard.rb +354 -0
- data/lib/gsc/cli/doctor.rb +129 -0
- data/lib/gsc/cli/eeat.rb +125 -0
- data/lib/gsc/cli/ga4.rb +852 -0
- data/lib/gsc/cli/growth.rb +650 -0
- data/lib/gsc/cli/hreflang.rb +164 -0
- data/lib/gsc/cli/image_seo.rb +162 -0
- data/lib/gsc/cli/indexing.rb +458 -0
- data/lib/gsc/cli/intent_shift.rb +125 -0
- data/lib/gsc/cli/keyword_value.rb +134 -0
- data/lib/gsc/cli/keywords.rb +795 -0
- data/lib/gsc/cli/landing_roi.rb +308 -0
- data/lib/gsc/cli/low_ctr.rb +213 -0
- data/lib/gsc/cli/mobile_parity.rb +150 -0
- data/lib/gsc/cli/report.rb +100 -0
- data/lib/gsc/cli/rich_results.rb +172 -0
- data/lib/gsc/cli/schema_generate.rb +149 -0
- data/lib/gsc/cli/seasonal.rb +232 -0
- data/lib/gsc/cli/security.rb +153 -0
- data/lib/gsc/cli/setup.rb +1291 -0
- data/lib/gsc/cli/sitemap_tree.rb +143 -0
- data/lib/gsc/cli/skill_pack.rb +62 -0
- data/lib/gsc/cli/soft_404.rb +199 -0
- data/lib/gsc/cli/sparkline.rb +227 -0
- data/lib/gsc/cli/watchdog.rb +150 -0
- data/lib/gsc/cli/zombie_purger.rb +208 -0
- data/lib/gsc/cli.rb +706 -5111
- data/lib/gsc/cli_advanced.rb +1513 -0
- data/lib/gsc/client.rb +17 -2
- data/lib/gsc/color.rb +16 -1
- data/lib/gsc/command_registry.rb +47 -9
- data/lib/gsc/config.rb +11 -2
- data/lib/gsc/content_gap.rb +112 -0
- data/lib/gsc/ctr_curve.rb +115 -0
- data/lib/gsc/decay_predictor.rb +322 -0
- data/lib/gsc/doctor.rb +434 -0
- data/lib/gsc/eeat_auditor.rb +428 -0
- data/lib/gsc/entity_auditor.rb +229 -0
- data/lib/gsc/firewall_scanner.rb +733 -0
- data/lib/gsc/geo_auditor.rb +368 -0
- data/lib/gsc/google_suggest.rb +109 -0
- data/lib/gsc/google_trends.rb +8 -1
- data/lib/gsc/heading_validator.rb +283 -0
- data/lib/gsc/hreflang_validator.rb +412 -0
- data/lib/gsc/image_seo.rb +286 -0
- data/lib/gsc/indexing_queue.rb +179 -0
- data/lib/gsc/indexnow.rb +93 -0
- data/lib/gsc/intent_shift.rb +188 -0
- data/lib/gsc/internal_links.rb +249 -0
- data/lib/gsc/keyword_value.rb +191 -0
- data/lib/gsc/landing_roi.rb +195 -0
- data/lib/gsc/llms_generator.rb +425 -0
- data/lib/gsc/low_ctr_rewriter.rb +408 -0
- data/lib/gsc/mobile_parity.rb +222 -0
- data/lib/gsc/network_tracer.rb +93 -0
- data/lib/gsc/open_page_rank.rb +72 -0
- data/lib/gsc/page_analyzer.rb +47 -7
- data/lib/gsc/page_comparator.rb +108 -0
- data/lib/gsc/page_speed.rb +110 -0
- data/lib/gsc/prompts.rb +38 -29
- data/lib/gsc/questions_harvester.rb +178 -0
- data/lib/gsc/report_generator.rb +461 -0
- data/lib/gsc/rich_results.rb +388 -0
- data/lib/gsc/robots_checker.rb +114 -0
- data/lib/gsc/schema_generator.rb +788 -0
- data/lib/gsc/schema_validator.rb +120 -0
- data/lib/gsc/seasonal_predictor.rb +381 -0
- data/lib/gsc/security_scanner.rb +496 -0
- data/lib/gsc/serp_feature_detector.rb +359 -0
- data/lib/gsc/serp_preview.rb +152 -0
- data/lib/gsc/site_crawler.rb +113 -21
- data/lib/gsc/sitemap_loader.rb +15 -4
- data/lib/gsc/sitemap_tree.rb +301 -0
- data/lib/gsc/skill_pack.rb +195 -0
- data/lib/gsc/soft_404_analyzer.rb +385 -0
- data/lib/gsc/sparkline.rb +171 -0
- data/lib/gsc/speed_correlator.rb +416 -0
- data/lib/gsc/striking_playbook.rb +190 -0
- data/lib/gsc/title_optimizer.rb +420 -0
- data/lib/gsc/vault.rb +260 -0
- data/lib/gsc/version.rb +1 -1
- data/lib/gsc/watchdog.rb +235 -0
- data/lib/gsc/zombie_purger.rb +366 -0
- data/lib/gsc.rb +144 -0
- metadata +91 -2
|
@@ -0,0 +1,283 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'json'
|
|
4
|
+
|
|
5
|
+
module GSC
|
|
6
|
+
class HeadingValidator
|
|
7
|
+
attr_reader :headings, :counts, :violations, :score, :grade, :depth
|
|
8
|
+
|
|
9
|
+
def self.analyze(html, target_keywords: nil)
|
|
10
|
+
new(html, target_keywords: target_keywords).audit
|
|
11
|
+
end
|
|
12
|
+
|
|
13
|
+
def initialize(html, target_keywords: nil)
|
|
14
|
+
@html = html.to_s.dup.force_encoding('UTF-8').scrub
|
|
15
|
+
@target_keywords = extract_keyword_tokens(target_keywords)
|
|
16
|
+
@headings = []
|
|
17
|
+
@counts = { 'h1' => 0, 'h2' => 0, 'h3' => 0, 'h4' => 0, 'h5' => 0, 'h6' => 0 }
|
|
18
|
+
@violations = []
|
|
19
|
+
@score = 100
|
|
20
|
+
@grade = 'A'
|
|
21
|
+
@depth = 0
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
def audit
|
|
25
|
+
parse_headings!
|
|
26
|
+
validate_hierarchy!
|
|
27
|
+
calculate_score!
|
|
28
|
+
audit_keywords! if @target_keywords && !@target_keywords.empty?
|
|
29
|
+
|
|
30
|
+
{
|
|
31
|
+
score: @score,
|
|
32
|
+
grade: @grade,
|
|
33
|
+
depth: @depth,
|
|
34
|
+
count: @headings.size,
|
|
35
|
+
counts: @counts,
|
|
36
|
+
h1_count: @counts['h1'],
|
|
37
|
+
empty_count: @headings.count { |h| h[:empty] },
|
|
38
|
+
headings: @headings,
|
|
39
|
+
violations: @violations,
|
|
40
|
+
keyword_analysis: @keyword_analysis,
|
|
41
|
+
summary: summary_text,
|
|
42
|
+
ascii_tree: render_ascii_tree(colorize: false),
|
|
43
|
+
color_tree: render_ascii_tree(colorize: true)
|
|
44
|
+
}
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
def render_ascii_tree(colorize: false)
|
|
48
|
+
return ' (No headings found on page)' if @headings.empty?
|
|
49
|
+
|
|
50
|
+
lines = []
|
|
51
|
+
@headings.each_with_index do |h, idx|
|
|
52
|
+
level = h[:level]
|
|
53
|
+
tag_str = "[#{h[:tag].upcase}]"
|
|
54
|
+
prefix = ' ' + (' ' * (level - 1)) + (level == 1 ? '■ ' : '├── ')
|
|
55
|
+
|
|
56
|
+
text_str = h[:empty] ? '(Empty Heading Tag)' : h[:text]
|
|
57
|
+
# Truncate very long heading in display
|
|
58
|
+
display_text = text_str.length > 75 ? "#{text_str[0..72]}..." : text_str
|
|
59
|
+
|
|
60
|
+
# Check if this heading has an attached violation
|
|
61
|
+
v_msgs = h[:violations].map { |v| v[:message] }
|
|
62
|
+
v_suffix = v_msgs.empty? ? '' : " ⚠️ #{v_msgs.join('; ')}"
|
|
63
|
+
|
|
64
|
+
if colorize && defined?(Color)
|
|
65
|
+
tag_color = case level
|
|
66
|
+
when 1 then Color::CYAN
|
|
67
|
+
when 2 then Color::BLUE
|
|
68
|
+
when 3 then Color::MAGENTA
|
|
69
|
+
else Color::GRAY
|
|
70
|
+
end
|
|
71
|
+
colored_tag = Color.c(tag_str, tag_color, Color::BOLD)
|
|
72
|
+
colored_text = h[:empty] ? Color.c(display_text, Color::RED, Color::DIM) : display_text
|
|
73
|
+
colored_suffix = v_suffix.empty? ? '' : Color.c(v_suffix, Color::YELLOW, Color::BOLD)
|
|
74
|
+
lines << "#{prefix}#{colored_tag} #{colored_text}#{colored_suffix}"
|
|
75
|
+
else
|
|
76
|
+
lines << "#{prefix}#{tag_str} #{display_text}#{v_suffix}"
|
|
77
|
+
end
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
lines.join("\n")
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
private
|
|
84
|
+
|
|
85
|
+
def parse_headings!
|
|
86
|
+
idx = 0
|
|
87
|
+
@html.scan(%r{<(h[1-6])(?:\s+[^>]*)?>(.*?)</\1>}im) do |tag, content|
|
|
88
|
+
idx += 1
|
|
89
|
+
clean = clean_text(content)
|
|
90
|
+
level = tag[1].to_i
|
|
91
|
+
is_empty = clean.empty?
|
|
92
|
+
|
|
93
|
+
h_info = {
|
|
94
|
+
index: idx,
|
|
95
|
+
tag: tag.downcase,
|
|
96
|
+
level: level,
|
|
97
|
+
text: clean,
|
|
98
|
+
length: clean.length,
|
|
99
|
+
empty: is_empty,
|
|
100
|
+
violations: []
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
@headings << h_info
|
|
104
|
+
@counts[tag.downcase] += 1
|
|
105
|
+
@depth = level if level > @depth
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
def validate_hierarchy!
|
|
110
|
+
return if @headings.empty?
|
|
111
|
+
|
|
112
|
+
# 1. Check if first heading is H1
|
|
113
|
+
first_h = @headings.first
|
|
114
|
+
if first_h[:level] != 1
|
|
115
|
+
v = {
|
|
116
|
+
type: :first_not_h1,
|
|
117
|
+
severity: :warning,
|
|
118
|
+
tag: first_h[:tag],
|
|
119
|
+
index: 1,
|
|
120
|
+
message: "Page starts with <#{first_h[:tag]}> instead of <h1>"
|
|
121
|
+
}
|
|
122
|
+
@violations << v
|
|
123
|
+
first_h[:violations] << v
|
|
124
|
+
end
|
|
125
|
+
|
|
126
|
+
# 2. Check H1 count
|
|
127
|
+
if @counts['h1'] == 0
|
|
128
|
+
@violations << {
|
|
129
|
+
type: :missing_h1,
|
|
130
|
+
severity: :critical,
|
|
131
|
+
message: 'Missing <h1> tag. Document has 0 top-level headings.'
|
|
132
|
+
}
|
|
133
|
+
elsif @counts['h1'] > 1
|
|
134
|
+
extra_h1s = @headings.select { |h| h[:tag] == 'h1' }[1..]
|
|
135
|
+
extra_h1s.each do |eh|
|
|
136
|
+
v = {
|
|
137
|
+
type: :multiple_h1,
|
|
138
|
+
severity: :warning,
|
|
139
|
+
tag: 'h1',
|
|
140
|
+
index: eh[:index],
|
|
141
|
+
message: "Multiple <h1> tag detected at position ##{eh[:index]} ('#{eh[:text][0..40]}...')"
|
|
142
|
+
}
|
|
143
|
+
@violations << v
|
|
144
|
+
eh[:violations] << v
|
|
145
|
+
end
|
|
146
|
+
end
|
|
147
|
+
|
|
148
|
+
# 3. Check sequential depth skips (e.g. H1 -> H3 or H2 -> H4)
|
|
149
|
+
prev_level = nil
|
|
150
|
+
@headings.each do |h|
|
|
151
|
+
curr_level = h[:level]
|
|
152
|
+
|
|
153
|
+
if prev_level && curr_level > prev_level + 1
|
|
154
|
+
skipped = (prev_level + 1...curr_level).map { |lvl| "<h#{lvl}>" }.join(', ')
|
|
155
|
+
v = {
|
|
156
|
+
type: :skipped_level,
|
|
157
|
+
severity: :warning,
|
|
158
|
+
tag: h[:tag],
|
|
159
|
+
index: h[:index],
|
|
160
|
+
from_level: prev_level,
|
|
161
|
+
to_level: curr_level,
|
|
162
|
+
message: "Skipped heading level: jumped from <h#{prev_level}> to <h#{curr_level}> (skipped #{skipped})"
|
|
163
|
+
}
|
|
164
|
+
@violations << v
|
|
165
|
+
h[:violations] << v
|
|
166
|
+
end
|
|
167
|
+
|
|
168
|
+
# 4. Check empty heading
|
|
169
|
+
if h[:empty]
|
|
170
|
+
v = {
|
|
171
|
+
type: :empty_heading,
|
|
172
|
+
severity: :error,
|
|
173
|
+
tag: h[:tag],
|
|
174
|
+
index: h[:index],
|
|
175
|
+
message: "Empty <#{h[:tag]}> tag with no text content"
|
|
176
|
+
}
|
|
177
|
+
@violations << v
|
|
178
|
+
h[:violations] << v
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
# 5. Check overlong heading (> 80 chars)
|
|
182
|
+
if h[:length] > 80
|
|
183
|
+
v = {
|
|
184
|
+
type: :overlong_heading,
|
|
185
|
+
severity: :info,
|
|
186
|
+
tag: h[:tag],
|
|
187
|
+
index: h[:index],
|
|
188
|
+
message: "Heading length (#{h[:length]} chars) exceeds 80 chars; risk of topical dilution"
|
|
189
|
+
}
|
|
190
|
+
@violations << v
|
|
191
|
+
h[:violations] << v
|
|
192
|
+
end
|
|
193
|
+
|
|
194
|
+
prev_level = curr_level
|
|
195
|
+
end
|
|
196
|
+
end
|
|
197
|
+
|
|
198
|
+
def calculate_score!
|
|
199
|
+
if @headings.empty?
|
|
200
|
+
@score = 0
|
|
201
|
+
@grade = 'F'
|
|
202
|
+
return
|
|
203
|
+
end
|
|
204
|
+
|
|
205
|
+
deductions = 0
|
|
206
|
+
@violations.each do |v|
|
|
207
|
+
deductions += case v[:type]
|
|
208
|
+
when :missing_h1 then 30
|
|
209
|
+
when :multiple_h1 then 15
|
|
210
|
+
when :skipped_level then 10
|
|
211
|
+
when :empty_heading then 10
|
|
212
|
+
when :first_not_h1 then 10
|
|
213
|
+
when :overlong_heading then 5
|
|
214
|
+
else 5
|
|
215
|
+
end
|
|
216
|
+
end
|
|
217
|
+
|
|
218
|
+
@score = [100 - deductions, 0].max
|
|
219
|
+
@grade = case @score
|
|
220
|
+
when 90..100 then 'A'
|
|
221
|
+
when 80..89 then 'B'
|
|
222
|
+
when 70..79 then 'C'
|
|
223
|
+
when 60..69 then 'D'
|
|
224
|
+
else 'F'
|
|
225
|
+
end
|
|
226
|
+
end
|
|
227
|
+
|
|
228
|
+
def audit_keywords!
|
|
229
|
+
h1 = @headings.find { |h| h[:tag] == 'h1' }
|
|
230
|
+
h1_matches = []
|
|
231
|
+
if h1
|
|
232
|
+
h1_text = h1[:text].downcase
|
|
233
|
+
h1_matches = @target_keywords.select { |kw| h1_text.include?(kw) }
|
|
234
|
+
end
|
|
235
|
+
|
|
236
|
+
h2_matches = {}
|
|
237
|
+
@headings.select { |h| h[:tag] == 'h2' }.each do |h2|
|
|
238
|
+
h2_text = h2[:text].downcase
|
|
239
|
+
matched = @target_keywords.select { |kw| h2_text.include?(kw) }
|
|
240
|
+
h2_matches[h2[:text]] = matched unless matched.empty?
|
|
241
|
+
end
|
|
242
|
+
|
|
243
|
+
@keyword_analysis = {
|
|
244
|
+
target_keywords: @target_keywords,
|
|
245
|
+
h1_has_keyword: !h1_matches.empty?,
|
|
246
|
+
h1_matched_keywords: h1_matches,
|
|
247
|
+
h2_matched_count: h2_matches.size,
|
|
248
|
+
h2_matches: h2_matches
|
|
249
|
+
}
|
|
250
|
+
end
|
|
251
|
+
|
|
252
|
+
def extract_keyword_tokens(keywords)
|
|
253
|
+
return [] if keywords.nil? || keywords.to_s.strip.empty?
|
|
254
|
+
|
|
255
|
+
if keywords.is_a?(Array)
|
|
256
|
+
keywords.map { |k| k.to_s.downcase.strip }.reject(&:empty?)
|
|
257
|
+
else
|
|
258
|
+
keywords.to_s.downcase.split(/[,|\s]+/).map(&:strip).reject { |w| w.length < 3 }
|
|
259
|
+
end
|
|
260
|
+
end
|
|
261
|
+
|
|
262
|
+
def clean_text(str)
|
|
263
|
+
str.to_s
|
|
264
|
+
.dup
|
|
265
|
+
.force_encoding('UTF-8')
|
|
266
|
+
.scrub
|
|
267
|
+
.gsub(/<[^>]+>/, ' ')
|
|
268
|
+
.gsub(/&/, '&')
|
|
269
|
+
.gsub(/</, '<')
|
|
270
|
+
.gsub(/>/, '>')
|
|
271
|
+
.gsub(/"/, '"')
|
|
272
|
+
.gsub(/'/, "'")
|
|
273
|
+
.gsub(/ /, ' ')
|
|
274
|
+
.gsub(/\s+/, ' ')
|
|
275
|
+
.strip
|
|
276
|
+
end
|
|
277
|
+
|
|
278
|
+
def summary_text
|
|
279
|
+
counts_str = @counts.select { |_k, v| v > 0 }.map { |k, v| "#{v}x #{k.upcase}" }.join(', ')
|
|
280
|
+
"Heading Health Score: #{@score}/100 (Grade #{@grade}) | Depth: H#{@depth} | Total: #{@headings.size} (#{counts_str}) | #{@violations.size} violations"
|
|
281
|
+
end
|
|
282
|
+
end
|
|
283
|
+
end
|
|
@@ -0,0 +1,412 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'net/http'
|
|
5
|
+
require 'uri'
|
|
6
|
+
require 'json'
|
|
7
|
+
require 'set'
|
|
8
|
+
|
|
9
|
+
module GSC
|
|
10
|
+
class HreflangValidator
|
|
11
|
+
# Common ISO 639-1 two-letter language codes
|
|
12
|
+
ISO_639_1_LANGUAGES = Set.new(%w[
|
|
13
|
+
aa ab ae af ak am an ar as av ay az ba be bg bh bi bm bn bo br bs ca ce ch co cr cs cu cv cy da de dv dz ee el
|
|
14
|
+
en eo es et eu fa ff fi fj fo fr fy ga gd gl gn gu gv ha he hi ho hr ht hu hy hz ia id ie ig ii ik io is it iu
|
|
15
|
+
ja jv ka kg ki kj kk kl km kn ko kr ks ku kv kw ky la lb lg li ln lo lt lu lv mg mh mi mk ml mn mr ms mt my na
|
|
16
|
+
nb nd ne ng nl nn no nr nv ny oc oj om or os pa pi pl ps pt qu rm rn ro ru rw sa sc sd se sg si sk sl sm sn so
|
|
17
|
+
sq sr ss st su sv sw ta te tg th ti tk tl tn to tr ts tt tw ty ug uk ur uz ve vi vo wa wo xh yi yo za zh zu
|
|
18
|
+
]).freeze
|
|
19
|
+
|
|
20
|
+
# Common script codes (4 letters, Titlecase)
|
|
21
|
+
SCRIPT_CODES = Set.new(%w[
|
|
22
|
+
Hans Hant Cyrl Latn Arab Deva
|
|
23
|
+
]).freeze
|
|
24
|
+
|
|
25
|
+
# ISO 3166-1 alpha-2 two-letter uppercase country/region codes
|
|
26
|
+
ISO_3166_1_REGIONS = Set.new(%w[
|
|
27
|
+
AD AE AF AG AI AL AM AO AQ AR AS AT AU AW AX AZ BA BB BD BE BF BG BH BI BJ BL BM BN BO BQ BR BS BT BV BW BY BZ
|
|
28
|
+
CA CC CD CF CG CH CI CK CL CM CN CO CR CU CV CW CX CY CZ DE DJ DK DM DO DZ EC EE EG EH ER ES ET FI FJ FK FM FO
|
|
29
|
+
FR GA GB GD GE GF GG GH GI GL GM GN GP GQ GR GS GT GU GW GY HK HM HN HR HT HU ID IE IL IM IN IO IQ IR IS IT JE
|
|
30
|
+
JM JO JP KE KG KH KI KM KN KP KR KW KY KZ LA LB LC LI LK LR LS LT LU LV LY MA MC MD ME MF MG MH MK ML MM MN MO
|
|
31
|
+
MP MQ MR MS MT MU MV MW MX MY MZ NA NC NE NF NG NI NL NO NP NR NU NZ OM PA PE PF PG PH PK PL PM PN PR PS PT PW
|
|
32
|
+
PY QA RE RO RS RU RW SA SB SC SD SE SG SH SI SJ SK SL SM SN SO SR SS ST SV SX SY SZ TC TD TF TG TH TJ TK TL TM
|
|
33
|
+
TN TO TR TT TV TW TZ UA UG UM US UY UZ VA VC VE VG VI VN VU WF WS YE YT ZA ZM ZW
|
|
34
|
+
]).freeze
|
|
35
|
+
|
|
36
|
+
# Known common mistakes & corrections
|
|
37
|
+
CODE_CORRECTIONS = {
|
|
38
|
+
'en-uk' => 'en-GB (The ISO 3166-1 region code for the UK is "GB", not "UK")',
|
|
39
|
+
'uk' => 'uk is Ukrainian; for United Kingdom English, use "en-GB"',
|
|
40
|
+
'jp' => 'jp is country code; for Japanese language, use "ja" or "ja-JP"',
|
|
41
|
+
'kr' => 'kr is country code; for Korean language, use "ko" or "ko-KR"',
|
|
42
|
+
'sp' => 'sp is invalid; for Spanish, use "es"',
|
|
43
|
+
'ge' => 'ge is Georgia; for German, use "de"',
|
|
44
|
+
'cz' => 'cz is Czech Republic country code; for Czech language, use "cs"',
|
|
45
|
+
'se' => 'se is Northern Sami; for Swedish, use "sv" or "sv-SE"',
|
|
46
|
+
'dk' => 'dk is Denmark country code; for Danish language, use "da" or "da-DK"',
|
|
47
|
+
'gr' => 'gr is Greece country code; for Greek language, use "el" or "el-GR"',
|
|
48
|
+
'es-la' => 'es-LA is invalid ("LA" is Laos, not Latin America; use "es-419" or country-specific codes)',
|
|
49
|
+
'es-sa' => 'es-SA is invalid ("SA" is Saudi Arabia; use country-specific codes like es-AR, es-CL)',
|
|
50
|
+
'cn' => 'cn is country code; for Chinese, use "zh", "zh-Hans", or "zh-Hant"'
|
|
51
|
+
}.freeze
|
|
52
|
+
|
|
53
|
+
attr_reader :options, :url, :html, :headers
|
|
54
|
+
|
|
55
|
+
def initialize(options = {})
|
|
56
|
+
@options = options
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def self.audit(target, options = {})
|
|
60
|
+
new(options).audit(target)
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
def audit(target)
|
|
64
|
+
@url, @html, @headers = load_target(target)
|
|
65
|
+
tags = extract_hreflang_tags(@html, @headers, @url)
|
|
66
|
+
|
|
67
|
+
# Validate language/region codes
|
|
68
|
+
tags.each { |tag| validate_tag_code(tag) }
|
|
69
|
+
|
|
70
|
+
# Self-referencing check
|
|
71
|
+
self_ref_found = check_self_reference(tags, @url)
|
|
72
|
+
|
|
73
|
+
# x-default check
|
|
74
|
+
x_default_tag = tags.find { |t| t[:hreflang].downcase == 'x-default' }
|
|
75
|
+
|
|
76
|
+
# Graph Reciprocity Check (for remote URLs)
|
|
77
|
+
if @options[:check_reciprocity] != false && @url.match?(%r{^https?://})
|
|
78
|
+
verify_reciprocity(tags, @url)
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
evaluate_health(tags, self_ref_found, x_default_tag)
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
private
|
|
85
|
+
|
|
86
|
+
def load_target(target)
|
|
87
|
+
target_str = target.to_s.strip
|
|
88
|
+
if target_str.match?(%r{^https?://})
|
|
89
|
+
uri = URI.parse(target_str)
|
|
90
|
+
req = Net::HTTP::Get.new(uri)
|
|
91
|
+
req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) gsc-cli/2.1'
|
|
92
|
+
req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
|
|
93
|
+
|
|
94
|
+
res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 8, read_timeout: 10) do |http|
|
|
95
|
+
http.request(req)
|
|
96
|
+
end
|
|
97
|
+
[target_str, res.body.to_s.dup.force_encoding('UTF-8').scrub, res.to_hash]
|
|
98
|
+
elsif File.exist?(target_str)
|
|
99
|
+
[target_str, File.read(target_str, encoding: 'UTF-8'), {}]
|
|
100
|
+
else
|
|
101
|
+
# In-memory HTML string (e.g. test fixtures)
|
|
102
|
+
base = @options[:url] || 'https://example.com/'
|
|
103
|
+
[base, target_str, {}]
|
|
104
|
+
end
|
|
105
|
+
rescue StandardError => e
|
|
106
|
+
[target_str.to_s, "<html><head><!-- Error: #{e.message} --></head><body></body></html>", {}]
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
def extract_hreflang_tags(html_content, resp_headers, base_url)
|
|
110
|
+
tags = []
|
|
111
|
+
|
|
112
|
+
# 1. Extract from HTML <link ...>
|
|
113
|
+
html_content.scan(/<link\b([^>]*?)>/im) do |match|
|
|
114
|
+
attrs_str = match.first
|
|
115
|
+
rel = extract_attr(attrs_str, 'rel')&.downcase
|
|
116
|
+
next unless rel == 'alternate'
|
|
117
|
+
|
|
118
|
+
hreflang = extract_attr(attrs_str, 'hreflang')
|
|
119
|
+
href = extract_attr(attrs_str, 'href')
|
|
120
|
+
next if hreflang.nil? || href.nil? || href.empty?
|
|
121
|
+
|
|
122
|
+
resolved_href = resolve_url(href, base_url)
|
|
123
|
+
tags << {
|
|
124
|
+
source: :html,
|
|
125
|
+
hreflang: hreflang.strip,
|
|
126
|
+
href: resolved_href,
|
|
127
|
+
raw_href: href,
|
|
128
|
+
issues: [],
|
|
129
|
+
status: :pending
|
|
130
|
+
}
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
# 2. Extract from HTTP response headers (Link: <...>; rel="alternate"; hreflang="...")
|
|
134
|
+
link_headers = Array(resp_headers['link']) + Array(resp_headers['Link'])
|
|
135
|
+
link_headers.each do |lh|
|
|
136
|
+
lh.split(',').each do |part|
|
|
137
|
+
if part =~ /<([^>]+)>\s*;\s*rel\s*=\s*(?:["']?alternate["']?)/i
|
|
138
|
+
href = $1
|
|
139
|
+
hreflang = nil
|
|
140
|
+
if part =~ /hreflang\s*=\s*["']?([^;"'\s]+)["']?/i
|
|
141
|
+
hreflang = $1
|
|
142
|
+
end
|
|
143
|
+
if hreflang && href
|
|
144
|
+
resolved_href = resolve_url(href, base_url)
|
|
145
|
+
tags << {
|
|
146
|
+
source: :http_header,
|
|
147
|
+
hreflang: hreflang.strip,
|
|
148
|
+
href: resolved_href,
|
|
149
|
+
raw_href: href,
|
|
150
|
+
issues: [],
|
|
151
|
+
status: :pending
|
|
152
|
+
}
|
|
153
|
+
end
|
|
154
|
+
end
|
|
155
|
+
end
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
tags
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
def validate_tag_code(tag)
|
|
162
|
+
code = tag[:hreflang].strip
|
|
163
|
+
return if code.downcase == 'x-default'
|
|
164
|
+
|
|
165
|
+
# Check for uppercase language code error (e.g., "EN", "EN-us")
|
|
166
|
+
if code =~ /^[A-Z]{2}(?:-|$)/
|
|
167
|
+
tag[:issues] << :uppercase_language
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
# Check for known common misconceptions
|
|
171
|
+
lowered = code.downcase
|
|
172
|
+
if CODE_CORRECTIONS.key?(lowered)
|
|
173
|
+
tag[:issues] << :common_mistake
|
|
174
|
+
tag[:correction] = CODE_CORRECTIONS[lowered]
|
|
175
|
+
end
|
|
176
|
+
|
|
177
|
+
# Split by hyphen
|
|
178
|
+
parts = code.split('-')
|
|
179
|
+
lang = parts[0].downcase
|
|
180
|
+
|
|
181
|
+
unless ISO_639_1_LANGUAGES.include?(lang)
|
|
182
|
+
tag[:issues] << :invalid_language unless tag[:issues].include?(:common_mistake)
|
|
183
|
+
end
|
|
184
|
+
|
|
185
|
+
if parts.size == 2
|
|
186
|
+
subtag = parts[1]
|
|
187
|
+
if subtag.length == 4 # Script (e.g. Hans, Hant)
|
|
188
|
+
unless SCRIPT_CODES.include?(subtag.capitalize)
|
|
189
|
+
tag[:issues] << :invalid_script
|
|
190
|
+
end
|
|
191
|
+
elsif subtag.length == 2 # Region (e.g. US, GB)
|
|
192
|
+
region_upper = subtag.upcase
|
|
193
|
+
unless ISO_3166_1_REGIONS.include?(region_upper)
|
|
194
|
+
tag[:issues] << :invalid_region unless tag[:issues].include?(:common_mistake)
|
|
195
|
+
end
|
|
196
|
+
if subtag =~ /^[a-z]{2}$/
|
|
197
|
+
tag[:issues] << :lowercase_region
|
|
198
|
+
end
|
|
199
|
+
elsif subtag == '419' # UN M.49 code for Latin America & Caribbean (valid in Google)
|
|
200
|
+
# valid
|
|
201
|
+
else
|
|
202
|
+
tag[:issues] << :invalid_format
|
|
203
|
+
end
|
|
204
|
+
elsif parts.size > 2
|
|
205
|
+
# e.g., zh-Hans-CN
|
|
206
|
+
script = parts[1]
|
|
207
|
+
region = parts[2]
|
|
208
|
+
unless SCRIPT_CODES.include?(script.capitalize)
|
|
209
|
+
tag[:issues] << :invalid_script
|
|
210
|
+
end
|
|
211
|
+
unless ISO_3166_1_REGIONS.include?(region.upcase)
|
|
212
|
+
tag[:issues] << :invalid_region
|
|
213
|
+
end
|
|
214
|
+
end
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
def check_self_reference(tags, origin_url)
|
|
218
|
+
norm_origin = normalize_url(origin_url)
|
|
219
|
+
match = tags.find do |t|
|
|
220
|
+
normalize_url(t[:href]) == norm_origin && t[:hreflang].downcase != 'x-default'
|
|
221
|
+
end
|
|
222
|
+
|
|
223
|
+
!match.nil?
|
|
224
|
+
end
|
|
225
|
+
|
|
226
|
+
def verify_reciprocity(tags, origin_url)
|
|
227
|
+
norm_origin = normalize_url(origin_url)
|
|
228
|
+
|
|
229
|
+
tags.each do |tag|
|
|
230
|
+
norm_target = normalize_url(tag[:href])
|
|
231
|
+
if norm_target == norm_origin
|
|
232
|
+
tag[:status] = :self_referential
|
|
233
|
+
next
|
|
234
|
+
end
|
|
235
|
+
|
|
236
|
+
# Fetch the alternate page and check for reciprocal link
|
|
237
|
+
begin
|
|
238
|
+
uri = URI.parse(tag[:href])
|
|
239
|
+
req = Net::HTTP::Get.new(uri)
|
|
240
|
+
req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) gsc-cli/2.1'
|
|
241
|
+
|
|
242
|
+
res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 4, read_timeout: 5) do |http|
|
|
243
|
+
http.request(req)
|
|
244
|
+
end
|
|
245
|
+
|
|
246
|
+
if res.code.to_i >= 400
|
|
247
|
+
tag[:status] = :http_error
|
|
248
|
+
tag[:issues] << "http_#{res.code}".to_sym
|
|
249
|
+
next
|
|
250
|
+
end
|
|
251
|
+
|
|
252
|
+
dest_html = res.body.to_s.dup.force_encoding('UTF-8').scrub
|
|
253
|
+
dest_tags = extract_hreflang_tags(dest_html, res.to_hash, tag[:href])
|
|
254
|
+
|
|
255
|
+
# Check if dest_tags contains a link back to origin_url
|
|
256
|
+
return_tag = dest_tags.find { |dt| normalize_url(dt[:href]) == norm_origin }
|
|
257
|
+
|
|
258
|
+
if return_tag
|
|
259
|
+
tag[:status] = :confirmed
|
|
260
|
+
tag[:return_hreflang] = return_tag[:hreflang]
|
|
261
|
+
else
|
|
262
|
+
tag[:status] = :missing_return
|
|
263
|
+
tag[:issues] << :missing_reciprocal_tag
|
|
264
|
+
end
|
|
265
|
+
rescue StandardError => e
|
|
266
|
+
tag[:status] = :unreachable
|
|
267
|
+
tag[:issues] << :network_error
|
|
268
|
+
end
|
|
269
|
+
end
|
|
270
|
+
end
|
|
271
|
+
|
|
272
|
+
def evaluate_health(tags, self_ref_found, x_default_tag)
|
|
273
|
+
total = tags.size
|
|
274
|
+
|
|
275
|
+
if total.zero?
|
|
276
|
+
return {
|
|
277
|
+
url: @url,
|
|
278
|
+
total_tags: 0,
|
|
279
|
+
health_score: 50.0,
|
|
280
|
+
grade: 'C',
|
|
281
|
+
self_reference: false,
|
|
282
|
+
has_x_default: false,
|
|
283
|
+
reciprocal_count: 0,
|
|
284
|
+
unreciprocated_count: 0,
|
|
285
|
+
invalid_code_count: 0,
|
|
286
|
+
issues_summary: { missing_hreflang: 1 },
|
|
287
|
+
prescriptions: ['Add bidirectional hreflang alternate tags to properly localize international audience.'],
|
|
288
|
+
cluster_snippets: { html: '', sitemap_xml: '' },
|
|
289
|
+
tags: []
|
|
290
|
+
}
|
|
291
|
+
end
|
|
292
|
+
|
|
293
|
+
invalid_codes = tags.count { |t| (t[:issues] & %i[invalid_language invalid_region invalid_script common_mistake uppercase_language]).any? }
|
|
294
|
+
unreciprocated = tags.count { |t| t[:status] == :missing_return }
|
|
295
|
+
confirmed = tags.count { |t| t[:status] == :confirmed || t[:status] == :self_referential }
|
|
296
|
+
|
|
297
|
+
score = 100.0
|
|
298
|
+
score -= 25.0 unless self_ref_found
|
|
299
|
+
score -= 15.0 unless x_default_tag
|
|
300
|
+
score -= [invalid_codes * 15.0, 30.0].min
|
|
301
|
+
score -= [unreciprocated * 20.0, 40.0].min
|
|
302
|
+
|
|
303
|
+
score = [[score.round(1), 100.0].min, 0.0].max
|
|
304
|
+
grade = compute_grade(score)
|
|
305
|
+
|
|
306
|
+
summary = {
|
|
307
|
+
total_tags: total,
|
|
308
|
+
self_referencing: self_ref_found,
|
|
309
|
+
x_default: !x_default_tag.nil?,
|
|
310
|
+
confirmed_reciprocal: confirmed,
|
|
311
|
+
unreciprocated: unreciprocated,
|
|
312
|
+
invalid_codes: invalid_codes
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
prescriptions = generate_prescriptions(self_ref_found, x_default_tag, invalid_codes, unreciprocated, tags)
|
|
316
|
+
snippets = generate_cluster_snippets(tags, @url)
|
|
317
|
+
|
|
318
|
+
{
|
|
319
|
+
url: @url,
|
|
320
|
+
total_tags: total,
|
|
321
|
+
health_score: score,
|
|
322
|
+
grade: grade,
|
|
323
|
+
self_reference: self_ref_found,
|
|
324
|
+
has_x_default: !x_default_tag.nil?,
|
|
325
|
+
reciprocal_count: confirmed,
|
|
326
|
+
unreciprocated_count: unreciprocated,
|
|
327
|
+
invalid_code_count: invalid_codes,
|
|
328
|
+
issues_summary: summary,
|
|
329
|
+
prescriptions: prescriptions,
|
|
330
|
+
cluster_snippets: snippets,
|
|
331
|
+
tags: tags
|
|
332
|
+
}
|
|
333
|
+
end
|
|
334
|
+
|
|
335
|
+
def generate_prescriptions(self_ref, x_def, invalid_count, unrecip_count, tags)
|
|
336
|
+
recs = []
|
|
337
|
+
|
|
338
|
+
unless self_ref
|
|
339
|
+
recs << "Add a self-referencing hreflang tag pointing back to #{@url}. Per Google guidelines, every language variant must include its own URL."
|
|
340
|
+
end
|
|
341
|
+
|
|
342
|
+
unless x_def
|
|
343
|
+
recs << "Include an 'x-default' hreflang tag to handle international users whose language/region is not explicitly matched."
|
|
344
|
+
end
|
|
345
|
+
|
|
346
|
+
if invalid_count > 0
|
|
347
|
+
mistakes = tags.select { |t| t[:correction] }
|
|
348
|
+
if mistakes.any?
|
|
349
|
+
mistakes.each do |m|
|
|
350
|
+
recs << "Fix invalid hreflang code '#{m[:hreflang]}': #{m[:correction]}."
|
|
351
|
+
end
|
|
352
|
+
else
|
|
353
|
+
recs << "Correct #{invalid_count} invalid ISO 639-1 language or ISO 3166-1 region codes."
|
|
354
|
+
end
|
|
355
|
+
end
|
|
356
|
+
|
|
357
|
+
if unrecip_count > 0
|
|
358
|
+
recs << "Resolve #{unrecip_count} missing return tags. Hreflang links must be bidirectional; alternate pages must point back to origin."
|
|
359
|
+
end
|
|
360
|
+
|
|
361
|
+
recs
|
|
362
|
+
end
|
|
363
|
+
|
|
364
|
+
def generate_cluster_snippets(tags, base_url)
|
|
365
|
+
html_lines = []
|
|
366
|
+
xml_lines = []
|
|
367
|
+
|
|
368
|
+
tags.each do |t|
|
|
369
|
+
html_lines << "<link rel=\"alternate\" hreflang=\"#{t[:hreflang]}\" href=\"#{t[:href]}\" />"
|
|
370
|
+
xml_lines << " <xhtml:link rel=\"alternate\" hreflang=\"#{t[:hreflang]}\" href=\"#{t[:href]}\" />"
|
|
371
|
+
end
|
|
372
|
+
|
|
373
|
+
{
|
|
374
|
+
html: html_lines.join("\n"),
|
|
375
|
+
sitemap_xml: xml_lines.join("\n")
|
|
376
|
+
}
|
|
377
|
+
end
|
|
378
|
+
|
|
379
|
+
def compute_grade(score)
|
|
380
|
+
case score
|
|
381
|
+
when 90.0..100.0 then 'A+'
|
|
382
|
+
when 80.0...90.0 then 'A'
|
|
383
|
+
when 70.0...80.0 then 'B'
|
|
384
|
+
when 55.0...70.0 then 'C'
|
|
385
|
+
when 40.0...55.0 then 'D'
|
|
386
|
+
else 'F'
|
|
387
|
+
end
|
|
388
|
+
end
|
|
389
|
+
|
|
390
|
+
def extract_attr(tag_str, attr_name)
|
|
391
|
+
if tag_str =~ /\b#{attr_name}\s*=\s*(['"])(.*?)\1/i
|
|
392
|
+
$2.strip
|
|
393
|
+
elsif tag_str =~ /\b#{attr_name}\s*=\s*([^\s>]+)/i
|
|
394
|
+
$1.strip
|
|
395
|
+
else
|
|
396
|
+
nil
|
|
397
|
+
end
|
|
398
|
+
end
|
|
399
|
+
|
|
400
|
+
def resolve_url(href, base)
|
|
401
|
+
return href if href.match?(%r{^https?://})
|
|
402
|
+
URI.join(base, href).to_s
|
|
403
|
+
rescue StandardError
|
|
404
|
+
href
|
|
405
|
+
end
|
|
406
|
+
|
|
407
|
+
def normalize_url(url)
|
|
408
|
+
u = url.to_s.strip.downcase.sub(%r{/$}, '')
|
|
409
|
+
u.split('#').first
|
|
410
|
+
end
|
|
411
|
+
end
|
|
412
|
+
end
|