gsc-cli 2.1.0 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +410 -414
  3. data/bin/gsc +28067 -5661
  4. data/dist/gsc +29121 -5046
  5. data/lib/gsc/aio_hunter.rb +343 -0
  6. data/lib/gsc/answer_synthesizer.rb +157 -0
  7. data/lib/gsc/api.rb +53 -1
  8. data/lib/gsc/auth.rb +26 -0
  9. data/lib/gsc/brand_segmenter.rb +140 -0
  10. data/lib/gsc/cache_manager.rb +806 -0
  11. data/lib/gsc/cannibalization_analyzer.rb +141 -0
  12. data/lib/gsc/canonical_chains.rb +367 -0
  13. data/lib/gsc/citation_simulator.rb +339 -0
  14. data/lib/gsc/cli/aio_hunter.rb +154 -0
  15. data/lib/gsc/cli/analytics.rb +788 -0
  16. data/lib/gsc/cli/audit.rb +1976 -0
  17. data/lib/gsc/cli/base.rb +384 -0
  18. data/lib/gsc/cli/cache.rb +266 -0
  19. data/lib/gsc/cli/canonical.rb +223 -0
  20. data/lib/gsc/cli/citation_simulator.rb +152 -0
  21. data/lib/gsc/cli/dashboard.rb +354 -0
  22. data/lib/gsc/cli/doctor.rb +129 -0
  23. data/lib/gsc/cli/eeat.rb +125 -0
  24. data/lib/gsc/cli/ga4.rb +852 -0
  25. data/lib/gsc/cli/growth.rb +650 -0
  26. data/lib/gsc/cli/hreflang.rb +164 -0
  27. data/lib/gsc/cli/image_seo.rb +162 -0
  28. data/lib/gsc/cli/indexing.rb +458 -0
  29. data/lib/gsc/cli/intent_shift.rb +125 -0
  30. data/lib/gsc/cli/keyword_value.rb +134 -0
  31. data/lib/gsc/cli/keywords.rb +795 -0
  32. data/lib/gsc/cli/landing_roi.rb +308 -0
  33. data/lib/gsc/cli/low_ctr.rb +213 -0
  34. data/lib/gsc/cli/mobile_parity.rb +150 -0
  35. data/lib/gsc/cli/report.rb +100 -0
  36. data/lib/gsc/cli/rich_results.rb +172 -0
  37. data/lib/gsc/cli/schema_generate.rb +149 -0
  38. data/lib/gsc/cli/seasonal.rb +232 -0
  39. data/lib/gsc/cli/security.rb +153 -0
  40. data/lib/gsc/cli/setup.rb +1291 -0
  41. data/lib/gsc/cli/sitemap_tree.rb +143 -0
  42. data/lib/gsc/cli/skill_pack.rb +62 -0
  43. data/lib/gsc/cli/soft_404.rb +199 -0
  44. data/lib/gsc/cli/sparkline.rb +227 -0
  45. data/lib/gsc/cli/watchdog.rb +150 -0
  46. data/lib/gsc/cli/zombie_purger.rb +208 -0
  47. data/lib/gsc/cli.rb +707 -5265
  48. data/lib/gsc/cli_advanced.rb +987 -44
  49. data/lib/gsc/client.rb +17 -2
  50. data/lib/gsc/color.rb +16 -1
  51. data/lib/gsc/command_registry.rb +47 -9
  52. data/lib/gsc/config.rb +2 -2
  53. data/lib/gsc/ctr_curve.rb +115 -0
  54. data/lib/gsc/decay_predictor.rb +322 -0
  55. data/lib/gsc/doctor.rb +434 -0
  56. data/lib/gsc/eeat_auditor.rb +428 -0
  57. data/lib/gsc/entity_auditor.rb +229 -0
  58. data/lib/gsc/firewall_scanner.rb +733 -0
  59. data/lib/gsc/geo_auditor.rb +368 -0
  60. data/lib/gsc/google_trends.rb +8 -1
  61. data/lib/gsc/heading_validator.rb +283 -0
  62. data/lib/gsc/hreflang_validator.rb +412 -0
  63. data/lib/gsc/image_seo.rb +286 -0
  64. data/lib/gsc/indexing_queue.rb +179 -0
  65. data/lib/gsc/indexnow.rb +93 -0
  66. data/lib/gsc/intent_shift.rb +188 -0
  67. data/lib/gsc/internal_links.rb +153 -36
  68. data/lib/gsc/keyword_value.rb +191 -0
  69. data/lib/gsc/landing_roi.rb +195 -0
  70. data/lib/gsc/llms_generator.rb +343 -22
  71. data/lib/gsc/low_ctr_rewriter.rb +408 -0
  72. data/lib/gsc/mobile_parity.rb +222 -0
  73. data/lib/gsc/network_tracer.rb +8 -1
  74. data/lib/gsc/page_analyzer.rb +47 -7
  75. data/lib/gsc/prompts.rb +38 -29
  76. data/lib/gsc/questions_harvester.rb +178 -0
  77. data/lib/gsc/report_generator.rb +461 -0
  78. data/lib/gsc/rich_results.rb +388 -0
  79. data/lib/gsc/robots_checker.rb +46 -15
  80. data/lib/gsc/schema_generator.rb +788 -0
  81. data/lib/gsc/schema_validator.rb +36 -38
  82. data/lib/gsc/seasonal_predictor.rb +381 -0
  83. data/lib/gsc/security_scanner.rb +496 -0
  84. data/lib/gsc/serp_feature_detector.rb +359 -0
  85. data/lib/gsc/serp_preview.rb +108 -22
  86. data/lib/gsc/site_crawler.rb +113 -21
  87. data/lib/gsc/sitemap_loader.rb +15 -4
  88. data/lib/gsc/sitemap_tree.rb +301 -0
  89. data/lib/gsc/skill_pack.rb +195 -0
  90. data/lib/gsc/soft_404_analyzer.rb +385 -0
  91. data/lib/gsc/sparkline.rb +171 -0
  92. data/lib/gsc/speed_correlator.rb +416 -0
  93. data/lib/gsc/striking_playbook.rb +190 -0
  94. data/lib/gsc/title_optimizer.rb +420 -0
  95. data/lib/gsc/vault.rb +260 -0
  96. data/lib/gsc/version.rb +1 -1
  97. data/lib/gsc/watchdog.rb +235 -0
  98. data/lib/gsc/zombie_purger.rb +366 -0
  99. data/lib/gsc.rb +118 -0
  100. metadata +75 -1
@@ -0,0 +1,412 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'net/http'
5
+ require 'uri'
6
+ require 'json'
7
+ require 'set'
8
+
9
+ module GSC
10
+ class HreflangValidator
11
+ # Common ISO 639-1 two-letter language codes
12
+ ISO_639_1_LANGUAGES = Set.new(%w[
13
+ aa ab ae af ak am an ar as av ay az ba be bg bh bi bm bn bo br bs ca ce ch co cr cs cu cv cy da de dv dz ee el
14
+ en eo es et eu fa ff fi fj fo fr fy ga gd gl gn gu gv ha he hi ho hr ht hu hy hz ia id ie ig ii ik io is it iu
15
+ ja jv ka kg ki kj kk kl km kn ko kr ks ku kv kw ky la lb lg li ln lo lt lu lv mg mh mi mk ml mn mr ms mt my na
16
+ nb nd ne ng nl nn no nr nv ny oc oj om or os pa pi pl ps pt qu rm rn ro ru rw sa sc sd se sg si sk sl sm sn so
17
+ sq sr ss st su sv sw ta te tg th ti tk tl tn to tr ts tt tw ty ug uk ur uz ve vi vo wa wo xh yi yo za zh zu
18
+ ]).freeze
19
+
20
+ # Common script codes (4 letters, Titlecase)
21
+ SCRIPT_CODES = Set.new(%w[
22
+ Hans Hant Cyrl Latn Arab Deva
23
+ ]).freeze
24
+
25
+ # ISO 3166-1 alpha-2 two-letter uppercase country/region codes
26
+ ISO_3166_1_REGIONS = Set.new(%w[
27
+ AD AE AF AG AI AL AM AO AQ AR AS AT AU AW AX AZ BA BB BD BE BF BG BH BI BJ BL BM BN BO BQ BR BS BT BV BW BY BZ
28
+ CA CC CD CF CG CH CI CK CL CM CN CO CR CU CV CW CX CY CZ DE DJ DK DM DO DZ EC EE EG EH ER ES ET FI FJ FK FM FO
29
+ FR GA GB GD GE GF GG GH GI GL GM GN GP GQ GR GS GT GU GW GY HK HM HN HR HT HU ID IE IL IM IN IO IQ IR IS IT JE
30
+ JM JO JP KE KG KH KI KM KN KP KR KW KY KZ LA LB LC LI LK LR LS LT LU LV LY MA MC MD ME MF MG MH MK ML MM MN MO
31
+ MP MQ MR MS MT MU MV MW MX MY MZ NA NC NE NF NG NI NL NO NP NR NU NZ OM PA PE PF PG PH PK PL PM PN PR PS PT PW
32
+ PY QA RE RO RS RU RW SA SB SC SD SE SG SH SI SJ SK SL SM SN SO SR SS ST SV SX SY SZ TC TD TF TG TH TJ TK TL TM
33
+ TN TO TR TT TV TW TZ UA UG UM US UY UZ VA VC VE VG VI VN VU WF WS YE YT ZA ZM ZW
34
+ ]).freeze
35
+
36
+ # Known common mistakes & corrections
37
+ CODE_CORRECTIONS = {
38
+ 'en-uk' => 'en-GB (The ISO 3166-1 region code for the UK is "GB", not "UK")',
39
+ 'uk' => 'uk is Ukrainian; for United Kingdom English, use "en-GB"',
40
+ 'jp' => 'jp is country code; for Japanese language, use "ja" or "ja-JP"',
41
+ 'kr' => 'kr is country code; for Korean language, use "ko" or "ko-KR"',
42
+ 'sp' => 'sp is invalid; for Spanish, use "es"',
43
+ 'ge' => 'ge is Georgia; for German, use "de"',
44
+ 'cz' => 'cz is Czech Republic country code; for Czech language, use "cs"',
45
+ 'se' => 'se is Northern Sami; for Swedish, use "sv" or "sv-SE"',
46
+ 'dk' => 'dk is Denmark country code; for Danish language, use "da" or "da-DK"',
47
+ 'gr' => 'gr is Greece country code; for Greek language, use "el" or "el-GR"',
48
+ 'es-la' => 'es-LA is invalid ("LA" is Laos, not Latin America; use "es-419" or country-specific codes)',
49
+ 'es-sa' => 'es-SA is invalid ("SA" is Saudi Arabia; use country-specific codes like es-AR, es-CL)',
50
+ 'cn' => 'cn is country code; for Chinese, use "zh", "zh-Hans", or "zh-Hant"'
51
+ }.freeze
52
+
53
+ attr_reader :options, :url, :html, :headers
54
+
55
+ def initialize(options = {})
56
+ @options = options
57
+ end
58
+
59
+ def self.audit(target, options = {})
60
+ new(options).audit(target)
61
+ end
62
+
63
+ def audit(target)
64
+ @url, @html, @headers = load_target(target)
65
+ tags = extract_hreflang_tags(@html, @headers, @url)
66
+
67
+ # Validate language/region codes
68
+ tags.each { |tag| validate_tag_code(tag) }
69
+
70
+ # Self-referencing check
71
+ self_ref_found = check_self_reference(tags, @url)
72
+
73
+ # x-default check
74
+ x_default_tag = tags.find { |t| t[:hreflang].downcase == 'x-default' }
75
+
76
+ # Graph Reciprocity Check (for remote URLs)
77
+ if @options[:check_reciprocity] != false && @url.match?(%r{^https?://})
78
+ verify_reciprocity(tags, @url)
79
+ end
80
+
81
+ evaluate_health(tags, self_ref_found, x_default_tag)
82
+ end
83
+
84
+ private
85
+
86
+ def load_target(target)
87
+ target_str = target.to_s.strip
88
+ if target_str.match?(%r{^https?://})
89
+ uri = URI.parse(target_str)
90
+ req = Net::HTTP::Get.new(uri)
91
+ req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) gsc-cli/2.1'
92
+ req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
93
+
94
+ res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 8, read_timeout: 10) do |http|
95
+ http.request(req)
96
+ end
97
+ [target_str, res.body.to_s.dup.force_encoding('UTF-8').scrub, res.to_hash]
98
+ elsif File.exist?(target_str)
99
+ [target_str, File.read(target_str, encoding: 'UTF-8'), {}]
100
+ else
101
+ # In-memory HTML string (e.g. test fixtures)
102
+ base = @options[:url] || 'https://example.com/'
103
+ [base, target_str, {}]
104
+ end
105
+ rescue StandardError => e
106
+ [target_str.to_s, "<html><head><!-- Error: #{e.message} --></head><body></body></html>", {}]
107
+ end
108
+
109
+ def extract_hreflang_tags(html_content, resp_headers, base_url)
110
+ tags = []
111
+
112
+ # 1. Extract from HTML <link ...>
113
+ html_content.scan(/<link\b([^>]*?)>/im) do |match|
114
+ attrs_str = match.first
115
+ rel = extract_attr(attrs_str, 'rel')&.downcase
116
+ next unless rel == 'alternate'
117
+
118
+ hreflang = extract_attr(attrs_str, 'hreflang')
119
+ href = extract_attr(attrs_str, 'href')
120
+ next if hreflang.nil? || href.nil? || href.empty?
121
+
122
+ resolved_href = resolve_url(href, base_url)
123
+ tags << {
124
+ source: :html,
125
+ hreflang: hreflang.strip,
126
+ href: resolved_href,
127
+ raw_href: href,
128
+ issues: [],
129
+ status: :pending
130
+ }
131
+ end
132
+
133
+ # 2. Extract from HTTP response headers (Link: <...>; rel="alternate"; hreflang="...")
134
+ link_headers = Array(resp_headers['link']) + Array(resp_headers['Link'])
135
+ link_headers.each do |lh|
136
+ lh.split(',').each do |part|
137
+ if part =~ /<([^>]+)>\s*;\s*rel\s*=\s*(?:["']?alternate["']?)/i
138
+ href = $1
139
+ hreflang = nil
140
+ if part =~ /hreflang\s*=\s*["']?([^;"'\s]+)["']?/i
141
+ hreflang = $1
142
+ end
143
+ if hreflang && href
144
+ resolved_href = resolve_url(href, base_url)
145
+ tags << {
146
+ source: :http_header,
147
+ hreflang: hreflang.strip,
148
+ href: resolved_href,
149
+ raw_href: href,
150
+ issues: [],
151
+ status: :pending
152
+ }
153
+ end
154
+ end
155
+ end
156
+ end
157
+
158
+ tags
159
+ end
160
+
161
+ def validate_tag_code(tag)
162
+ code = tag[:hreflang].strip
163
+ return if code.downcase == 'x-default'
164
+
165
+ # Check for uppercase language code error (e.g., "EN", "EN-us")
166
+ if code =~ /^[A-Z]{2}(?:-|$)/
167
+ tag[:issues] << :uppercase_language
168
+ end
169
+
170
+ # Check for known common misconceptions
171
+ lowered = code.downcase
172
+ if CODE_CORRECTIONS.key?(lowered)
173
+ tag[:issues] << :common_mistake
174
+ tag[:correction] = CODE_CORRECTIONS[lowered]
175
+ end
176
+
177
+ # Split by hyphen
178
+ parts = code.split('-')
179
+ lang = parts[0].downcase
180
+
181
+ unless ISO_639_1_LANGUAGES.include?(lang)
182
+ tag[:issues] << :invalid_language unless tag[:issues].include?(:common_mistake)
183
+ end
184
+
185
+ if parts.size == 2
186
+ subtag = parts[1]
187
+ if subtag.length == 4 # Script (e.g. Hans, Hant)
188
+ unless SCRIPT_CODES.include?(subtag.capitalize)
189
+ tag[:issues] << :invalid_script
190
+ end
191
+ elsif subtag.length == 2 # Region (e.g. US, GB)
192
+ region_upper = subtag.upcase
193
+ unless ISO_3166_1_REGIONS.include?(region_upper)
194
+ tag[:issues] << :invalid_region unless tag[:issues].include?(:common_mistake)
195
+ end
196
+ if subtag =~ /^[a-z]{2}$/
197
+ tag[:issues] << :lowercase_region
198
+ end
199
+ elsif subtag == '419' # UN M.49 code for Latin America & Caribbean (valid in Google)
200
+ # valid
201
+ else
202
+ tag[:issues] << :invalid_format
203
+ end
204
+ elsif parts.size > 2
205
+ # e.g., zh-Hans-CN
206
+ script = parts[1]
207
+ region = parts[2]
208
+ unless SCRIPT_CODES.include?(script.capitalize)
209
+ tag[:issues] << :invalid_script
210
+ end
211
+ unless ISO_3166_1_REGIONS.include?(region.upcase)
212
+ tag[:issues] << :invalid_region
213
+ end
214
+ end
215
+ end
216
+
217
+ def check_self_reference(tags, origin_url)
218
+ norm_origin = normalize_url(origin_url)
219
+ match = tags.find do |t|
220
+ normalize_url(t[:href]) == norm_origin && t[:hreflang].downcase != 'x-default'
221
+ end
222
+
223
+ !match.nil?
224
+ end
225
+
226
+ def verify_reciprocity(tags, origin_url)
227
+ norm_origin = normalize_url(origin_url)
228
+
229
+ tags.each do |tag|
230
+ norm_target = normalize_url(tag[:href])
231
+ if norm_target == norm_origin
232
+ tag[:status] = :self_referential
233
+ next
234
+ end
235
+
236
+ # Fetch the alternate page and check for reciprocal link
237
+ begin
238
+ uri = URI.parse(tag[:href])
239
+ req = Net::HTTP::Get.new(uri)
240
+ req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) gsc-cli/2.1'
241
+
242
+ res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 4, read_timeout: 5) do |http|
243
+ http.request(req)
244
+ end
245
+
246
+ if res.code.to_i >= 400
247
+ tag[:status] = :http_error
248
+ tag[:issues] << "http_#{res.code}".to_sym
249
+ next
250
+ end
251
+
252
+ dest_html = res.body.to_s.dup.force_encoding('UTF-8').scrub
253
+ dest_tags = extract_hreflang_tags(dest_html, res.to_hash, tag[:href])
254
+
255
+ # Check if dest_tags contains a link back to origin_url
256
+ return_tag = dest_tags.find { |dt| normalize_url(dt[:href]) == norm_origin }
257
+
258
+ if return_tag
259
+ tag[:status] = :confirmed
260
+ tag[:return_hreflang] = return_tag[:hreflang]
261
+ else
262
+ tag[:status] = :missing_return
263
+ tag[:issues] << :missing_reciprocal_tag
264
+ end
265
+ rescue StandardError => e
266
+ tag[:status] = :unreachable
267
+ tag[:issues] << :network_error
268
+ end
269
+ end
270
+ end
271
+
272
+ def evaluate_health(tags, self_ref_found, x_default_tag)
273
+ total = tags.size
274
+
275
+ if total.zero?
276
+ return {
277
+ url: @url,
278
+ total_tags: 0,
279
+ health_score: 50.0,
280
+ grade: 'C',
281
+ self_reference: false,
282
+ has_x_default: false,
283
+ reciprocal_count: 0,
284
+ unreciprocated_count: 0,
285
+ invalid_code_count: 0,
286
+ issues_summary: { missing_hreflang: 1 },
287
+ prescriptions: ['Add bidirectional hreflang alternate tags to properly localize international audience.'],
288
+ cluster_snippets: { html: '', sitemap_xml: '' },
289
+ tags: []
290
+ }
291
+ end
292
+
293
+ invalid_codes = tags.count { |t| (t[:issues] & %i[invalid_language invalid_region invalid_script common_mistake uppercase_language]).any? }
294
+ unreciprocated = tags.count { |t| t[:status] == :missing_return }
295
+ confirmed = tags.count { |t| t[:status] == :confirmed || t[:status] == :self_referential }
296
+
297
+ score = 100.0
298
+ score -= 25.0 unless self_ref_found
299
+ score -= 15.0 unless x_default_tag
300
+ score -= [invalid_codes * 15.0, 30.0].min
301
+ score -= [unreciprocated * 20.0, 40.0].min
302
+
303
+ score = [[score.round(1), 100.0].min, 0.0].max
304
+ grade = compute_grade(score)
305
+
306
+ summary = {
307
+ total_tags: total,
308
+ self_referencing: self_ref_found,
309
+ x_default: !x_default_tag.nil?,
310
+ confirmed_reciprocal: confirmed,
311
+ unreciprocated: unreciprocated,
312
+ invalid_codes: invalid_codes
313
+ }
314
+
315
+ prescriptions = generate_prescriptions(self_ref_found, x_default_tag, invalid_codes, unreciprocated, tags)
316
+ snippets = generate_cluster_snippets(tags, @url)
317
+
318
+ {
319
+ url: @url,
320
+ total_tags: total,
321
+ health_score: score,
322
+ grade: grade,
323
+ self_reference: self_ref_found,
324
+ has_x_default: !x_default_tag.nil?,
325
+ reciprocal_count: confirmed,
326
+ unreciprocated_count: unreciprocated,
327
+ invalid_code_count: invalid_codes,
328
+ issues_summary: summary,
329
+ prescriptions: prescriptions,
330
+ cluster_snippets: snippets,
331
+ tags: tags
332
+ }
333
+ end
334
+
335
+ def generate_prescriptions(self_ref, x_def, invalid_count, unrecip_count, tags)
336
+ recs = []
337
+
338
+ unless self_ref
339
+ recs << "Add a self-referencing hreflang tag pointing back to #{@url}. Per Google guidelines, every language variant must include its own URL."
340
+ end
341
+
342
+ unless x_def
343
+ recs << "Include an 'x-default' hreflang tag to handle international users whose language/region is not explicitly matched."
344
+ end
345
+
346
+ if invalid_count > 0
347
+ mistakes = tags.select { |t| t[:correction] }
348
+ if mistakes.any?
349
+ mistakes.each do |m|
350
+ recs << "Fix invalid hreflang code '#{m[:hreflang]}': #{m[:correction]}."
351
+ end
352
+ else
353
+ recs << "Correct #{invalid_count} invalid ISO 639-1 language or ISO 3166-1 region codes."
354
+ end
355
+ end
356
+
357
+ if unrecip_count > 0
358
+ recs << "Resolve #{unrecip_count} missing return tags. Hreflang links must be bidirectional; alternate pages must point back to origin."
359
+ end
360
+
361
+ recs
362
+ end
363
+
364
+ def generate_cluster_snippets(tags, base_url)
365
+ html_lines = []
366
+ xml_lines = []
367
+
368
+ tags.each do |t|
369
+ html_lines << "<link rel=\"alternate\" hreflang=\"#{t[:hreflang]}\" href=\"#{t[:href]}\" />"
370
+ xml_lines << " <xhtml:link rel=\"alternate\" hreflang=\"#{t[:hreflang]}\" href=\"#{t[:href]}\" />"
371
+ end
372
+
373
+ {
374
+ html: html_lines.join("\n"),
375
+ sitemap_xml: xml_lines.join("\n")
376
+ }
377
+ end
378
+
379
+ def compute_grade(score)
380
+ case score
381
+ when 90.0..100.0 then 'A+'
382
+ when 80.0...90.0 then 'A'
383
+ when 70.0...80.0 then 'B'
384
+ when 55.0...70.0 then 'C'
385
+ when 40.0...55.0 then 'D'
386
+ else 'F'
387
+ end
388
+ end
389
+
390
+ def extract_attr(tag_str, attr_name)
391
+ if tag_str =~ /\b#{attr_name}\s*=\s*(['"])(.*?)\1/i
392
+ $2.strip
393
+ elsif tag_str =~ /\b#{attr_name}\s*=\s*([^\s>]+)/i
394
+ $1.strip
395
+ else
396
+ nil
397
+ end
398
+ end
399
+
400
+ def resolve_url(href, base)
401
+ return href if href.match?(%r{^https?://})
402
+ URI.join(base, href).to_s
403
+ rescue StandardError
404
+ href
405
+ end
406
+
407
+ def normalize_url(url)
408
+ u = url.to_s.strip.downcase.sub(%r{/$}, '')
409
+ u.split('#').first
410
+ end
411
+ end
412
+ end
@@ -0,0 +1,286 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'net/http'
5
+ require 'uri'
6
+ require 'json'
7
+
8
+ module GSC
9
+ class ImageSeo
10
+ MODERN_FORMATS = %w[webp avif svg].freeze
11
+ LEGACY_FORMATS = %w[png jpg jpeg gif bmp webp_fallback].freeze
12
+
13
+ attr_reader :options, :url, :html
14
+
15
+ def initialize(options = {})
16
+ @options = options
17
+ end
18
+
19
+ def self.audit(target, options = {})
20
+ new(options).audit(target)
21
+ end
22
+
23
+ def audit(target)
24
+ @url, @html = load_content(target)
25
+ images = extract_images(@html, @url)
26
+
27
+ # Optionally check file size via HTTP HEAD
28
+ if @options[:check_size]
29
+ audit_image_sizes(images)
30
+ end
31
+
32
+ evaluate_health(images)
33
+ end
34
+
35
+ private
36
+
37
+ def load_content(target)
38
+ target_str = target.to_s.strip
39
+ if target_str.match?(%r{^https?://})
40
+ uri = URI.parse(target_str)
41
+ req = Net::HTTP::Get.new(uri)
42
+ req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36'
43
+ req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
44
+
45
+ res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 8, read_timeout: 10) do |http|
46
+ http.request(req)
47
+ end
48
+ [target_str, res.body.to_s.dup.force_encoding('UTF-8').scrub]
49
+ elsif File.exist?(target_str)
50
+ [target_str, File.read(target_str, encoding: 'UTF-8')]
51
+ else
52
+ [target_str.start_with?('http') ? target_str : 'local-document', target_str]
53
+ end
54
+ rescue StandardError => e
55
+ [target_str.to_s, "<html><body><!-- Error: #{e.message} --></body></html>"]
56
+ end
57
+
58
+ def extract_images(html_str, base_url)
59
+ images = []
60
+ index = 0
61
+
62
+ # Match all <img> tags
63
+ html_str.scan(/<img\b([^>]*?)>/im) do |match|
64
+ tag_attrs = match.first
65
+ src = extract_attr(tag_attrs, 'src')
66
+ next if src.nil? || src.empty?
67
+
68
+ alt = extract_attr(tag_attrs, 'alt')
69
+ width = extract_attr(tag_attrs, 'width')
70
+ height = extract_attr(tag_attrs, 'height')
71
+ loading = extract_attr(tag_attrs, 'loading')&.downcase
72
+ fetchpriority = extract_attr(tag_attrs, 'fetchpriority')&.downcase
73
+ decoding = extract_attr(tag_attrs, 'decoding')&.downcase
74
+ role = extract_attr(tag_attrs, 'role')&.downcase
75
+ aria_hidden = extract_attr(tag_attrs, 'aria-hidden')&.downcase
76
+
77
+ is_decorative = (role == 'presentation' || role == 'none' || aria_hidden == 'true')
78
+ resolved_src = resolve_url(src, base_url)
79
+ ext = extract_extension(resolved_src)
80
+
81
+ is_hero = (index == 0)
82
+
83
+ issues = []
84
+ issues << :missing_alt if alt.nil? && !is_decorative
85
+ issues << :empty_alt if alt == '' && !is_decorative
86
+ issues << :long_alt if alt && alt.length > 125
87
+ issues << :legacy_format if LEGACY_FORMATS.include?(ext)
88
+ issues << :missing_dimensions if (width.nil? || height.nil?)
89
+ issues << :lcp_lazy_loaded if is_hero && loading == 'lazy'
90
+ issues << :hero_missing_priority if is_hero && fetchpriority != 'high'
91
+ issues << :missing_lazy if !is_hero && loading != 'lazy'
92
+
93
+ images << {
94
+ index: index + 1,
95
+ src: resolved_src,
96
+ raw_src: src,
97
+ alt: alt,
98
+ width: width ? width.to_i : nil,
99
+ height: height ? height.to_i : nil,
100
+ format: ext,
101
+ loading: loading,
102
+ fetchpriority: fetchpriority,
103
+ decoding: decoding,
104
+ is_decorative: is_decorative,
105
+ is_hero: is_hero,
106
+ issues: issues,
107
+ issue_count: issues.size,
108
+ picture_tag_snippet: build_picture_snippet(resolved_src, alt, width, height, is_hero),
109
+ nextjs_snippet: build_nextjs_snippet(resolved_src, alt, width, height, is_hero)
110
+ }
111
+ index += 1
112
+ end
113
+
114
+ images
115
+ end
116
+
117
+ def audit_image_sizes(images)
118
+ images.each do |img|
119
+ next unless img[:src].match?(%r{^https?://})
120
+ begin
121
+ uri = URI.parse(img[:src])
122
+ res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 4, read_timeout: 4) do |http|
123
+ http.head(uri.request_uri)
124
+ end
125
+ if res['content-length']
126
+ bytes = res['content-length'].to_i
127
+ img[:bytes] = bytes
128
+ img[:size_kb] = (bytes / 1024.0).round(1)
129
+ img[:issues] << :oversized_payload if bytes > 200 * 1024
130
+ img[:issues] << :critical_payload if bytes > 500 * 1024
131
+ end
132
+ rescue StandardError
133
+ img[:bytes] = nil
134
+ end
135
+ end
136
+ end
137
+
138
+ def evaluate_health(images)
139
+ total = images.size
140
+ if total.zero?
141
+ return {
142
+ url: @url,
143
+ total_images: 0,
144
+ health_score: 100.0,
145
+ grade: 'A+',
146
+ metrics: { alt_coverage_pct: 100.0, dimension_coverage_pct: 100.0, modern_format_pct: 100.0 },
147
+ issues_summary: {},
148
+ images: []
149
+ }
150
+ end
151
+
152
+ valid_alt_count = images.count { |img| !img[:issues].include?(:missing_alt) && !img[:issues].include?(:empty_alt) }
153
+ valid_dims_count = images.count { |img| !img[:issues].include?(:missing_dimensions) }
154
+ modern_format_count = images.count { |img| MODERN_FORMATS.include?(img[:format]) }
155
+ proper_lazy_count = images.count { |img| (!img[:is_hero] && img[:loading] == 'lazy') || (img[:is_hero] && img[:loading] != 'lazy') }
156
+
157
+ alt_coverage = ((valid_alt_count.to_f / total) * 100.0).round(1)
158
+ dim_coverage = ((valid_dims_count.to_f / total) * 100.0).round(1)
159
+ format_coverage = ((modern_format_count.to_f / total) * 100.0).round(1)
160
+ lazy_coverage = ((proper_lazy_count.to_f / total) * 100.0).round(1)
161
+
162
+ # Deductions
163
+ score = 100.0
164
+ score -= (100.0 - alt_coverage) * 0.35 # Alt text is critical (up to 35 pts)
165
+ score -= (100.0 - dim_coverage) * 0.25 # CLS dimensions (up to 25 pts)
166
+ score -= (100.0 - format_coverage) * 0.25 # Next-gen formats (up to 25 pts)
167
+ score -= (100.0 - lazy_coverage) * 0.15 # Lazy loading & LCP priority (up to 15 pts)
168
+
169
+ score = [[score.round(1), 100.0].min, 0.0].max
170
+ grade = compute_grade(score)
171
+
172
+ summary = {
173
+ missing_alt: images.count { |i| i[:issues].include?(:missing_alt) },
174
+ empty_alt: images.count { |i| i[:issues].include?(:empty_alt) },
175
+ missing_dimensions: images.count { |i| i[:issues].include?(:missing_dimensions) },
176
+ legacy_format: images.count { |i| i[:issues].include?(:legacy_format) },
177
+ lcp_lazy_loaded: images.count { |i| i[:issues].include?(:lcp_lazy_loaded) },
178
+ hero_missing_priority: images.count { |i| i[:issues].include?(:hero_missing_priority) },
179
+ oversized: images.count { |i| i[:issues].include?(:oversized_payload) }
180
+ }
181
+
182
+ prescriptions = generate_prescriptions(summary, images)
183
+
184
+ limit = (@options[:limit] || 20).to_i
185
+ displayed = images.first(limit)
186
+
187
+ {
188
+ url: @url,
189
+ total_images: total,
190
+ health_score: score,
191
+ grade: grade,
192
+ metrics: {
193
+ alt_coverage_pct: alt_coverage,
194
+ dimension_coverage_pct: dim_coverage,
195
+ modern_format_pct: format_coverage,
196
+ lazy_loading_pct: lazy_coverage
197
+ },
198
+ issues_summary: summary,
199
+ prescriptions: prescriptions,
200
+ images: displayed
201
+ }
202
+ end
203
+
204
+ def generate_prescriptions(sum, images)
205
+ recs = []
206
+
207
+ if sum[:missing_alt] > 0
208
+ recs << "Add descriptive, keyword-rich alt text to #{sum[:missing_alt]} images missing the alt attribute."
209
+ end
210
+
211
+ if sum[:missing_dimensions] > 0
212
+ recs << "Specify explicit width and height attributes on #{sum[:missing_dimensions]} images to eliminate Cumulative Layout Shift (CLS)."
213
+ end
214
+
215
+ if sum[:legacy_format] > 0
216
+ recs << "Convert #{sum[:legacy_format]} PNG/JPEG images to modern WebP or AVIF formats for 65–80% byte reduction."
217
+ end
218
+
219
+ if sum[:lcp_lazy_loaded] > 0
220
+ recs << "Remove loading=\"lazy\" from the primary above-the-fold hero image to accelerate Largest Contentful Paint (LCP)."
221
+ end
222
+
223
+ if sum[:hero_missing_priority] > 0
224
+ recs << "Add fetchpriority=\"high\" to the primary hero image to trigger immediate preloading by the browser."
225
+ end
226
+
227
+ recs
228
+ end
229
+
230
+ def build_picture_snippet(src, alt, width, height, is_hero)
231
+ base = src.sub(/\.[^.]+$/, '')
232
+ w_attr = width ? " width=\"#{width}\"" : ""
233
+ h_attr = height ? " height=\"#{height}\"" : ""
234
+ loading_attr = is_hero ? " fetchpriority=\"high\"" : " loading=\"lazy\" decoding=\"async\""
235
+
236
+ <<~HTML.strip
237
+ <picture>
238
+ <source srcset="#{base}.avif" type="image/avif">
239
+ <source srcset="#{base}.webp" type="image/webp">
240
+ <img src="#{src}" alt="#{alt || 'Descriptive image text'}"#{w_attr}#{h_attr}#{loading_attr}>
241
+ </picture>
242
+ HTML
243
+ end
244
+
245
+ def build_nextjs_snippet(src, alt, width, height, is_hero)
246
+ w = width || 800
247
+ h = height || 600
248
+ priority_attr = is_hero ? " priority" : ""
249
+
250
+ "<Image src=\"#{src}\" alt=\"#{alt || 'Descriptive image text'}\" width={#{w}} height={#{h}}#{priority_attr} />"
251
+ end
252
+
253
+ def compute_grade(score)
254
+ case score
255
+ when 90.0..100.0 then 'A+'
256
+ when 80.0...90.0 then 'A'
257
+ when 70.0...80.0 then 'B'
258
+ when 55.0...70.0 then 'C'
259
+ when 40.0...55.0 then 'D'
260
+ else 'F'
261
+ end
262
+ end
263
+
264
+ def extract_attr(tag_str, attr_name)
265
+ if tag_str =~ /\b#{attr_name}\s*=\s*(['"])(.*?)\1/i
266
+ $2.strip
267
+ elsif tag_str =~ /\b#{attr_name}\s*=\s*([^\s>]+)/i
268
+ $1.strip
269
+ else
270
+ nil
271
+ end
272
+ end
273
+
274
+ def resolve_url(src, base)
275
+ return src if src.match?(%r{^https?://})
276
+ URI.join(base, src).to_s
277
+ rescue StandardError
278
+ src
279
+ end
280
+
281
+ def extract_extension(url_str)
282
+ clean = url_str.split('?').first.split('#').first
283
+ File.extname(clean).sub(/^\./, '').downcase
284
+ end
285
+ end
286
+ end