gsc-cli 2.1.0 → 2.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. checksums.yaml +4 -4
  2. data/AUTH.md +4 -1
  3. data/README.md +448 -408
  4. data/bin/gsc +28067 -5661
  5. data/dist/gsc +29121 -5046
  6. data/lib/gsc/aio_hunter.rb +343 -0
  7. data/lib/gsc/answer_synthesizer.rb +157 -0
  8. data/lib/gsc/api.rb +53 -1
  9. data/lib/gsc/auth.rb +26 -0
  10. data/lib/gsc/brand_segmenter.rb +140 -0
  11. data/lib/gsc/cache_manager.rb +806 -0
  12. data/lib/gsc/cannibalization_analyzer.rb +141 -0
  13. data/lib/gsc/canonical_chains.rb +367 -0
  14. data/lib/gsc/citation_simulator.rb +339 -0
  15. data/lib/gsc/cli/aio_hunter.rb +154 -0
  16. data/lib/gsc/cli/analytics.rb +788 -0
  17. data/lib/gsc/cli/audit.rb +1976 -0
  18. data/lib/gsc/cli/base.rb +384 -0
  19. data/lib/gsc/cli/cache.rb +266 -0
  20. data/lib/gsc/cli/canonical.rb +223 -0
  21. data/lib/gsc/cli/citation_simulator.rb +152 -0
  22. data/lib/gsc/cli/dashboard.rb +354 -0
  23. data/lib/gsc/cli/doctor.rb +129 -0
  24. data/lib/gsc/cli/eeat.rb +125 -0
  25. data/lib/gsc/cli/ga4.rb +852 -0
  26. data/lib/gsc/cli/growth.rb +650 -0
  27. data/lib/gsc/cli/hreflang.rb +164 -0
  28. data/lib/gsc/cli/image_seo.rb +162 -0
  29. data/lib/gsc/cli/indexing.rb +458 -0
  30. data/lib/gsc/cli/intent_shift.rb +125 -0
  31. data/lib/gsc/cli/keyword_value.rb +134 -0
  32. data/lib/gsc/cli/keywords.rb +795 -0
  33. data/lib/gsc/cli/landing_roi.rb +308 -0
  34. data/lib/gsc/cli/low_ctr.rb +213 -0
  35. data/lib/gsc/cli/mobile_parity.rb +150 -0
  36. data/lib/gsc/cli/report.rb +100 -0
  37. data/lib/gsc/cli/rich_results.rb +172 -0
  38. data/lib/gsc/cli/schema_generate.rb +149 -0
  39. data/lib/gsc/cli/seasonal.rb +232 -0
  40. data/lib/gsc/cli/security.rb +153 -0
  41. data/lib/gsc/cli/setup.rb +1291 -0
  42. data/lib/gsc/cli/sitemap_tree.rb +143 -0
  43. data/lib/gsc/cli/skill_pack.rb +62 -0
  44. data/lib/gsc/cli/soft_404.rb +199 -0
  45. data/lib/gsc/cli/sparkline.rb +227 -0
  46. data/lib/gsc/cli/watchdog.rb +150 -0
  47. data/lib/gsc/cli/zombie_purger.rb +208 -0
  48. data/lib/gsc/cli.rb +707 -5265
  49. data/lib/gsc/cli_advanced.rb +987 -44
  50. data/lib/gsc/client.rb +17 -2
  51. data/lib/gsc/color.rb +16 -1
  52. data/lib/gsc/command_registry.rb +47 -9
  53. data/lib/gsc/config.rb +2 -2
  54. data/lib/gsc/ctr_curve.rb +115 -0
  55. data/lib/gsc/decay_predictor.rb +322 -0
  56. data/lib/gsc/doctor.rb +434 -0
  57. data/lib/gsc/eeat_auditor.rb +428 -0
  58. data/lib/gsc/entity_auditor.rb +229 -0
  59. data/lib/gsc/firewall_scanner.rb +733 -0
  60. data/lib/gsc/geo_auditor.rb +368 -0
  61. data/lib/gsc/google_trends.rb +8 -1
  62. data/lib/gsc/heading_validator.rb +283 -0
  63. data/lib/gsc/hreflang_validator.rb +412 -0
  64. data/lib/gsc/image_seo.rb +286 -0
  65. data/lib/gsc/indexing_queue.rb +179 -0
  66. data/lib/gsc/indexnow.rb +93 -0
  67. data/lib/gsc/intent_shift.rb +188 -0
  68. data/lib/gsc/internal_links.rb +153 -36
  69. data/lib/gsc/keyword_value.rb +191 -0
  70. data/lib/gsc/landing_roi.rb +195 -0
  71. data/lib/gsc/llms_generator.rb +343 -22
  72. data/lib/gsc/low_ctr_rewriter.rb +408 -0
  73. data/lib/gsc/mobile_parity.rb +222 -0
  74. data/lib/gsc/network_tracer.rb +8 -1
  75. data/lib/gsc/page_analyzer.rb +47 -7
  76. data/lib/gsc/prompts.rb +38 -29
  77. data/lib/gsc/questions_harvester.rb +178 -0
  78. data/lib/gsc/report_generator.rb +461 -0
  79. data/lib/gsc/rich_results.rb +388 -0
  80. data/lib/gsc/robots_checker.rb +46 -15
  81. data/lib/gsc/schema_generator.rb +788 -0
  82. data/lib/gsc/schema_validator.rb +36 -38
  83. data/lib/gsc/seasonal_predictor.rb +381 -0
  84. data/lib/gsc/security_scanner.rb +496 -0
  85. data/lib/gsc/serp_feature_detector.rb +359 -0
  86. data/lib/gsc/serp_preview.rb +108 -22
  87. data/lib/gsc/site_crawler.rb +113 -21
  88. data/lib/gsc/sitemap_loader.rb +15 -4
  89. data/lib/gsc/sitemap_tree.rb +301 -0
  90. data/lib/gsc/skill_pack.rb +195 -0
  91. data/lib/gsc/soft_404_analyzer.rb +385 -0
  92. data/lib/gsc/sparkline.rb +171 -0
  93. data/lib/gsc/speed_correlator.rb +416 -0
  94. data/lib/gsc/striking_playbook.rb +190 -0
  95. data/lib/gsc/title_optimizer.rb +420 -0
  96. data/lib/gsc/vault.rb +260 -0
  97. data/lib/gsc/version.rb +1 -1
  98. data/lib/gsc/watchdog.rb +235 -0
  99. data/lib/gsc/zombie_purger.rb +366 -0
  100. data/lib/gsc.rb +118 -0
  101. metadata +75 -1
@@ -0,0 +1,368 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'net/http'
5
+ require 'uri'
6
+ require 'json'
7
+ require 'zlib'
8
+ require 'stringio'
9
+
10
+ module GSC
11
+ class GeoAuditor
12
+ AI_BOTS = {
13
+ 'GPTBot' => { provider: 'OpenAI (ChatGPT Search)', critical: true },
14
+ 'ClaudeBot' => { provider: 'Anthropic (Claude AI)', critical: true },
15
+ 'PerplexityBot' => { provider: 'Perplexity AI Search', critical: true },
16
+ 'Google-Extended' => { provider: 'Google (Gemini & AI Overviews)', critical: true },
17
+ 'Applebot-Extended' => { provider: 'Apple Intelligence', critical: false },
18
+ 'CCBot' => { provider: 'Common Crawl (Open LLM Datasets)', critical: false }
19
+ }.freeze
20
+
21
+ QUESTION_HEADING_REGEX = /\b(what|why|how|can|does|is|are|which|best|vs|versus|difference|guide|review|pricing|steps|definition)\b/i
22
+
23
+ attr_reader :url, :html, :http_status, :headers, :response_time_ms
24
+
25
+ def initialize(url_or_domain)
26
+ raw = url_or_domain.to_s.strip
27
+ raw = "https://#{raw}" unless raw =~ %r{^https?://}
28
+ @url = raw
29
+ @uri = URI.parse(@url)
30
+ end
31
+
32
+ def audit
33
+ fetch_page
34
+ return { error: true, message: "HTTP #{@http_status} fetching #{@url}" } unless @http_status == 200 && !@html.empty?
35
+
36
+ ai_crawler_status = audit_ai_crawlers
37
+ answer_density = audit_direct_answers
38
+ fact_density = audit_factual_citability
39
+ entity_authority = audit_entity_schema
40
+ temporal_freshness = audit_freshness
41
+
42
+ # Scoring (0-100)
43
+ crawler_score = calculate_crawler_score(ai_crawler_status)
44
+ answer_score = calculate_answer_score(answer_density)
45
+ fact_score = calculate_fact_score(fact_density)
46
+ entity_score = calculate_entity_score(entity_authority)
47
+
48
+ total_score = crawler_score + answer_score + fact_score + entity_score
49
+
50
+ recommendations = generate_recommendations(
51
+ ai_crawler_status,
52
+ answer_density,
53
+ fact_density,
54
+ entity_authority,
55
+ temporal_freshness
56
+ )
57
+
58
+ {
59
+ url: @url,
60
+ score: total_score,
61
+ grade: score_grade(total_score),
62
+ category_scores: {
63
+ ai_bot_access: { score: crawler_score, max: 25 },
64
+ direct_answers: { score: answer_score, max: 25 },
65
+ facts_and_data: { score: fact_score, max: 25 },
66
+ entity_schema: { score: entity_score, max: 25 }
67
+ },
68
+ ai_crawlers: ai_crawler_status,
69
+ direct_answers: answer_density,
70
+ facts_and_citability: fact_density,
71
+ entity_knowledge_graph: entity_authority,
72
+ freshness: temporal_freshness,
73
+ recommendations: recommendations
74
+ }
75
+ end
76
+
77
+ private
78
+
79
+ def fetch_page
80
+ start_t = Time.now
81
+ http = Net::HTTP.new(@uri.host, @uri.port)
82
+ http.use_ssl = (@uri.scheme == 'https')
83
+ http.open_timeout = 8
84
+ http.read_timeout = 15
85
+
86
+ req = Net::HTTP::Get.new(@uri.request_uri.empty? ? '/' : @uri.request_uri)
87
+ req['User-Agent'] = "Mozilla/5.0 (compatible; GSC-GEO-Auditor/#{GSC::VERSION}; +https://apolloswave.com)"
88
+ req['Accept-Encoding'] = 'gzip'
89
+
90
+ res = http.request(req)
91
+ @response_time_ms = ((Time.now - start_t) * 1000).round(1)
92
+ @http_status = res.code.to_i
93
+ @headers = res.to_hash
94
+
95
+ raw_body = res.body || ''
96
+ body_str = if res['content-encoding'] =~ /gzip/i && !raw_body.empty?
97
+ begin
98
+ Zlib::GzipReader.new(StringIO.new(raw_body)).read
99
+ rescue StandardError
100
+ raw_body
101
+ end
102
+ else
103
+ raw_body
104
+ end
105
+ @html = body_str.to_s.dup.force_encoding('UTF-8').scrub
106
+ rescue StandardError => e
107
+ @http_status = 0
108
+ @html = ''
109
+ @error_msg = e.message
110
+ end
111
+
112
+ def audit_ai_crawlers
113
+ robots_url = "#{@uri.scheme}://#{@uri.host}:#{@uri.port}/robots.txt"
114
+ robots_txt = ''
115
+ begin
116
+ res = Net::HTTP.get_response(URI.parse(robots_url))
117
+ robots_txt = res.body.to_s.dup.force_encoding('UTF-8').scrub if res.code == '200'
118
+ rescue StandardError
119
+ robots_txt = ''
120
+ end
121
+
122
+ # Check robots meta in page
123
+ robots_meta = @html[/<meta\s+[^>]*name=['"]robots['"][^>]*content=['"]([^'"]+)['"]/i, 1] || ''
124
+ noindex = robots_meta =~ /noindex/i
125
+ nosnippet = robots_meta =~ /nosnippet/i
126
+
127
+ checker = RobotsChecker.new(@url)
128
+ bot_results = {}
129
+
130
+ AI_BOTS.each do |bot_name, meta|
131
+ rule = checker.check(@uri.path, bot_name)
132
+ allowed = rule[:allowed] && !noindex
133
+ bot_results[bot_name] = {
134
+ provider: meta[:provider],
135
+ allowed: allowed,
136
+ critical: meta[:critical],
137
+ rule: rule[:matched_rule] ? "#{rule[:matched_rule][:type]}: #{rule[:matched_rule][:path]}" : 'Default (Allowed)'
138
+ }
139
+ end
140
+
141
+ {
142
+ robots_txt_found: !robots_txt.empty?,
143
+ meta_noindex: !!noindex,
144
+ meta_nosnippet: !!nosnippet,
145
+ bots: bot_results,
146
+ all_critical_allowed: bot_results.select { |_, v| v[:critical] }.all? { |_, v| v[:allowed] }
147
+ }
148
+ end
149
+
150
+ def audit_direct_answers
151
+ # Find headings that look like user questions
152
+ headings = []
153
+ @html.scan(/<(h[23])[^>]*>(.*?)<\/\1>/im) do |tag, text|
154
+ clean = text.gsub(/<[^>]+>/, '').strip
155
+ next if clean.empty?
156
+
157
+ is_question = (clean =~ QUESTION_HEADING_REGEX) || clean.end_with?('?')
158
+ headings << { tag: tag, text: clean, is_question: !!is_question }
159
+ end
160
+
161
+ # Analyze paragraphs following question headings
162
+ answer_blocks = []
163
+ headings.select { |h| h[:is_question] }.each do |h|
164
+ pattern = /<#{h[:tag]}[^>]*>#{Regexp.escape(h[:text])}<\/#{h[:tag]}>\s*<p[^>]*>(.*?)<\/p>/im
165
+ match = @html[pattern, 1]
166
+ if match
167
+ ans_text = match.gsub(/<[^>]+>/, '').strip
168
+ words = ans_text.split(/\s+/).size
169
+ # Ideal direct answer for LLM citation is 30-80 words
170
+ optimal = words >= 30 && words <= 80
171
+ answer_blocks << {
172
+ question: h[:text],
173
+ answer_preview: ans_text[0..120] + (ans_text.length > 120 ? '...' : ''),
174
+ word_count: words,
175
+ optimal_length: optimal
176
+ }
177
+ end
178
+ end
179
+
180
+ {
181
+ total_question_headings: headings.count { |h| h[:is_question] },
182
+ direct_answer_blocks_found: answer_blocks.size,
183
+ optimal_answers_count: answer_blocks.count { |a| a[:optimal_length] },
184
+ samples: answer_blocks.first(3)
185
+ }
186
+ end
187
+
188
+ def audit_factual_citability
189
+ clean = @html.gsub(/<script\b[^>]*>.*?<\/script>/im, ' ')
190
+ .gsub(/<style\b[^>]*>.*?<\/style>/im, ' ')
191
+
192
+ # Count statistical data points
193
+ percentages = clean.scan(/\b\d+(?:\.\d+)?%/).size
194
+ currency_points = clean.scan(/(?:\$|€|£)\s*\d+(?:,\d{3})*(?:\.\d+)?/).size
195
+ numbers = clean.scan(/\b\d+(?:,\d{3})+(?:\.\d+)?\b/).size
196
+ recent_years = clean.scan(/\b(202[4-6])\b/).size
197
+
198
+ # Count structured list items
199
+ list_items = clean.scan(/<li\b[^>]*>(.*?)<\/li>/im).size
200
+
201
+ # Count table rows
202
+ table_rows = clean.scan(/<tr\b[^>]*>/im).size
203
+
204
+ {
205
+ percentage_mentions: percentages,
206
+ currency_mentions: currency_points,
207
+ large_numbers: numbers,
208
+ recent_year_mentions: recent_years,
209
+ bullet_list_items: list_items,
210
+ comparison_table_rows: table_rows,
211
+ high_fact_density: (percentages + currency_points + numbers >= 5) || (list_items >= 6)
212
+ }
213
+ end
214
+
215
+ def audit_entity_schema
216
+ schemas = []
217
+ @html.scan(/<script\s+[^>]*type=['"]application\/ld\+json['"][^>]*>(.*?)<\/script>/im) do |m|
218
+ content = m.first.to_s.strip
219
+ begin
220
+ parsed = JSON.parse(content)
221
+ schemas.concat(parsed.is_a?(Array) ? parsed : [parsed])
222
+ rescue StandardError
223
+ nil
224
+ end
225
+ end
226
+
227
+ types = schemas.map { |s| s['@type'] }.compact.flatten
228
+ same_as = []
229
+ author_info = nil
230
+ brand_info = nil
231
+
232
+ schemas.each do |s|
233
+ same_as.concat(Array(s['sameAs'])) if s['sameAs']
234
+ author_info ||= s['author'] if s['author']
235
+ brand_info ||= (s['brand'] || s['name']) if %w[Organization Brand WebSite Product].include?(s['@type'])
236
+ end
237
+
238
+ same_as_domains = same_as.map do |link|
239
+ begin
240
+ URI.parse(link.to_s).host
241
+ rescue StandardError
242
+ link.to_s
243
+ end
244
+ end.compact.uniq
245
+
246
+ has_wiki = same_as_domains.any? { |d| d =~ /wikipedia|wikidata/i }
247
+ has_social_entity = same_as_domains.any? { |d| d =~ /linkedin|twitter|x\.com|youtube|github/i }
248
+
249
+ {
250
+ schema_count: schemas.size,
251
+ schema_types: types.uniq,
252
+ has_organization_or_brand: types.any? { |t| %w[Organization Brand WebSite].include?(t) },
253
+ has_faq_or_howto: types.any? { |t| %w[FAQPage HowTo QAPage].include?(t) },
254
+ has_article_schema: types.any? { |t| %w[Article NewsArticle BlogPosting TechArticle].include?(t) },
255
+ has_author: !author_info.nil?,
256
+ same_as_links_count: same_as.size,
257
+ same_as_domains: same_as_domains,
258
+ wikidata_or_wikipedia_linked: has_wiki,
259
+ social_entity_linked: has_social_entity
260
+ }
261
+ end
262
+
263
+ def audit_freshness
264
+ published = @html[/<meta\s+[^>]*property=['"]article:published_time['"][^>]*content=['"]([^'"]+)['"]/i, 1] ||
265
+ @html[/<meta\s+[^>]*name=['"]pubdate['"][^>]*content=['"]([^'"]+)['"]/i, 1]
266
+ modified = @html[/<meta\s+[^>]*property=['"]article:modified_time['"][^>]*content=['"]([^'"]+)['"]/i, 1] ||
267
+ @html[/<meta\s+[^>]*name=['"]last-modified['"][^>]*content=['"]([^'"]+)['"]/i, 1]
268
+
269
+ {
270
+ published_date: published,
271
+ modified_date: modified,
272
+ has_dates: !published.nil? || !modified.nil?
273
+ }
274
+ end
275
+
276
+ def calculate_crawler_score(ai_crawlers)
277
+ score = 0
278
+ critical_bots = ai_crawlers[:bots].select { |_, b| b[:critical] }
279
+ allowed_count = critical_bots.count { |_, b| b[:allowed] }
280
+
281
+ score += (allowed_count.to_f / [critical_bots.size, 1].max * 20).round
282
+ score += 5 unless ai_crawlers[:meta_nosnippet] || ai_crawlers[:meta_noindex]
283
+ score
284
+ end
285
+
286
+ def calculate_answer_score(answers)
287
+ score = 0
288
+ score += 8 if answers[:total_question_headings] >= 2
289
+ score += 10 if answers[:direct_answer_blocks_found] >= 2
290
+ score += 7 if answers[:optimal_answers_count] >= 1
291
+ [score, 25].min
292
+ end
293
+
294
+ def calculate_fact_score(facts)
295
+ score = 0
296
+ score += 8 if (facts[:percentage_mentions] + facts[:currency_mentions] + facts[:large_numbers]) >= 3
297
+ score += 9 if facts[:bullet_list_items] >= 5
298
+ score += 5 if facts[:recent_year_mentions] >= 1
299
+ score += 3 if facts[:comparison_table_rows] >= 2
300
+ [score, 25].min
301
+ end
302
+
303
+ def calculate_entity_score(entity)
304
+ score = 0
305
+ score += 7 if entity[:has_organization_or_brand]
306
+ score += 7 if entity[:has_faq_or_howto] || entity[:has_article_schema]
307
+ score += 6 if entity[:same_as_links_count] >= 1
308
+ score += 5 if entity[:wikidata_or_wikipedia_linked] || entity[:social_entity_linked]
309
+ [score, 25].min
310
+ end
311
+
312
+ def score_grade(score)
313
+ case score
314
+ when 85..100 then 'A (Exceptional AI Search Citability)'
315
+ when 70..84 then 'B (Good - Minor Entity & Direct Answer Gaps)'
316
+ when 50..69 then 'C (Moderate - Missing Key AI Bot Access or Schema)'
317
+ else 'D/F (Poor - High Risk of Being Ignored by LLMs)'
318
+ end
319
+ end
320
+
321
+ def generate_recommendations(crawlers, answers, facts, entity, freshness)
322
+ recs = []
323
+
324
+ # AI Crawlers
325
+ blocked = crawlers[:bots].select { |_, b| !b[:allowed] }
326
+ if blocked.any?
327
+ names = blocked.keys.join(', ')
328
+ recs << "Unblock #{names} in /robots.txt to permit ChatGPT, Claude, and Perplexity from quoting this URL."
329
+ end
330
+ if crawlers[:meta_nosnippet]
331
+ recs << "Remove 'nosnippet' directive from robots meta tag; LLMs require snippet extraction to cite your content."
332
+ end
333
+
334
+ # Direct Answers
335
+ if answers[:direct_answer_blocks_found] == 0
336
+ recs << "Add 2+ question headings (H2/H3 'What is...', 'How to...') followed immediately by a concise 40-60 word definition paragraph."
337
+ elsif answers[:optimal_answers_count] == 0
338
+ recs << "Tighten paragraph lengths following question headings to 40-70 words. Overly long prose reduces LLM snippet selection."
339
+ end
340
+
341
+ # Factual Citability
342
+ if !facts[:high_fact_density]
343
+ recs << "Inject concrete statistics (percentages, metrics, pricing) and bulleted key takeaways; LLMs favor citing verified data points over generic prose."
344
+ end
345
+ if facts[:bullet_list_items] < 4
346
+ recs << "Add structured bulleted summaries (<ul>/<li>) beneath major section headers for easy machine extraction."
347
+ end
348
+
349
+ # Entity Schema
350
+ if !entity[:has_organization_or_brand]
351
+ recs << "Add JSON-LD 'Organization' or 'Brand' structured data with official entity name and logo."
352
+ end
353
+ if entity[:same_as_links_count] == 0
354
+ recs << "Add 'sameAs' links inside your Organization JSON-LD pointing to Wikipedia, Wikidata, LinkedIn, or Twitter/X to disambiguate your brand in LLM Knowledge Graphs."
355
+ end
356
+ if !entity[:has_faq_or_howto]
357
+ recs << "Implement FAQPage JSON-LD schema wrapping your common questions and answers."
358
+ end
359
+
360
+ # Freshness
361
+ unless freshness[:has_dates]
362
+ recs << "Include visible publication and modification dates (and 'datePublished'/'dateModified' in schema) to signal content freshness to AI answer engines."
363
+ end
364
+
365
+ recs
366
+ end
367
+ end
368
+ end
@@ -63,7 +63,14 @@ class GoogleTrends
63
63
  explore_req['Cookie'] = cookies if cookies
64
64
 
65
65
  explore_res = http.request(explore_req)
66
- return { ok: false, error: "Google Trends Explore API error (HTTP #{explore_res.code})" } unless explore_res.is_a?(Net::HTTPSuccess)
66
+ if explore_res.code == '302'
67
+ loc = explore_res['location'].to_s
68
+ return { ok: false, error: "Google Trends rate limit or bot challenge triggered (HTTP 302 redirect to #{loc.include?('sorry') ? 'Google CAPTCHA' : 'redirect'}). Google is temporarily throttling Trends requests from this network. Try again in a few minutes, or use 'gsc suggest' / 'gsc planner'." }
69
+ elsif explore_res.code == '429' || init_res.code == '429'
70
+ return { ok: false, error: "Google Trends rate limit reached (HTTP 429 - Too Many Requests). Google is temporarily throttling Trends requests from this network. Try again in a few minutes, or use 'gsc suggest' / 'gsc planner'." }
71
+ elsif !explore_res.is_a?(Net::HTTPSuccess)
72
+ return { ok: false, error: "Google Trends Explore API error (HTTP #{explore_res.code})" }
73
+ end
67
74
 
68
75
  raw_exp = explore_res.body.to_s
69
76
  clean_body = raw_exp.index('{') ? raw_exp[raw_exp.index('{')..] : raw_exp
@@ -0,0 +1,283 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'json'
4
+
5
+ module GSC
6
+ class HeadingValidator
7
+ attr_reader :headings, :counts, :violations, :score, :grade, :depth
8
+
9
+ def self.analyze(html, target_keywords: nil)
10
+ new(html, target_keywords: target_keywords).audit
11
+ end
12
+
13
+ def initialize(html, target_keywords: nil)
14
+ @html = html.to_s.dup.force_encoding('UTF-8').scrub
15
+ @target_keywords = extract_keyword_tokens(target_keywords)
16
+ @headings = []
17
+ @counts = { 'h1' => 0, 'h2' => 0, 'h3' => 0, 'h4' => 0, 'h5' => 0, 'h6' => 0 }
18
+ @violations = []
19
+ @score = 100
20
+ @grade = 'A'
21
+ @depth = 0
22
+ end
23
+
24
+ def audit
25
+ parse_headings!
26
+ validate_hierarchy!
27
+ calculate_score!
28
+ audit_keywords! if @target_keywords && !@target_keywords.empty?
29
+
30
+ {
31
+ score: @score,
32
+ grade: @grade,
33
+ depth: @depth,
34
+ count: @headings.size,
35
+ counts: @counts,
36
+ h1_count: @counts['h1'],
37
+ empty_count: @headings.count { |h| h[:empty] },
38
+ headings: @headings,
39
+ violations: @violations,
40
+ keyword_analysis: @keyword_analysis,
41
+ summary: summary_text,
42
+ ascii_tree: render_ascii_tree(colorize: false),
43
+ color_tree: render_ascii_tree(colorize: true)
44
+ }
45
+ end
46
+
47
+ def render_ascii_tree(colorize: false)
48
+ return ' (No headings found on page)' if @headings.empty?
49
+
50
+ lines = []
51
+ @headings.each_with_index do |h, idx|
52
+ level = h[:level]
53
+ tag_str = "[#{h[:tag].upcase}]"
54
+ prefix = ' ' + (' ' * (level - 1)) + (level == 1 ? '■ ' : '├── ')
55
+
56
+ text_str = h[:empty] ? '(Empty Heading Tag)' : h[:text]
57
+ # Truncate very long heading in display
58
+ display_text = text_str.length > 75 ? "#{text_str[0..72]}..." : text_str
59
+
60
+ # Check if this heading has an attached violation
61
+ v_msgs = h[:violations].map { |v| v[:message] }
62
+ v_suffix = v_msgs.empty? ? '' : " ⚠️ #{v_msgs.join('; ')}"
63
+
64
+ if colorize && defined?(Color)
65
+ tag_color = case level
66
+ when 1 then Color::CYAN
67
+ when 2 then Color::BLUE
68
+ when 3 then Color::MAGENTA
69
+ else Color::GRAY
70
+ end
71
+ colored_tag = Color.c(tag_str, tag_color, Color::BOLD)
72
+ colored_text = h[:empty] ? Color.c(display_text, Color::RED, Color::DIM) : display_text
73
+ colored_suffix = v_suffix.empty? ? '' : Color.c(v_suffix, Color::YELLOW, Color::BOLD)
74
+ lines << "#{prefix}#{colored_tag} #{colored_text}#{colored_suffix}"
75
+ else
76
+ lines << "#{prefix}#{tag_str} #{display_text}#{v_suffix}"
77
+ end
78
+ end
79
+
80
+ lines.join("\n")
81
+ end
82
+
83
+ private
84
+
85
+ def parse_headings!
86
+ idx = 0
87
+ @html.scan(%r{<(h[1-6])(?:\s+[^>]*)?>(.*?)</\1>}im) do |tag, content|
88
+ idx += 1
89
+ clean = clean_text(content)
90
+ level = tag[1].to_i
91
+ is_empty = clean.empty?
92
+
93
+ h_info = {
94
+ index: idx,
95
+ tag: tag.downcase,
96
+ level: level,
97
+ text: clean,
98
+ length: clean.length,
99
+ empty: is_empty,
100
+ violations: []
101
+ }
102
+
103
+ @headings << h_info
104
+ @counts[tag.downcase] += 1
105
+ @depth = level if level > @depth
106
+ end
107
+ end
108
+
109
+ def validate_hierarchy!
110
+ return if @headings.empty?
111
+
112
+ # 1. Check if first heading is H1
113
+ first_h = @headings.first
114
+ if first_h[:level] != 1
115
+ v = {
116
+ type: :first_not_h1,
117
+ severity: :warning,
118
+ tag: first_h[:tag],
119
+ index: 1,
120
+ message: "Page starts with <#{first_h[:tag]}> instead of <h1>"
121
+ }
122
+ @violations << v
123
+ first_h[:violations] << v
124
+ end
125
+
126
+ # 2. Check H1 count
127
+ if @counts['h1'] == 0
128
+ @violations << {
129
+ type: :missing_h1,
130
+ severity: :critical,
131
+ message: 'Missing <h1> tag. Document has 0 top-level headings.'
132
+ }
133
+ elsif @counts['h1'] > 1
134
+ extra_h1s = @headings.select { |h| h[:tag] == 'h1' }[1..]
135
+ extra_h1s.each do |eh|
136
+ v = {
137
+ type: :multiple_h1,
138
+ severity: :warning,
139
+ tag: 'h1',
140
+ index: eh[:index],
141
+ message: "Multiple <h1> tag detected at position ##{eh[:index]} ('#{eh[:text][0..40]}...')"
142
+ }
143
+ @violations << v
144
+ eh[:violations] << v
145
+ end
146
+ end
147
+
148
+ # 3. Check sequential depth skips (e.g. H1 -> H3 or H2 -> H4)
149
+ prev_level = nil
150
+ @headings.each do |h|
151
+ curr_level = h[:level]
152
+
153
+ if prev_level && curr_level > prev_level + 1
154
+ skipped = (prev_level + 1...curr_level).map { |lvl| "<h#{lvl}>" }.join(', ')
155
+ v = {
156
+ type: :skipped_level,
157
+ severity: :warning,
158
+ tag: h[:tag],
159
+ index: h[:index],
160
+ from_level: prev_level,
161
+ to_level: curr_level,
162
+ message: "Skipped heading level: jumped from <h#{prev_level}> to <h#{curr_level}> (skipped #{skipped})"
163
+ }
164
+ @violations << v
165
+ h[:violations] << v
166
+ end
167
+
168
+ # 4. Check empty heading
169
+ if h[:empty]
170
+ v = {
171
+ type: :empty_heading,
172
+ severity: :error,
173
+ tag: h[:tag],
174
+ index: h[:index],
175
+ message: "Empty <#{h[:tag]}> tag with no text content"
176
+ }
177
+ @violations << v
178
+ h[:violations] << v
179
+ end
180
+
181
+ # 5. Check overlong heading (> 80 chars)
182
+ if h[:length] > 80
183
+ v = {
184
+ type: :overlong_heading,
185
+ severity: :info,
186
+ tag: h[:tag],
187
+ index: h[:index],
188
+ message: "Heading length (#{h[:length]} chars) exceeds 80 chars; risk of topical dilution"
189
+ }
190
+ @violations << v
191
+ h[:violations] << v
192
+ end
193
+
194
+ prev_level = curr_level
195
+ end
196
+ end
197
+
198
+ def calculate_score!
199
+ if @headings.empty?
200
+ @score = 0
201
+ @grade = 'F'
202
+ return
203
+ end
204
+
205
+ deductions = 0
206
+ @violations.each do |v|
207
+ deductions += case v[:type]
208
+ when :missing_h1 then 30
209
+ when :multiple_h1 then 15
210
+ when :skipped_level then 10
211
+ when :empty_heading then 10
212
+ when :first_not_h1 then 10
213
+ when :overlong_heading then 5
214
+ else 5
215
+ end
216
+ end
217
+
218
+ @score = [100 - deductions, 0].max
219
+ @grade = case @score
220
+ when 90..100 then 'A'
221
+ when 80..89 then 'B'
222
+ when 70..79 then 'C'
223
+ when 60..69 then 'D'
224
+ else 'F'
225
+ end
226
+ end
227
+
228
+ def audit_keywords!
229
+ h1 = @headings.find { |h| h[:tag] == 'h1' }
230
+ h1_matches = []
231
+ if h1
232
+ h1_text = h1[:text].downcase
233
+ h1_matches = @target_keywords.select { |kw| h1_text.include?(kw) }
234
+ end
235
+
236
+ h2_matches = {}
237
+ @headings.select { |h| h[:tag] == 'h2' }.each do |h2|
238
+ h2_text = h2[:text].downcase
239
+ matched = @target_keywords.select { |kw| h2_text.include?(kw) }
240
+ h2_matches[h2[:text]] = matched unless matched.empty?
241
+ end
242
+
243
+ @keyword_analysis = {
244
+ target_keywords: @target_keywords,
245
+ h1_has_keyword: !h1_matches.empty?,
246
+ h1_matched_keywords: h1_matches,
247
+ h2_matched_count: h2_matches.size,
248
+ h2_matches: h2_matches
249
+ }
250
+ end
251
+
252
+ def extract_keyword_tokens(keywords)
253
+ return [] if keywords.nil? || keywords.to_s.strip.empty?
254
+
255
+ if keywords.is_a?(Array)
256
+ keywords.map { |k| k.to_s.downcase.strip }.reject(&:empty?)
257
+ else
258
+ keywords.to_s.downcase.split(/[,|\s]+/).map(&:strip).reject { |w| w.length < 3 }
259
+ end
260
+ end
261
+
262
+ def clean_text(str)
263
+ str.to_s
264
+ .dup
265
+ .force_encoding('UTF-8')
266
+ .scrub
267
+ .gsub(/<[^>]+>/, ' ')
268
+ .gsub(/&amp;/, '&')
269
+ .gsub(/&lt;/, '<')
270
+ .gsub(/&gt;/, '>')
271
+ .gsub(/&quot;/, '"')
272
+ .gsub(/&#39;/, "'")
273
+ .gsub(/&nbsp;/, ' ')
274
+ .gsub(/\s+/, ' ')
275
+ .strip
276
+ end
277
+
278
+ def summary_text
279
+ counts_str = @counts.select { |_k, v| v > 0 }.map { |k, v| "#{v}x #{k.upcase}" }.join(', ')
280
+ "Heading Health Score: #{@score}/100 (Grade #{@grade}) | Depth: H#{@depth} | Total: #{@headings.size} (#{counts_str}) | #{@violations.size} violations"
281
+ end
282
+ end
283
+ end