gsc-cli 2.0.2 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. checksums.yaml +4 -4
  2. data/AUTH.md +205 -0
  3. data/FUNDING.md +120 -0
  4. data/README.md +463 -299
  5. data/bin/gsc +29158 -4921
  6. data/dist/gsc +29158 -4921
  7. data/lib/gsc/aio_hunter.rb +343 -0
  8. data/lib/gsc/answer_synthesizer.rb +157 -0
  9. data/lib/gsc/api.rb +53 -1
  10. data/lib/gsc/auth.rb +26 -0
  11. data/lib/gsc/backlinks_manager.rb +96 -0
  12. data/lib/gsc/brand_segmenter.rb +140 -0
  13. data/lib/gsc/cache_manager.rb +806 -0
  14. data/lib/gsc/cannibalization_analyzer.rb +141 -0
  15. data/lib/gsc/canonical_chains.rb +367 -0
  16. data/lib/gsc/citation_simulator.rb +339 -0
  17. data/lib/gsc/cli/aio_hunter.rb +154 -0
  18. data/lib/gsc/cli/analytics.rb +788 -0
  19. data/lib/gsc/cli/audit.rb +1976 -0
  20. data/lib/gsc/cli/base.rb +384 -0
  21. data/lib/gsc/cli/cache.rb +266 -0
  22. data/lib/gsc/cli/canonical.rb +223 -0
  23. data/lib/gsc/cli/citation_simulator.rb +152 -0
  24. data/lib/gsc/cli/dashboard.rb +354 -0
  25. data/lib/gsc/cli/doctor.rb +129 -0
  26. data/lib/gsc/cli/eeat.rb +125 -0
  27. data/lib/gsc/cli/ga4.rb +852 -0
  28. data/lib/gsc/cli/growth.rb +650 -0
  29. data/lib/gsc/cli/hreflang.rb +164 -0
  30. data/lib/gsc/cli/image_seo.rb +162 -0
  31. data/lib/gsc/cli/indexing.rb +458 -0
  32. data/lib/gsc/cli/intent_shift.rb +125 -0
  33. data/lib/gsc/cli/keyword_value.rb +134 -0
  34. data/lib/gsc/cli/keywords.rb +795 -0
  35. data/lib/gsc/cli/landing_roi.rb +308 -0
  36. data/lib/gsc/cli/low_ctr.rb +213 -0
  37. data/lib/gsc/cli/mobile_parity.rb +150 -0
  38. data/lib/gsc/cli/report.rb +100 -0
  39. data/lib/gsc/cli/rich_results.rb +172 -0
  40. data/lib/gsc/cli/schema_generate.rb +149 -0
  41. data/lib/gsc/cli/seasonal.rb +232 -0
  42. data/lib/gsc/cli/security.rb +153 -0
  43. data/lib/gsc/cli/setup.rb +1291 -0
  44. data/lib/gsc/cli/sitemap_tree.rb +143 -0
  45. data/lib/gsc/cli/skill_pack.rb +62 -0
  46. data/lib/gsc/cli/soft_404.rb +199 -0
  47. data/lib/gsc/cli/sparkline.rb +227 -0
  48. data/lib/gsc/cli/watchdog.rb +150 -0
  49. data/lib/gsc/cli/zombie_purger.rb +208 -0
  50. data/lib/gsc/cli.rb +706 -5111
  51. data/lib/gsc/cli_advanced.rb +1513 -0
  52. data/lib/gsc/client.rb +17 -2
  53. data/lib/gsc/color.rb +16 -1
  54. data/lib/gsc/command_registry.rb +47 -9
  55. data/lib/gsc/config.rb +11 -2
  56. data/lib/gsc/content_gap.rb +112 -0
  57. data/lib/gsc/ctr_curve.rb +115 -0
  58. data/lib/gsc/decay_predictor.rb +322 -0
  59. data/lib/gsc/doctor.rb +434 -0
  60. data/lib/gsc/eeat_auditor.rb +428 -0
  61. data/lib/gsc/entity_auditor.rb +229 -0
  62. data/lib/gsc/firewall_scanner.rb +733 -0
  63. data/lib/gsc/geo_auditor.rb +368 -0
  64. data/lib/gsc/google_suggest.rb +109 -0
  65. data/lib/gsc/google_trends.rb +8 -1
  66. data/lib/gsc/heading_validator.rb +283 -0
  67. data/lib/gsc/hreflang_validator.rb +412 -0
  68. data/lib/gsc/image_seo.rb +286 -0
  69. data/lib/gsc/indexing_queue.rb +179 -0
  70. data/lib/gsc/indexnow.rb +93 -0
  71. data/lib/gsc/intent_shift.rb +188 -0
  72. data/lib/gsc/internal_links.rb +249 -0
  73. data/lib/gsc/keyword_value.rb +191 -0
  74. data/lib/gsc/landing_roi.rb +195 -0
  75. data/lib/gsc/llms_generator.rb +425 -0
  76. data/lib/gsc/low_ctr_rewriter.rb +408 -0
  77. data/lib/gsc/mobile_parity.rb +222 -0
  78. data/lib/gsc/network_tracer.rb +93 -0
  79. data/lib/gsc/open_page_rank.rb +72 -0
  80. data/lib/gsc/page_analyzer.rb +47 -7
  81. data/lib/gsc/page_comparator.rb +108 -0
  82. data/lib/gsc/page_speed.rb +110 -0
  83. data/lib/gsc/prompts.rb +38 -29
  84. data/lib/gsc/questions_harvester.rb +178 -0
  85. data/lib/gsc/report_generator.rb +461 -0
  86. data/lib/gsc/rich_results.rb +388 -0
  87. data/lib/gsc/robots_checker.rb +114 -0
  88. data/lib/gsc/schema_generator.rb +788 -0
  89. data/lib/gsc/schema_validator.rb +120 -0
  90. data/lib/gsc/seasonal_predictor.rb +381 -0
  91. data/lib/gsc/security_scanner.rb +496 -0
  92. data/lib/gsc/serp_feature_detector.rb +359 -0
  93. data/lib/gsc/serp_preview.rb +152 -0
  94. data/lib/gsc/site_crawler.rb +113 -21
  95. data/lib/gsc/sitemap_loader.rb +15 -4
  96. data/lib/gsc/sitemap_tree.rb +301 -0
  97. data/lib/gsc/skill_pack.rb +195 -0
  98. data/lib/gsc/soft_404_analyzer.rb +385 -0
  99. data/lib/gsc/sparkline.rb +171 -0
  100. data/lib/gsc/speed_correlator.rb +416 -0
  101. data/lib/gsc/striking_playbook.rb +190 -0
  102. data/lib/gsc/title_optimizer.rb +420 -0
  103. data/lib/gsc/vault.rb +260 -0
  104. data/lib/gsc/version.rb +1 -1
  105. data/lib/gsc/watchdog.rb +235 -0
  106. data/lib/gsc/zombie_purger.rb +366 -0
  107. data/lib/gsc.rb +144 -0
  108. metadata +91 -2
@@ -0,0 +1,195 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'json'
5
+ require 'uri'
6
+
7
+ module GSC
8
+ class LandingRoi
9
+ DEFAULT_AOV = 75.0 # Average Order Value / Customer LTV ($)
10
+ DEFAULT_CONV_RATE = 0.025 # 2.5% Conversion rate of engaged visitors
11
+ DEFAULT_BENCHMARK_BOUNCE = 45.0 # 45% standard healthy bounce rate
12
+ DEFAULT_CPC = 1.50 # Replacement PPC cost per click
13
+
14
+ attr_reader :options, :aov, :conv_rate, :benchmark_bounce, :cpc
15
+
16
+ def initialize(options = {})
17
+ @options = options
18
+ @aov = (options[:aov] || DEFAULT_AOV).to_f
19
+ @conv_rate = (options[:conv_rate] || DEFAULT_CONV_RATE).to_f
20
+ @benchmark_bounce = (options[:benchmark_bounce] || DEFAULT_BENCHMARK_BOUNCE).to_f
21
+ @cpc = (options[:cpc] || DEFAULT_CPC).to_f
22
+ end
23
+
24
+ # Analyzes combined GSC pre-click and GA4 post-click data
25
+ def analyze_pages(pages_data)
26
+ analyzed = pages_data.map do |page|
27
+ analyze_single_page(page)
28
+ end
29
+
30
+ # Sort by highest monthly revenue leak by default
31
+ analyzed.sort_by! { |p| -(p[:monthly_revenue_leak] || 0.0) }
32
+
33
+ total_clicks = analyzed.sum { |p| p[:clicks] || 0 }
34
+ total_sessions = analyzed.sum { |p| p[:sessions] || 0 }
35
+ total_leak = analyzed.sum { |p| p[:monthly_revenue_leak] || 0.0 }
36
+ total_actual_rev = analyzed.sum { |p| p[:actual_revenue_est] || 0.0 }
37
+ total_lost_visitors = analyzed.sum { |p| p[:lost_visitors] || 0 }
38
+
39
+ quadrant_summary = {
40
+ cash_cows: analyzed.count { |p| p[:quadrant] == :cash_cow },
41
+ revenue_leakers: analyzed.count { |p| p[:quadrant] == :revenue_leaker },
42
+ hidden_gems: analyzed.count { |p| p[:quadrant] == :hidden_gem },
43
+ zombies: analyzed.count { |p| p[:quadrant] == :zombie }
44
+ }
45
+
46
+ {
47
+ total_pages: analyzed.size,
48
+ total_clicks: total_clicks,
49
+ total_sessions: total_sessions,
50
+ total_monthly_leak: total_leak.round(2),
51
+ total_annual_leak: (total_leak * 12.0).round(2),
52
+ total_actual_revenue_est: total_actual_rev.round(2),
53
+ total_lost_visitors: total_lost_visitors,
54
+ quadrants: quadrant_summary,
55
+ assumptions: {
56
+ aov: @aov,
57
+ conv_rate_pct: (@conv_rate * 100.0).round(2),
58
+ benchmark_bounce_pct: @benchmark_bounce,
59
+ estimated_cpc: @cpc
60
+ },
61
+ pages: analyzed
62
+ }
63
+ end
64
+
65
+ def analyze_single_page(page)
66
+ url = page[:url] || page['url'] || page[:path] || page['path'] || '/'
67
+ clicks = (page[:clicks] || page['clicks'] || 0).to_i
68
+ impressions = (page[:impressions] || page['impressions'] || 0).to_i
69
+ position = (page[:position] || page['position'] || 0.0).to_f.round(1)
70
+ ctr = (page[:ctr] || page['ctr'] || 0.0).to_f.round(2)
71
+ sessions = (page[:sessions] || page['sessions'] || clicks).to_i
72
+ bounce_rate = (page[:bounce_rate] || page['bounce_rate'] || estimate_bounce_rate(clicks, position)).to_f.round(1)
73
+ duration = (page[:duration] || page['duration'] || page[:duration_seconds] || 60.0).to_f.round(1)
74
+
75
+ # Economic Calculations
76
+ excess_bounce = [0.0, bounce_rate - @benchmark_bounce].max
77
+ lost_visitors = (clicks * (excess_bounce / 100.0)).round
78
+ monthly_revenue_leak = (lost_visitors * @conv_rate * @aov).round(2)
79
+ annual_revenue_leak = (monthly_revenue_leak * 12.0).round(2)
80
+
81
+ engaged_clicks = [0, clicks - lost_visitors].max
82
+ actual_revenue_est = (engaged_clicks * @conv_rate * @aov).round(2)
83
+ traffic_asset_value = (clicks * @cpc).round(2)
84
+
85
+ # Page Economic Health Index (PEHI 0-100)
86
+ # High clicks + low bounce + high duration = 100
87
+ pehi_score = calculate_pehi(clicks, bounce_rate, duration)
88
+
89
+ # Strategic Quadrant
90
+ quadrant = classify_quadrant(clicks, bounce_rate, duration, position)
91
+
92
+ # Conversion Prescriptions & Fixes
93
+ prescriptions = generate_cro_prescriptions(bounce_rate, duration, clicks, position)
94
+
95
+ {
96
+ url: url,
97
+ clicks: clicks,
98
+ impressions: impressions,
99
+ position: position,
100
+ ctr: ctr,
101
+ sessions: sessions,
102
+ bounce_rate: bounce_rate,
103
+ duration_seconds: duration,
104
+ pehi_score: pehi_score,
105
+ quadrant: quadrant,
106
+ lost_visitors: lost_visitors,
107
+ monthly_revenue_leak: monthly_revenue_leak,
108
+ annual_revenue_leak: annual_revenue_leak,
109
+ actual_revenue_est: actual_revenue_est,
110
+ traffic_asset_value: traffic_asset_value,
111
+ prescriptions: prescriptions
112
+ }
113
+ end
114
+
115
+ private
116
+
117
+ def calculate_pehi(clicks, bounce_rate, duration)
118
+ score = 50.0
119
+
120
+ # Bounce Rate Component (max +/- 30pts)
121
+ if bounce_rate <= 35.0
122
+ score += 30.0
123
+ elsif bounce_rate <= 45.0
124
+ score += 15.0
125
+ elsif bounce_rate >= 75.0
126
+ score -= 30.0
127
+ elsif bounce_rate >= 60.0
128
+ score -= 15.0
129
+ end
130
+
131
+ # Duration Component (max +/- 15pts)
132
+ if duration >= 120.0
133
+ score += 15.0
134
+ elsif duration >= 60.0
135
+ score += 8.0
136
+ elsif duration <= 25.0
137
+ score -= 15.0
138
+ elsif duration <= 45.0
139
+ score -= 8.0
140
+ end
141
+
142
+ # Volume Bonus (max +5pts)
143
+ score += 5.0 if clicks >= 100
144
+
145
+ score.clamp(0.0, 100.0).round
146
+ end
147
+
148
+ def classify_quadrant(clicks, bounce_rate, duration, position)
149
+ if clicks >= 25 && bounce_rate <= @benchmark_bounce && duration >= 45.0
150
+ :cash_cow
151
+ elsif clicks >= 25 && bounce_rate > @benchmark_bounce
152
+ :revenue_leaker
153
+ elsif clicks < 25 && bounce_rate <= @benchmark_bounce && duration >= 60.0
154
+ :hidden_gem
155
+ else
156
+ :zombie
157
+ end
158
+ end
159
+
160
+ def generate_cro_prescriptions(bounce_rate, duration, clicks, position)
161
+ fixes = []
162
+
163
+ if bounce_rate >= 70.0
164
+ fixes << "🚨 HIGH BOUNCE TRAP: Place sticky, high-contrast CTA button above 450px fold."
165
+ fixes << "🛡️ ADD SOCIAL PROOF: Insert customer rating stars, logos, or guarantee badge in the top viewport."
166
+ elsif bounce_rate > 55.0
167
+ fixes << "⚡ REDUCE BOUNCE: Add interactive jump-links / Table of Contents to lower drop-off."
168
+ end
169
+
170
+ if duration < 30.0
171
+ fixes << "⏱️ LOW TIME ON PAGE: Searchers are not finding answers instantly; add a 2-sentence executive summary under H1."
172
+ end
173
+
174
+ if clicks >= 50 && bounce_rate > 60.0
175
+ fixes << "💰 HIGH TRAFFIC BLEED: Priority CRO target. Run A/B test on hero headline and primary action."
176
+ end
177
+
178
+ if position >= 7.0 && bounce_rate <= 40.0
179
+ fixes << "⭐ HIDDEN HIGH CONVERTER: Searchers love this page (low bounce). Build 3 internal links to push from pos #{position} to Top 3."
180
+ end
181
+
182
+ fixes << "✅ HEALTHY ENGAGEMENT: Maintain current content structure and monitor core queries." if fixes.empty?
183
+ fixes
184
+ end
185
+
186
+ def estimate_bounce_rate(clicks, position)
187
+ # Fallback heuristic when GA4 is not linked:
188
+ # Informational / deeper SERP positions generally experience higher bounce rates
189
+ base = 50.0
190
+ base += 5.0 if position > 5.0
191
+ base += 8.0 if position > 10.0
192
+ base.clamp(35.0, 80.0)
193
+ end
194
+ end
195
+ end
@@ -0,0 +1,425 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'uri'
5
+ require 'net/http'
6
+ require 'time'
7
+ require 'fileutils'
8
+ require_relative 'sitemap_loader'
9
+ require_relative 'page_analyzer'
10
+
11
+ module GSC
12
+ class LlmsGenerator
13
+ attr_reader :base_url, :options
14
+
15
+ def initialize(base_url, options = {})
16
+ @base_url = base_url.to_s.strip
17
+ @base_url = "https://#{@base_url}" unless @base_url =~ %r{^https?://}
18
+ @options = options || {}
19
+ end
20
+
21
+ # Generate standard /llms.txt index file per specification
22
+ def generate_llms_txt(title: nil, summary: nil, limit: 25)
23
+ pages_meta = collect_pages_metadata(limit: limit)
24
+
25
+ site_title = title || URI.parse(@base_url).host.sub(/^www\./, '').capitalize
26
+ site_summary = summary || "Official developer documentation, guides, and knowledge base for #{site_title}."
27
+
28
+ out = []
29
+ out << "# #{site_title}"
30
+ out << ""
31
+ out << "> #{site_summary}"
32
+ out << ""
33
+ out << "## Core Documentation"
34
+ out << ""
35
+
36
+ core_pages = pages_meta.first(12)
37
+ core_pages.each do |pm|
38
+ out << "- [#{pm[:title]}](#{pm[:url]}): #{pm[:description]}"
39
+ end
40
+
41
+ optional_pages = pages_meta[12..-1] || []
42
+ if optional_pages.any?
43
+ out << ""
44
+ out << "## Extended Guides & Articles"
45
+ out << ""
46
+ optional_pages.each do |pm|
47
+ out << "- [#{pm[:title]}](#{pm[:url]}): #{pm[:description]}"
48
+ end
49
+ end
50
+
51
+ out << ""
52
+ out << "## Optional"
53
+ out << ""
54
+ out << "- [Full Consolidated Knowledge Base](#{@base_url}/llms-full.txt): Complete multi-page markdown knowledge bundle for LLM context ingestion."
55
+ out << ""
56
+
57
+ out.join("\n")
58
+ end
59
+
60
+ # Generate complete /llms-full.txt multi-page consolidated knowledge bundle
61
+ def generate_llms_full_txt(title: nil, summary: nil, limit: 25, gsc_rows: nil)
62
+ pages_meta = collect_pages_metadata(limit: limit)
63
+ site_title = title || URI.parse(@base_url).host.sub(/^www\./, '').capitalize
64
+ site_summary = summary || "Comprehensive knowledge base and technical specifications for #{site_title}."
65
+
66
+ toc = []
67
+ docs_content = []
68
+
69
+ pages_meta.each_with_index do |pm, idx|
70
+ section_id = "section-#{idx + 1}"
71
+ toc << "#{idx + 1}. [#{pm[:title]}](##{section_id})"
72
+
73
+ html = fetch_html(pm[:url])
74
+ md_body = html_to_clean_markdown(html, pm[:url])
75
+
76
+ # FAQ injection from GSC queries if available
77
+ faq_block = generate_gsc_faq_block(pm[:url], gsc_rows)
78
+
79
+ doc_section = []
80
+ doc_section << "<a name=\"#{section_id}\"></a>"
81
+ doc_section << "# #{pm[:title]}"
82
+ doc_section << ""
83
+ doc_section << "**Source URL**: #{pm[:url]}"
84
+ doc_section << "**Description**: #{pm[:description]}"
85
+ doc_section << ""
86
+ doc_section << md_body
87
+ doc_section << ""
88
+ doc_section << faq_block if faq_block
89
+ doc_section << ""
90
+ doc_section << "---"
91
+ doc_section << ""
92
+
93
+ docs_content << doc_section.join("\n")
94
+ end
95
+
96
+ out = []
97
+ out << "# #{site_title} — Full Knowledge Base (`/llms-full.txt`)"
98
+ out << ""
99
+ out << "> #{site_summary}"
100
+ out << "> Generated automatically by Google Search Console CLI (`gsc llms --full`) on #{Time.now.strftime('%Y-%m-%d')}."
101
+ out << ""
102
+ out << "## Table of Contents"
103
+ out << ""
104
+ out << toc.join("\n")
105
+ out << ""
106
+ out << "---"
107
+ out << ""
108
+ out << docs_content.join("\n")
109
+
110
+ out.join("\n")
111
+ end
112
+
113
+ # Package complete bundle (both llms.txt and llms-full.txt) with token analytics
114
+ def package_agent_bundle(output_dir: nil, limit: 25, gsc_rows: nil)
115
+ llms_txt = generate_llms_txt(limit: limit)
116
+ llms_full_txt = generate_llms_full_txt(limit: limit, gsc_rows: gsc_rows)
117
+
118
+ # Token estimation (standard 4 chars per token)
119
+ txt_tokens = (llms_txt.length / 4.0).ceil
120
+ full_tokens = (llms_full_txt.length / 4.0).ceil
121
+
122
+ saved_files = []
123
+ if output_dir
124
+ FileUtils.mkdir_p(output_dir)
125
+ txt_path = File.join(output_dir, "llms.txt")
126
+ full_path = File.join(output_dir, "llms-full.txt")
127
+
128
+ File.write(txt_path, llms_txt, encoding: 'UTF-8')
129
+ File.write(full_path, llms_full_txt, encoding: 'UTF-8')
130
+
131
+ saved_files << { path: txt_path, size_bytes: File.size(txt_path) }
132
+ saved_files << { path: full_path, size_bytes: File.size(full_path) }
133
+ end
134
+
135
+ context_windows = {
136
+ claude_3_5_sonnet_200k: (full_tokens <= 200_000),
137
+ gpt_4o_128k: (full_tokens <= 128_000),
138
+ gemini_1_5_pro_1m: (full_tokens <= 1_000_000)
139
+ }
140
+
141
+ {
142
+ base_url: @base_url,
143
+ total_pages_packaged: limit,
144
+ llms_txt: {
145
+ chars: llms_txt.length,
146
+ tokens: txt_tokens,
147
+ content: llms_txt
148
+ },
149
+ llms_full_txt: {
150
+ chars: llms_full_txt.length,
151
+ tokens: full_tokens,
152
+ content: llms_full_txt
153
+ },
154
+ context_window_fit: context_windows,
155
+ saved_files: saved_files
156
+ }
157
+ end
158
+
159
+ # HTML to Clean Markdown Converter (Zero Gem Dependencies)
160
+ def html_to_clean_markdown(html, base_url = '')
161
+ return "" if html.nil? || html.empty?
162
+
163
+ # 1. Strip comments, scripts, styles, nav, footer, header, svg, noscript
164
+ clean = html.to_s.dup.force_encoding('UTF-8').scrub
165
+ clean.gsub!(/<!--.*?-->/m, '')
166
+ clean.gsub!(/<script\b[^>]*>.*?<\/script>/mi, '')
167
+ clean.gsub!(/<style\b[^>]*>.*?<\/style>/mi, '')
168
+ clean.gsub!(/<noscript\b[^>]*>.*?<\/noscript>/mi, '')
169
+ clean.gsub!(/<svg\b[^>]*>.*?<\/svg>/mi, '')
170
+ clean.gsub!(/<iframe\b[^>]*>.*?<\/iframe>/mi, '')
171
+ clean.gsub!(/<nav\b[^>]*>.*?<\/nav>/mi, '')
172
+ clean.gsub!(/<footer\b[^>]*>.*?<\/footer>/mi, '')
173
+ clean.gsub!(/<header\b[^>]*>.*?<\/header>/mi, '')
174
+ clean.gsub!(/<aside\b[^>]*>.*?<\/aside>/mi, '')
175
+
176
+ # 2. Convert Headings (h1 - h6)
177
+ (1..6).each do |level|
178
+ clean.gsub!(/<h#{level}\b[^>]*>(.*?)<\/h#{level}>/mi) do
179
+ heading_text = strip_tags($1).strip
180
+ "\n\n#{'#' * level} #{heading_text}\n\n"
181
+ end
182
+ end
183
+
184
+ # 3. Convert Tables to GFM Markdown
185
+ clean.gsub!(/<table\b[^>]*>(.*?)<\/table>/mi) do
186
+ table_html = $1
187
+ convert_table_to_markdown(table_html)
188
+ end
189
+
190
+ # 4. Convert Lists
191
+ clean.gsub!(/<li\b[^>]*>(.*?)<\/li>/mi) do
192
+ li_text = strip_tags($1).strip
193
+ "\n- #{li_text}"
194
+ end
195
+ clean.gsub!(/<\/?(ul|ol)\b[^>]*>/mi, "\n")
196
+
197
+ # 5. Convert Code blocks
198
+ clean.gsub!(/<pre\b[^>]*><code\b[^>]*>(.*?)<\/code><\/pre>/mi) do
199
+ code = decode_html_entities($1).strip
200
+ "\n```\n#{code}\n```\n"
201
+ end
202
+ clean.gsub!(/<code\b[^>]*>(.*?)<\/code>/mi) do
203
+ code = decode_html_entities($1).strip
204
+ "`#{code}`"
205
+ end
206
+
207
+ # 6. Convert Bold and Italic
208
+ clean.gsub!(/<(b|strong)\b[^>]*>(.*?)<\/\1>/mi) { "**#{strip_tags($2).strip}**" }
209
+ clean.gsub!(/<(i|em)\b[^>]*>(.*?)<\/\1>/mi) { "*#{strip_tags($2).strip}*" }
210
+
211
+ # 7. Convert Links
212
+ clean.gsub!(/<a\b[^>]*href=["']([^"']+)["'][^>]*>(.*?)<\/a>/mi) do
213
+ href = $1.strip
214
+ text = strip_tags($2).strip
215
+ next text if text.empty?
216
+ resolved_href = URI.join(base_url, href).to_s rescue href
217
+ safe_text = text.gsub('[', '\[').gsub(']', '\]')
218
+ "[#{safe_text}](#{resolved_href})"
219
+ end
220
+
221
+ # 8. Convert Paragraphs and line breaks
222
+ clean.gsub!(/<br\s*\/?>/mi, " \n")
223
+ clean.gsub!(/<p\b[^>]*>(.*?)<\/p>/mi) { "\n\n#{strip_tags($1).strip}\n\n" }
224
+
225
+ # 9. Strip any remaining HTML tags
226
+ clean = strip_tags(clean)
227
+
228
+ # 10. Decode HTML Entities
229
+ clean = decode_html_entities(clean)
230
+
231
+ # 11. Normalize excessive whitespace
232
+ clean.gsub!(/\r\n/, "\n")
233
+ clean.gsub!(/\n{3,}/, "\n\n")
234
+ clean.strip
235
+ end
236
+
237
+ def audit_ai_readability(url)
238
+ pa = GSC::PageAnalyzer.new(url)
239
+ data = pa.fetch_and_analyze
240
+
241
+ html = pa.html || ''
242
+ has_tables = html.include?('<table')
243
+ has_lists = html.include?('<ul') || html.include?('<ol')
244
+ schemas = data.dig(:structured_data, :schemas) || []
245
+ h1_count = (data.dig(:headings, :h1) || []).length
246
+
247
+ score = 100
248
+ issues = []
249
+
250
+ if h1_count != 1
251
+ score -= 20
252
+ issues << "H1 count is #{h1_count} (Must be exactly 1 for clean LLM hierarchy)"
253
+ end
254
+
255
+ unless has_tables
256
+ score -= 15
257
+ issues << "No <table> found (tables increase LLM citation and fact extraction by 3x)"
258
+ end
259
+
260
+ unless has_lists
261
+ score -= 15
262
+ issues << "No bullet lists (<ul> or <ol>) found for quick entity consumption"
263
+ end
264
+
265
+ if schemas.empty?
266
+ score -= 20
267
+ issues << "No JSON-LD schemas detected (structured data accelerates AI knowledge graph inclusion)"
268
+ end
269
+
270
+ {
271
+ url: url,
272
+ ai_readability_score: [score, 0].max,
273
+ grade: score >= 80 ? 'A (Excellent)' : (score >= 60 ? 'B (Acceptable)' : 'C (Needs Work)'),
274
+ issues: issues,
275
+ features: {
276
+ has_tables: has_tables,
277
+ has_lists: has_lists,
278
+ schemas_found: schemas.length,
279
+ h1_count: h1_count
280
+ }
281
+ }
282
+ end
283
+
284
+ private
285
+
286
+ def collect_pages_metadata(limit: 25)
287
+ sitemap_target = "#{@base_url.chomp('/')}/sitemap.xml"
288
+ urls = (GSC::SitemapLoader.resolve_urls(sitemap_target, @base_url, quiet: true) rescue [@base_url]) || [@base_url]
289
+ urls = [@base_url] if urls.empty?
290
+
291
+ # Also add standard routes if sitemap is small
292
+ default_routes = ["/about", "/features", "/pricing", "/faq", "/docs", "/contact"].map do |r|
293
+ URI.join(@base_url, r).to_s rescue nil
294
+ end.compact
295
+
296
+ candidate_urls = (urls + default_routes).uniq.first(limit)
297
+ metadata = []
298
+
299
+ candidate_urls.each do |url|
300
+ html = fetch_html(url)
301
+ next unless html && !html.empty?
302
+
303
+ title = extract_title(html) || url
304
+ desc = extract_description(html) || "Guide and specifications for #{title}."
305
+
306
+ metadata << {
307
+ url: url,
308
+ title: title,
309
+ description: desc
310
+ }
311
+ end
312
+
313
+ metadata.empty? ? [{ url: @base_url, title: "Home", description: "Home page" }] : metadata
314
+ end
315
+
316
+ def fetch_html(url)
317
+ uri = URI.parse(url) rescue nil
318
+ return "" unless uri && uri.host
319
+
320
+ begin
321
+ Net::HTTP.start(uri.hostname, uri.port, use_ssl: (uri.scheme == 'https'), open_timeout: 5, read_timeout: 8) do |http|
322
+ req = Net::HTTP::Get.new(uri.request_uri)
323
+ req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) gsc-cli/2.1'
324
+ req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
325
+ res = http.request(req)
326
+ res.body.to_s[0..60_000] if res.body
327
+ end
328
+ rescue StandardError
329
+ ""
330
+ end
331
+ end
332
+
333
+ def extract_title(html)
334
+ if html =~ /<title\b[^>]*>(.*?)<\/title>/mi
335
+ strip_tags(decode_html_entities($1)).strip
336
+ end
337
+ end
338
+
339
+ def extract_description(html)
340
+ if html =~ /<meta\b[^>]*name=["']description["'][^>]*content=["']([^"']+)["']/mi
341
+ strip_tags(decode_html_entities($1)).strip
342
+ elsif html =~ /<meta\b[^>]*content=["']([^"']+)["'][^>]*name=["']description["']/mi
343
+ strip_tags(decode_html_entities($1)).strip
344
+ end
345
+ end
346
+
347
+ def convert_table_to_markdown(table_html)
348
+ rows = []
349
+ table_html.scan(/<tr\b[^>]*>(.*?)<\/tr>/mi).each do |tr_match|
350
+ row_content = tr_match.first
351
+ cells = []
352
+ row_content.scan(/<(th|td)\b[^>]*>(.*?)<\/\1>/mi).each do |_tag, content|
353
+ cells << strip_tags(content).gsub('|', '\\|').strip
354
+ end
355
+ rows << cells unless cells.empty?
356
+ end
357
+
358
+ return "" if rows.empty?
359
+
360
+ col_count = rows.map(&:size).max
361
+ rows.each { |r| r.fill('', r.size...col_count) }
362
+
363
+ md = []
364
+ # Header
365
+ header = rows.first
366
+ md << "| #{header.join(' | ')} |"
367
+ md << "| #{Array.new(col_count, '---').join(' | ')} |"
368
+
369
+ # Body
370
+ rows[1..-1].each do |r|
371
+ md << "| #{r.join(' | ')} |"
372
+ end
373
+
374
+ "\n\n" + md.join("\n") + "\n\n"
375
+ end
376
+
377
+ def generate_gsc_faq_block(url, gsc_rows)
378
+ return nil unless gsc_rows && gsc_rows.is_a?(Array)
379
+
380
+ clean_target = url.downcase.sub(%r{/$}, '')
381
+ matching_queries = []
382
+
383
+ gsc_rows.each do |r|
384
+ keys = r['keys'] || []
385
+ page = keys.find { |k| k =~ %r{^https?://} }
386
+ next unless page && page.downcase.sub(%r{/$}, '') == clean_target
387
+
388
+ q = keys.find { |k| k !~ %r{^https?://} }
389
+ clicks = (r['clicks'] || 0).to_i
390
+ matching_queries << { query: q, clicks: clicks } if q
391
+ end
392
+
393
+ return nil if matching_queries.empty?
394
+
395
+ top_q = matching_queries.sort_by { |item| -item[:clicks] }.first(5)
396
+ faq = []
397
+ faq << "### Search Questions & High-Intent Queries (Verified from Google Search Console)"
398
+ faq << ""
399
+ top_q.each do |item|
400
+ faq << "- **Q: How does #{item[:query]} work?**"
401
+ faq << " *A: Detailed in the core specifications above.*"
402
+ end
403
+ faq.join("\n")
404
+ end
405
+
406
+ def strip_tags(text)
407
+ text.to_s.gsub(/<[^>]+>/, ' ')
408
+ end
409
+
410
+ def decode_html_entities(text)
411
+ t = text.to_s.dup
412
+ t.gsub!('&amp;', '&')
413
+ t.gsub!('&quot;', '"')
414
+ t.gsub!('&#39;', "'")
415
+ t.gsub!('&apos;', "'")
416
+ t.gsub!('&lt;', '<')
417
+ t.gsub!('&gt;', '>')
418
+ t.gsub!('&nbsp;', ' ')
419
+ t.gsub!('&#8217;', "'")
420
+ t.gsub!('&#8220;', '"')
421
+ t.gsub!('&#8221;', '"')
422
+ t
423
+ end
424
+ end
425
+ end