gsc-cli 2.0.2 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/AUTH.md +205 -0
- data/FUNDING.md +120 -0
- data/README.md +463 -299
- data/bin/gsc +29158 -4921
- data/dist/gsc +29158 -4921
- data/lib/gsc/aio_hunter.rb +343 -0
- data/lib/gsc/answer_synthesizer.rb +157 -0
- data/lib/gsc/api.rb +53 -1
- data/lib/gsc/auth.rb +26 -0
- data/lib/gsc/backlinks_manager.rb +96 -0
- data/lib/gsc/brand_segmenter.rb +140 -0
- data/lib/gsc/cache_manager.rb +806 -0
- data/lib/gsc/cannibalization_analyzer.rb +141 -0
- data/lib/gsc/canonical_chains.rb +367 -0
- data/lib/gsc/citation_simulator.rb +339 -0
- data/lib/gsc/cli/aio_hunter.rb +154 -0
- data/lib/gsc/cli/analytics.rb +788 -0
- data/lib/gsc/cli/audit.rb +1976 -0
- data/lib/gsc/cli/base.rb +384 -0
- data/lib/gsc/cli/cache.rb +266 -0
- data/lib/gsc/cli/canonical.rb +223 -0
- data/lib/gsc/cli/citation_simulator.rb +152 -0
- data/lib/gsc/cli/dashboard.rb +354 -0
- data/lib/gsc/cli/doctor.rb +129 -0
- data/lib/gsc/cli/eeat.rb +125 -0
- data/lib/gsc/cli/ga4.rb +852 -0
- data/lib/gsc/cli/growth.rb +650 -0
- data/lib/gsc/cli/hreflang.rb +164 -0
- data/lib/gsc/cli/image_seo.rb +162 -0
- data/lib/gsc/cli/indexing.rb +458 -0
- data/lib/gsc/cli/intent_shift.rb +125 -0
- data/lib/gsc/cli/keyword_value.rb +134 -0
- data/lib/gsc/cli/keywords.rb +795 -0
- data/lib/gsc/cli/landing_roi.rb +308 -0
- data/lib/gsc/cli/low_ctr.rb +213 -0
- data/lib/gsc/cli/mobile_parity.rb +150 -0
- data/lib/gsc/cli/report.rb +100 -0
- data/lib/gsc/cli/rich_results.rb +172 -0
- data/lib/gsc/cli/schema_generate.rb +149 -0
- data/lib/gsc/cli/seasonal.rb +232 -0
- data/lib/gsc/cli/security.rb +153 -0
- data/lib/gsc/cli/setup.rb +1291 -0
- data/lib/gsc/cli/sitemap_tree.rb +143 -0
- data/lib/gsc/cli/skill_pack.rb +62 -0
- data/lib/gsc/cli/soft_404.rb +199 -0
- data/lib/gsc/cli/sparkline.rb +227 -0
- data/lib/gsc/cli/watchdog.rb +150 -0
- data/lib/gsc/cli/zombie_purger.rb +208 -0
- data/lib/gsc/cli.rb +706 -5111
- data/lib/gsc/cli_advanced.rb +1513 -0
- data/lib/gsc/client.rb +17 -2
- data/lib/gsc/color.rb +16 -1
- data/lib/gsc/command_registry.rb +47 -9
- data/lib/gsc/config.rb +11 -2
- data/lib/gsc/content_gap.rb +112 -0
- data/lib/gsc/ctr_curve.rb +115 -0
- data/lib/gsc/decay_predictor.rb +322 -0
- data/lib/gsc/doctor.rb +434 -0
- data/lib/gsc/eeat_auditor.rb +428 -0
- data/lib/gsc/entity_auditor.rb +229 -0
- data/lib/gsc/firewall_scanner.rb +733 -0
- data/lib/gsc/geo_auditor.rb +368 -0
- data/lib/gsc/google_suggest.rb +109 -0
- data/lib/gsc/google_trends.rb +8 -1
- data/lib/gsc/heading_validator.rb +283 -0
- data/lib/gsc/hreflang_validator.rb +412 -0
- data/lib/gsc/image_seo.rb +286 -0
- data/lib/gsc/indexing_queue.rb +179 -0
- data/lib/gsc/indexnow.rb +93 -0
- data/lib/gsc/intent_shift.rb +188 -0
- data/lib/gsc/internal_links.rb +249 -0
- data/lib/gsc/keyword_value.rb +191 -0
- data/lib/gsc/landing_roi.rb +195 -0
- data/lib/gsc/llms_generator.rb +425 -0
- data/lib/gsc/low_ctr_rewriter.rb +408 -0
- data/lib/gsc/mobile_parity.rb +222 -0
- data/lib/gsc/network_tracer.rb +93 -0
- data/lib/gsc/open_page_rank.rb +72 -0
- data/lib/gsc/page_analyzer.rb +47 -7
- data/lib/gsc/page_comparator.rb +108 -0
- data/lib/gsc/page_speed.rb +110 -0
- data/lib/gsc/prompts.rb +38 -29
- data/lib/gsc/questions_harvester.rb +178 -0
- data/lib/gsc/report_generator.rb +461 -0
- data/lib/gsc/rich_results.rb +388 -0
- data/lib/gsc/robots_checker.rb +114 -0
- data/lib/gsc/schema_generator.rb +788 -0
- data/lib/gsc/schema_validator.rb +120 -0
- data/lib/gsc/seasonal_predictor.rb +381 -0
- data/lib/gsc/security_scanner.rb +496 -0
- data/lib/gsc/serp_feature_detector.rb +359 -0
- data/lib/gsc/serp_preview.rb +152 -0
- data/lib/gsc/site_crawler.rb +113 -21
- data/lib/gsc/sitemap_loader.rb +15 -4
- data/lib/gsc/sitemap_tree.rb +301 -0
- data/lib/gsc/skill_pack.rb +195 -0
- data/lib/gsc/soft_404_analyzer.rb +385 -0
- data/lib/gsc/sparkline.rb +171 -0
- data/lib/gsc/speed_correlator.rb +416 -0
- data/lib/gsc/striking_playbook.rb +190 -0
- data/lib/gsc/title_optimizer.rb +420 -0
- data/lib/gsc/vault.rb +260 -0
- data/lib/gsc/version.rb +1 -1
- data/lib/gsc/watchdog.rb +235 -0
- data/lib/gsc/zombie_purger.rb +366 -0
- data/lib/gsc.rb +144 -0
- metadata +91 -2
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'json'
|
|
5
|
+
require 'uri'
|
|
6
|
+
|
|
7
|
+
module GSC
|
|
8
|
+
class LandingRoi
|
|
9
|
+
DEFAULT_AOV = 75.0 # Average Order Value / Customer LTV ($)
|
|
10
|
+
DEFAULT_CONV_RATE = 0.025 # 2.5% Conversion rate of engaged visitors
|
|
11
|
+
DEFAULT_BENCHMARK_BOUNCE = 45.0 # 45% standard healthy bounce rate
|
|
12
|
+
DEFAULT_CPC = 1.50 # Replacement PPC cost per click
|
|
13
|
+
|
|
14
|
+
attr_reader :options, :aov, :conv_rate, :benchmark_bounce, :cpc
|
|
15
|
+
|
|
16
|
+
def initialize(options = {})
|
|
17
|
+
@options = options
|
|
18
|
+
@aov = (options[:aov] || DEFAULT_AOV).to_f
|
|
19
|
+
@conv_rate = (options[:conv_rate] || DEFAULT_CONV_RATE).to_f
|
|
20
|
+
@benchmark_bounce = (options[:benchmark_bounce] || DEFAULT_BENCHMARK_BOUNCE).to_f
|
|
21
|
+
@cpc = (options[:cpc] || DEFAULT_CPC).to_f
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
# Analyzes combined GSC pre-click and GA4 post-click data
|
|
25
|
+
def analyze_pages(pages_data)
|
|
26
|
+
analyzed = pages_data.map do |page|
|
|
27
|
+
analyze_single_page(page)
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
# Sort by highest monthly revenue leak by default
|
|
31
|
+
analyzed.sort_by! { |p| -(p[:monthly_revenue_leak] || 0.0) }
|
|
32
|
+
|
|
33
|
+
total_clicks = analyzed.sum { |p| p[:clicks] || 0 }
|
|
34
|
+
total_sessions = analyzed.sum { |p| p[:sessions] || 0 }
|
|
35
|
+
total_leak = analyzed.sum { |p| p[:monthly_revenue_leak] || 0.0 }
|
|
36
|
+
total_actual_rev = analyzed.sum { |p| p[:actual_revenue_est] || 0.0 }
|
|
37
|
+
total_lost_visitors = analyzed.sum { |p| p[:lost_visitors] || 0 }
|
|
38
|
+
|
|
39
|
+
quadrant_summary = {
|
|
40
|
+
cash_cows: analyzed.count { |p| p[:quadrant] == :cash_cow },
|
|
41
|
+
revenue_leakers: analyzed.count { |p| p[:quadrant] == :revenue_leaker },
|
|
42
|
+
hidden_gems: analyzed.count { |p| p[:quadrant] == :hidden_gem },
|
|
43
|
+
zombies: analyzed.count { |p| p[:quadrant] == :zombie }
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
{
|
|
47
|
+
total_pages: analyzed.size,
|
|
48
|
+
total_clicks: total_clicks,
|
|
49
|
+
total_sessions: total_sessions,
|
|
50
|
+
total_monthly_leak: total_leak.round(2),
|
|
51
|
+
total_annual_leak: (total_leak * 12.0).round(2),
|
|
52
|
+
total_actual_revenue_est: total_actual_rev.round(2),
|
|
53
|
+
total_lost_visitors: total_lost_visitors,
|
|
54
|
+
quadrants: quadrant_summary,
|
|
55
|
+
assumptions: {
|
|
56
|
+
aov: @aov,
|
|
57
|
+
conv_rate_pct: (@conv_rate * 100.0).round(2),
|
|
58
|
+
benchmark_bounce_pct: @benchmark_bounce,
|
|
59
|
+
estimated_cpc: @cpc
|
|
60
|
+
},
|
|
61
|
+
pages: analyzed
|
|
62
|
+
}
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
def analyze_single_page(page)
|
|
66
|
+
url = page[:url] || page['url'] || page[:path] || page['path'] || '/'
|
|
67
|
+
clicks = (page[:clicks] || page['clicks'] || 0).to_i
|
|
68
|
+
impressions = (page[:impressions] || page['impressions'] || 0).to_i
|
|
69
|
+
position = (page[:position] || page['position'] || 0.0).to_f.round(1)
|
|
70
|
+
ctr = (page[:ctr] || page['ctr'] || 0.0).to_f.round(2)
|
|
71
|
+
sessions = (page[:sessions] || page['sessions'] || clicks).to_i
|
|
72
|
+
bounce_rate = (page[:bounce_rate] || page['bounce_rate'] || estimate_bounce_rate(clicks, position)).to_f.round(1)
|
|
73
|
+
duration = (page[:duration] || page['duration'] || page[:duration_seconds] || 60.0).to_f.round(1)
|
|
74
|
+
|
|
75
|
+
# Economic Calculations
|
|
76
|
+
excess_bounce = [0.0, bounce_rate - @benchmark_bounce].max
|
|
77
|
+
lost_visitors = (clicks * (excess_bounce / 100.0)).round
|
|
78
|
+
monthly_revenue_leak = (lost_visitors * @conv_rate * @aov).round(2)
|
|
79
|
+
annual_revenue_leak = (monthly_revenue_leak * 12.0).round(2)
|
|
80
|
+
|
|
81
|
+
engaged_clicks = [0, clicks - lost_visitors].max
|
|
82
|
+
actual_revenue_est = (engaged_clicks * @conv_rate * @aov).round(2)
|
|
83
|
+
traffic_asset_value = (clicks * @cpc).round(2)
|
|
84
|
+
|
|
85
|
+
# Page Economic Health Index (PEHI 0-100)
|
|
86
|
+
# High clicks + low bounce + high duration = 100
|
|
87
|
+
pehi_score = calculate_pehi(clicks, bounce_rate, duration)
|
|
88
|
+
|
|
89
|
+
# Strategic Quadrant
|
|
90
|
+
quadrant = classify_quadrant(clicks, bounce_rate, duration, position)
|
|
91
|
+
|
|
92
|
+
# Conversion Prescriptions & Fixes
|
|
93
|
+
prescriptions = generate_cro_prescriptions(bounce_rate, duration, clicks, position)
|
|
94
|
+
|
|
95
|
+
{
|
|
96
|
+
url: url,
|
|
97
|
+
clicks: clicks,
|
|
98
|
+
impressions: impressions,
|
|
99
|
+
position: position,
|
|
100
|
+
ctr: ctr,
|
|
101
|
+
sessions: sessions,
|
|
102
|
+
bounce_rate: bounce_rate,
|
|
103
|
+
duration_seconds: duration,
|
|
104
|
+
pehi_score: pehi_score,
|
|
105
|
+
quadrant: quadrant,
|
|
106
|
+
lost_visitors: lost_visitors,
|
|
107
|
+
monthly_revenue_leak: monthly_revenue_leak,
|
|
108
|
+
annual_revenue_leak: annual_revenue_leak,
|
|
109
|
+
actual_revenue_est: actual_revenue_est,
|
|
110
|
+
traffic_asset_value: traffic_asset_value,
|
|
111
|
+
prescriptions: prescriptions
|
|
112
|
+
}
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
private
|
|
116
|
+
|
|
117
|
+
def calculate_pehi(clicks, bounce_rate, duration)
|
|
118
|
+
score = 50.0
|
|
119
|
+
|
|
120
|
+
# Bounce Rate Component (max +/- 30pts)
|
|
121
|
+
if bounce_rate <= 35.0
|
|
122
|
+
score += 30.0
|
|
123
|
+
elsif bounce_rate <= 45.0
|
|
124
|
+
score += 15.0
|
|
125
|
+
elsif bounce_rate >= 75.0
|
|
126
|
+
score -= 30.0
|
|
127
|
+
elsif bounce_rate >= 60.0
|
|
128
|
+
score -= 15.0
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
# Duration Component (max +/- 15pts)
|
|
132
|
+
if duration >= 120.0
|
|
133
|
+
score += 15.0
|
|
134
|
+
elsif duration >= 60.0
|
|
135
|
+
score += 8.0
|
|
136
|
+
elsif duration <= 25.0
|
|
137
|
+
score -= 15.0
|
|
138
|
+
elsif duration <= 45.0
|
|
139
|
+
score -= 8.0
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
# Volume Bonus (max +5pts)
|
|
143
|
+
score += 5.0 if clicks >= 100
|
|
144
|
+
|
|
145
|
+
score.clamp(0.0, 100.0).round
|
|
146
|
+
end
|
|
147
|
+
|
|
148
|
+
def classify_quadrant(clicks, bounce_rate, duration, position)
|
|
149
|
+
if clicks >= 25 && bounce_rate <= @benchmark_bounce && duration >= 45.0
|
|
150
|
+
:cash_cow
|
|
151
|
+
elsif clicks >= 25 && bounce_rate > @benchmark_bounce
|
|
152
|
+
:revenue_leaker
|
|
153
|
+
elsif clicks < 25 && bounce_rate <= @benchmark_bounce && duration >= 60.0
|
|
154
|
+
:hidden_gem
|
|
155
|
+
else
|
|
156
|
+
:zombie
|
|
157
|
+
end
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
def generate_cro_prescriptions(bounce_rate, duration, clicks, position)
|
|
161
|
+
fixes = []
|
|
162
|
+
|
|
163
|
+
if bounce_rate >= 70.0
|
|
164
|
+
fixes << "🚨 HIGH BOUNCE TRAP: Place sticky, high-contrast CTA button above 450px fold."
|
|
165
|
+
fixes << "🛡️ ADD SOCIAL PROOF: Insert customer rating stars, logos, or guarantee badge in the top viewport."
|
|
166
|
+
elsif bounce_rate > 55.0
|
|
167
|
+
fixes << "⚡ REDUCE BOUNCE: Add interactive jump-links / Table of Contents to lower drop-off."
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
if duration < 30.0
|
|
171
|
+
fixes << "⏱️ LOW TIME ON PAGE: Searchers are not finding answers instantly; add a 2-sentence executive summary under H1."
|
|
172
|
+
end
|
|
173
|
+
|
|
174
|
+
if clicks >= 50 && bounce_rate > 60.0
|
|
175
|
+
fixes << "💰 HIGH TRAFFIC BLEED: Priority CRO target. Run A/B test on hero headline and primary action."
|
|
176
|
+
end
|
|
177
|
+
|
|
178
|
+
if position >= 7.0 && bounce_rate <= 40.0
|
|
179
|
+
fixes << "⭐ HIDDEN HIGH CONVERTER: Searchers love this page (low bounce). Build 3 internal links to push from pos #{position} to Top 3."
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
fixes << "✅ HEALTHY ENGAGEMENT: Maintain current content structure and monitor core queries." if fixes.empty?
|
|
183
|
+
fixes
|
|
184
|
+
end
|
|
185
|
+
|
|
186
|
+
def estimate_bounce_rate(clicks, position)
|
|
187
|
+
# Fallback heuristic when GA4 is not linked:
|
|
188
|
+
# Informational / deeper SERP positions generally experience higher bounce rates
|
|
189
|
+
base = 50.0
|
|
190
|
+
base += 5.0 if position > 5.0
|
|
191
|
+
base += 8.0 if position > 10.0
|
|
192
|
+
base.clamp(35.0, 80.0)
|
|
193
|
+
end
|
|
194
|
+
end
|
|
195
|
+
end
|
|
@@ -0,0 +1,425 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'uri'
|
|
5
|
+
require 'net/http'
|
|
6
|
+
require 'time'
|
|
7
|
+
require 'fileutils'
|
|
8
|
+
require_relative 'sitemap_loader'
|
|
9
|
+
require_relative 'page_analyzer'
|
|
10
|
+
|
|
11
|
+
module GSC
|
|
12
|
+
class LlmsGenerator
|
|
13
|
+
attr_reader :base_url, :options
|
|
14
|
+
|
|
15
|
+
def initialize(base_url, options = {})
|
|
16
|
+
@base_url = base_url.to_s.strip
|
|
17
|
+
@base_url = "https://#{@base_url}" unless @base_url =~ %r{^https?://}
|
|
18
|
+
@options = options || {}
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
# Generate standard /llms.txt index file per specification
|
|
22
|
+
def generate_llms_txt(title: nil, summary: nil, limit: 25)
|
|
23
|
+
pages_meta = collect_pages_metadata(limit: limit)
|
|
24
|
+
|
|
25
|
+
site_title = title || URI.parse(@base_url).host.sub(/^www\./, '').capitalize
|
|
26
|
+
site_summary = summary || "Official developer documentation, guides, and knowledge base for #{site_title}."
|
|
27
|
+
|
|
28
|
+
out = []
|
|
29
|
+
out << "# #{site_title}"
|
|
30
|
+
out << ""
|
|
31
|
+
out << "> #{site_summary}"
|
|
32
|
+
out << ""
|
|
33
|
+
out << "## Core Documentation"
|
|
34
|
+
out << ""
|
|
35
|
+
|
|
36
|
+
core_pages = pages_meta.first(12)
|
|
37
|
+
core_pages.each do |pm|
|
|
38
|
+
out << "- [#{pm[:title]}](#{pm[:url]}): #{pm[:description]}"
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
optional_pages = pages_meta[12..-1] || []
|
|
42
|
+
if optional_pages.any?
|
|
43
|
+
out << ""
|
|
44
|
+
out << "## Extended Guides & Articles"
|
|
45
|
+
out << ""
|
|
46
|
+
optional_pages.each do |pm|
|
|
47
|
+
out << "- [#{pm[:title]}](#{pm[:url]}): #{pm[:description]}"
|
|
48
|
+
end
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
out << ""
|
|
52
|
+
out << "## Optional"
|
|
53
|
+
out << ""
|
|
54
|
+
out << "- [Full Consolidated Knowledge Base](#{@base_url}/llms-full.txt): Complete multi-page markdown knowledge bundle for LLM context ingestion."
|
|
55
|
+
out << ""
|
|
56
|
+
|
|
57
|
+
out.join("\n")
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
# Generate complete /llms-full.txt multi-page consolidated knowledge bundle
|
|
61
|
+
def generate_llms_full_txt(title: nil, summary: nil, limit: 25, gsc_rows: nil)
|
|
62
|
+
pages_meta = collect_pages_metadata(limit: limit)
|
|
63
|
+
site_title = title || URI.parse(@base_url).host.sub(/^www\./, '').capitalize
|
|
64
|
+
site_summary = summary || "Comprehensive knowledge base and technical specifications for #{site_title}."
|
|
65
|
+
|
|
66
|
+
toc = []
|
|
67
|
+
docs_content = []
|
|
68
|
+
|
|
69
|
+
pages_meta.each_with_index do |pm, idx|
|
|
70
|
+
section_id = "section-#{idx + 1}"
|
|
71
|
+
toc << "#{idx + 1}. [#{pm[:title]}](##{section_id})"
|
|
72
|
+
|
|
73
|
+
html = fetch_html(pm[:url])
|
|
74
|
+
md_body = html_to_clean_markdown(html, pm[:url])
|
|
75
|
+
|
|
76
|
+
# FAQ injection from GSC queries if available
|
|
77
|
+
faq_block = generate_gsc_faq_block(pm[:url], gsc_rows)
|
|
78
|
+
|
|
79
|
+
doc_section = []
|
|
80
|
+
doc_section << "<a name=\"#{section_id}\"></a>"
|
|
81
|
+
doc_section << "# #{pm[:title]}"
|
|
82
|
+
doc_section << ""
|
|
83
|
+
doc_section << "**Source URL**: #{pm[:url]}"
|
|
84
|
+
doc_section << "**Description**: #{pm[:description]}"
|
|
85
|
+
doc_section << ""
|
|
86
|
+
doc_section << md_body
|
|
87
|
+
doc_section << ""
|
|
88
|
+
doc_section << faq_block if faq_block
|
|
89
|
+
doc_section << ""
|
|
90
|
+
doc_section << "---"
|
|
91
|
+
doc_section << ""
|
|
92
|
+
|
|
93
|
+
docs_content << doc_section.join("\n")
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
out = []
|
|
97
|
+
out << "# #{site_title} — Full Knowledge Base (`/llms-full.txt`)"
|
|
98
|
+
out << ""
|
|
99
|
+
out << "> #{site_summary}"
|
|
100
|
+
out << "> Generated automatically by Google Search Console CLI (`gsc llms --full`) on #{Time.now.strftime('%Y-%m-%d')}."
|
|
101
|
+
out << ""
|
|
102
|
+
out << "## Table of Contents"
|
|
103
|
+
out << ""
|
|
104
|
+
out << toc.join("\n")
|
|
105
|
+
out << ""
|
|
106
|
+
out << "---"
|
|
107
|
+
out << ""
|
|
108
|
+
out << docs_content.join("\n")
|
|
109
|
+
|
|
110
|
+
out.join("\n")
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
# Package complete bundle (both llms.txt and llms-full.txt) with token analytics
|
|
114
|
+
def package_agent_bundle(output_dir: nil, limit: 25, gsc_rows: nil)
|
|
115
|
+
llms_txt = generate_llms_txt(limit: limit)
|
|
116
|
+
llms_full_txt = generate_llms_full_txt(limit: limit, gsc_rows: gsc_rows)
|
|
117
|
+
|
|
118
|
+
# Token estimation (standard 4 chars per token)
|
|
119
|
+
txt_tokens = (llms_txt.length / 4.0).ceil
|
|
120
|
+
full_tokens = (llms_full_txt.length / 4.0).ceil
|
|
121
|
+
|
|
122
|
+
saved_files = []
|
|
123
|
+
if output_dir
|
|
124
|
+
FileUtils.mkdir_p(output_dir)
|
|
125
|
+
txt_path = File.join(output_dir, "llms.txt")
|
|
126
|
+
full_path = File.join(output_dir, "llms-full.txt")
|
|
127
|
+
|
|
128
|
+
File.write(txt_path, llms_txt, encoding: 'UTF-8')
|
|
129
|
+
File.write(full_path, llms_full_txt, encoding: 'UTF-8')
|
|
130
|
+
|
|
131
|
+
saved_files << { path: txt_path, size_bytes: File.size(txt_path) }
|
|
132
|
+
saved_files << { path: full_path, size_bytes: File.size(full_path) }
|
|
133
|
+
end
|
|
134
|
+
|
|
135
|
+
context_windows = {
|
|
136
|
+
claude_3_5_sonnet_200k: (full_tokens <= 200_000),
|
|
137
|
+
gpt_4o_128k: (full_tokens <= 128_000),
|
|
138
|
+
gemini_1_5_pro_1m: (full_tokens <= 1_000_000)
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
{
|
|
142
|
+
base_url: @base_url,
|
|
143
|
+
total_pages_packaged: limit,
|
|
144
|
+
llms_txt: {
|
|
145
|
+
chars: llms_txt.length,
|
|
146
|
+
tokens: txt_tokens,
|
|
147
|
+
content: llms_txt
|
|
148
|
+
},
|
|
149
|
+
llms_full_txt: {
|
|
150
|
+
chars: llms_full_txt.length,
|
|
151
|
+
tokens: full_tokens,
|
|
152
|
+
content: llms_full_txt
|
|
153
|
+
},
|
|
154
|
+
context_window_fit: context_windows,
|
|
155
|
+
saved_files: saved_files
|
|
156
|
+
}
|
|
157
|
+
end
|
|
158
|
+
|
|
159
|
+
# HTML to Clean Markdown Converter (Zero Gem Dependencies)
|
|
160
|
+
def html_to_clean_markdown(html, base_url = '')
|
|
161
|
+
return "" if html.nil? || html.empty?
|
|
162
|
+
|
|
163
|
+
# 1. Strip comments, scripts, styles, nav, footer, header, svg, noscript
|
|
164
|
+
clean = html.to_s.dup.force_encoding('UTF-8').scrub
|
|
165
|
+
clean.gsub!(/<!--.*?-->/m, '')
|
|
166
|
+
clean.gsub!(/<script\b[^>]*>.*?<\/script>/mi, '')
|
|
167
|
+
clean.gsub!(/<style\b[^>]*>.*?<\/style>/mi, '')
|
|
168
|
+
clean.gsub!(/<noscript\b[^>]*>.*?<\/noscript>/mi, '')
|
|
169
|
+
clean.gsub!(/<svg\b[^>]*>.*?<\/svg>/mi, '')
|
|
170
|
+
clean.gsub!(/<iframe\b[^>]*>.*?<\/iframe>/mi, '')
|
|
171
|
+
clean.gsub!(/<nav\b[^>]*>.*?<\/nav>/mi, '')
|
|
172
|
+
clean.gsub!(/<footer\b[^>]*>.*?<\/footer>/mi, '')
|
|
173
|
+
clean.gsub!(/<header\b[^>]*>.*?<\/header>/mi, '')
|
|
174
|
+
clean.gsub!(/<aside\b[^>]*>.*?<\/aside>/mi, '')
|
|
175
|
+
|
|
176
|
+
# 2. Convert Headings (h1 - h6)
|
|
177
|
+
(1..6).each do |level|
|
|
178
|
+
clean.gsub!(/<h#{level}\b[^>]*>(.*?)<\/h#{level}>/mi) do
|
|
179
|
+
heading_text = strip_tags($1).strip
|
|
180
|
+
"\n\n#{'#' * level} #{heading_text}\n\n"
|
|
181
|
+
end
|
|
182
|
+
end
|
|
183
|
+
|
|
184
|
+
# 3. Convert Tables to GFM Markdown
|
|
185
|
+
clean.gsub!(/<table\b[^>]*>(.*?)<\/table>/mi) do
|
|
186
|
+
table_html = $1
|
|
187
|
+
convert_table_to_markdown(table_html)
|
|
188
|
+
end
|
|
189
|
+
|
|
190
|
+
# 4. Convert Lists
|
|
191
|
+
clean.gsub!(/<li\b[^>]*>(.*?)<\/li>/mi) do
|
|
192
|
+
li_text = strip_tags($1).strip
|
|
193
|
+
"\n- #{li_text}"
|
|
194
|
+
end
|
|
195
|
+
clean.gsub!(/<\/?(ul|ol)\b[^>]*>/mi, "\n")
|
|
196
|
+
|
|
197
|
+
# 5. Convert Code blocks
|
|
198
|
+
clean.gsub!(/<pre\b[^>]*><code\b[^>]*>(.*?)<\/code><\/pre>/mi) do
|
|
199
|
+
code = decode_html_entities($1).strip
|
|
200
|
+
"\n```\n#{code}\n```\n"
|
|
201
|
+
end
|
|
202
|
+
clean.gsub!(/<code\b[^>]*>(.*?)<\/code>/mi) do
|
|
203
|
+
code = decode_html_entities($1).strip
|
|
204
|
+
"`#{code}`"
|
|
205
|
+
end
|
|
206
|
+
|
|
207
|
+
# 6. Convert Bold and Italic
|
|
208
|
+
clean.gsub!(/<(b|strong)\b[^>]*>(.*?)<\/\1>/mi) { "**#{strip_tags($2).strip}**" }
|
|
209
|
+
clean.gsub!(/<(i|em)\b[^>]*>(.*?)<\/\1>/mi) { "*#{strip_tags($2).strip}*" }
|
|
210
|
+
|
|
211
|
+
# 7. Convert Links
|
|
212
|
+
clean.gsub!(/<a\b[^>]*href=["']([^"']+)["'][^>]*>(.*?)<\/a>/mi) do
|
|
213
|
+
href = $1.strip
|
|
214
|
+
text = strip_tags($2).strip
|
|
215
|
+
next text if text.empty?
|
|
216
|
+
resolved_href = URI.join(base_url, href).to_s rescue href
|
|
217
|
+
safe_text = text.gsub('[', '\[').gsub(']', '\]')
|
|
218
|
+
"[#{safe_text}](#{resolved_href})"
|
|
219
|
+
end
|
|
220
|
+
|
|
221
|
+
# 8. Convert Paragraphs and line breaks
|
|
222
|
+
clean.gsub!(/<br\s*\/?>/mi, " \n")
|
|
223
|
+
clean.gsub!(/<p\b[^>]*>(.*?)<\/p>/mi) { "\n\n#{strip_tags($1).strip}\n\n" }
|
|
224
|
+
|
|
225
|
+
# 9. Strip any remaining HTML tags
|
|
226
|
+
clean = strip_tags(clean)
|
|
227
|
+
|
|
228
|
+
# 10. Decode HTML Entities
|
|
229
|
+
clean = decode_html_entities(clean)
|
|
230
|
+
|
|
231
|
+
# 11. Normalize excessive whitespace
|
|
232
|
+
clean.gsub!(/\r\n/, "\n")
|
|
233
|
+
clean.gsub!(/\n{3,}/, "\n\n")
|
|
234
|
+
clean.strip
|
|
235
|
+
end
|
|
236
|
+
|
|
237
|
+
def audit_ai_readability(url)
|
|
238
|
+
pa = GSC::PageAnalyzer.new(url)
|
|
239
|
+
data = pa.fetch_and_analyze
|
|
240
|
+
|
|
241
|
+
html = pa.html || ''
|
|
242
|
+
has_tables = html.include?('<table')
|
|
243
|
+
has_lists = html.include?('<ul') || html.include?('<ol')
|
|
244
|
+
schemas = data.dig(:structured_data, :schemas) || []
|
|
245
|
+
h1_count = (data.dig(:headings, :h1) || []).length
|
|
246
|
+
|
|
247
|
+
score = 100
|
|
248
|
+
issues = []
|
|
249
|
+
|
|
250
|
+
if h1_count != 1
|
|
251
|
+
score -= 20
|
|
252
|
+
issues << "H1 count is #{h1_count} (Must be exactly 1 for clean LLM hierarchy)"
|
|
253
|
+
end
|
|
254
|
+
|
|
255
|
+
unless has_tables
|
|
256
|
+
score -= 15
|
|
257
|
+
issues << "No <table> found (tables increase LLM citation and fact extraction by 3x)"
|
|
258
|
+
end
|
|
259
|
+
|
|
260
|
+
unless has_lists
|
|
261
|
+
score -= 15
|
|
262
|
+
issues << "No bullet lists (<ul> or <ol>) found for quick entity consumption"
|
|
263
|
+
end
|
|
264
|
+
|
|
265
|
+
if schemas.empty?
|
|
266
|
+
score -= 20
|
|
267
|
+
issues << "No JSON-LD schemas detected (structured data accelerates AI knowledge graph inclusion)"
|
|
268
|
+
end
|
|
269
|
+
|
|
270
|
+
{
|
|
271
|
+
url: url,
|
|
272
|
+
ai_readability_score: [score, 0].max,
|
|
273
|
+
grade: score >= 80 ? 'A (Excellent)' : (score >= 60 ? 'B (Acceptable)' : 'C (Needs Work)'),
|
|
274
|
+
issues: issues,
|
|
275
|
+
features: {
|
|
276
|
+
has_tables: has_tables,
|
|
277
|
+
has_lists: has_lists,
|
|
278
|
+
schemas_found: schemas.length,
|
|
279
|
+
h1_count: h1_count
|
|
280
|
+
}
|
|
281
|
+
}
|
|
282
|
+
end
|
|
283
|
+
|
|
284
|
+
private
|
|
285
|
+
|
|
286
|
+
def collect_pages_metadata(limit: 25)
|
|
287
|
+
sitemap_target = "#{@base_url.chomp('/')}/sitemap.xml"
|
|
288
|
+
urls = (GSC::SitemapLoader.resolve_urls(sitemap_target, @base_url, quiet: true) rescue [@base_url]) || [@base_url]
|
|
289
|
+
urls = [@base_url] if urls.empty?
|
|
290
|
+
|
|
291
|
+
# Also add standard routes if sitemap is small
|
|
292
|
+
default_routes = ["/about", "/features", "/pricing", "/faq", "/docs", "/contact"].map do |r|
|
|
293
|
+
URI.join(@base_url, r).to_s rescue nil
|
|
294
|
+
end.compact
|
|
295
|
+
|
|
296
|
+
candidate_urls = (urls + default_routes).uniq.first(limit)
|
|
297
|
+
metadata = []
|
|
298
|
+
|
|
299
|
+
candidate_urls.each do |url|
|
|
300
|
+
html = fetch_html(url)
|
|
301
|
+
next unless html && !html.empty?
|
|
302
|
+
|
|
303
|
+
title = extract_title(html) || url
|
|
304
|
+
desc = extract_description(html) || "Guide and specifications for #{title}."
|
|
305
|
+
|
|
306
|
+
metadata << {
|
|
307
|
+
url: url,
|
|
308
|
+
title: title,
|
|
309
|
+
description: desc
|
|
310
|
+
}
|
|
311
|
+
end
|
|
312
|
+
|
|
313
|
+
metadata.empty? ? [{ url: @base_url, title: "Home", description: "Home page" }] : metadata
|
|
314
|
+
end
|
|
315
|
+
|
|
316
|
+
def fetch_html(url)
|
|
317
|
+
uri = URI.parse(url) rescue nil
|
|
318
|
+
return "" unless uri && uri.host
|
|
319
|
+
|
|
320
|
+
begin
|
|
321
|
+
Net::HTTP.start(uri.hostname, uri.port, use_ssl: (uri.scheme == 'https'), open_timeout: 5, read_timeout: 8) do |http|
|
|
322
|
+
req = Net::HTTP::Get.new(uri.request_uri)
|
|
323
|
+
req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) gsc-cli/2.1'
|
|
324
|
+
req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
|
|
325
|
+
res = http.request(req)
|
|
326
|
+
res.body.to_s[0..60_000] if res.body
|
|
327
|
+
end
|
|
328
|
+
rescue StandardError
|
|
329
|
+
""
|
|
330
|
+
end
|
|
331
|
+
end
|
|
332
|
+
|
|
333
|
+
def extract_title(html)
|
|
334
|
+
if html =~ /<title\b[^>]*>(.*?)<\/title>/mi
|
|
335
|
+
strip_tags(decode_html_entities($1)).strip
|
|
336
|
+
end
|
|
337
|
+
end
|
|
338
|
+
|
|
339
|
+
def extract_description(html)
|
|
340
|
+
if html =~ /<meta\b[^>]*name=["']description["'][^>]*content=["']([^"']+)["']/mi
|
|
341
|
+
strip_tags(decode_html_entities($1)).strip
|
|
342
|
+
elsif html =~ /<meta\b[^>]*content=["']([^"']+)["'][^>]*name=["']description["']/mi
|
|
343
|
+
strip_tags(decode_html_entities($1)).strip
|
|
344
|
+
end
|
|
345
|
+
end
|
|
346
|
+
|
|
347
|
+
def convert_table_to_markdown(table_html)
|
|
348
|
+
rows = []
|
|
349
|
+
table_html.scan(/<tr\b[^>]*>(.*?)<\/tr>/mi).each do |tr_match|
|
|
350
|
+
row_content = tr_match.first
|
|
351
|
+
cells = []
|
|
352
|
+
row_content.scan(/<(th|td)\b[^>]*>(.*?)<\/\1>/mi).each do |_tag, content|
|
|
353
|
+
cells << strip_tags(content).gsub('|', '\\|').strip
|
|
354
|
+
end
|
|
355
|
+
rows << cells unless cells.empty?
|
|
356
|
+
end
|
|
357
|
+
|
|
358
|
+
return "" if rows.empty?
|
|
359
|
+
|
|
360
|
+
col_count = rows.map(&:size).max
|
|
361
|
+
rows.each { |r| r.fill('', r.size...col_count) }
|
|
362
|
+
|
|
363
|
+
md = []
|
|
364
|
+
# Header
|
|
365
|
+
header = rows.first
|
|
366
|
+
md << "| #{header.join(' | ')} |"
|
|
367
|
+
md << "| #{Array.new(col_count, '---').join(' | ')} |"
|
|
368
|
+
|
|
369
|
+
# Body
|
|
370
|
+
rows[1..-1].each do |r|
|
|
371
|
+
md << "| #{r.join(' | ')} |"
|
|
372
|
+
end
|
|
373
|
+
|
|
374
|
+
"\n\n" + md.join("\n") + "\n\n"
|
|
375
|
+
end
|
|
376
|
+
|
|
377
|
+
def generate_gsc_faq_block(url, gsc_rows)
|
|
378
|
+
return nil unless gsc_rows && gsc_rows.is_a?(Array)
|
|
379
|
+
|
|
380
|
+
clean_target = url.downcase.sub(%r{/$}, '')
|
|
381
|
+
matching_queries = []
|
|
382
|
+
|
|
383
|
+
gsc_rows.each do |r|
|
|
384
|
+
keys = r['keys'] || []
|
|
385
|
+
page = keys.find { |k| k =~ %r{^https?://} }
|
|
386
|
+
next unless page && page.downcase.sub(%r{/$}, '') == clean_target
|
|
387
|
+
|
|
388
|
+
q = keys.find { |k| k !~ %r{^https?://} }
|
|
389
|
+
clicks = (r['clicks'] || 0).to_i
|
|
390
|
+
matching_queries << { query: q, clicks: clicks } if q
|
|
391
|
+
end
|
|
392
|
+
|
|
393
|
+
return nil if matching_queries.empty?
|
|
394
|
+
|
|
395
|
+
top_q = matching_queries.sort_by { |item| -item[:clicks] }.first(5)
|
|
396
|
+
faq = []
|
|
397
|
+
faq << "### Search Questions & High-Intent Queries (Verified from Google Search Console)"
|
|
398
|
+
faq << ""
|
|
399
|
+
top_q.each do |item|
|
|
400
|
+
faq << "- **Q: How does #{item[:query]} work?**"
|
|
401
|
+
faq << " *A: Detailed in the core specifications above.*"
|
|
402
|
+
end
|
|
403
|
+
faq.join("\n")
|
|
404
|
+
end
|
|
405
|
+
|
|
406
|
+
def strip_tags(text)
|
|
407
|
+
text.to_s.gsub(/<[^>]+>/, ' ')
|
|
408
|
+
end
|
|
409
|
+
|
|
410
|
+
def decode_html_entities(text)
|
|
411
|
+
t = text.to_s.dup
|
|
412
|
+
t.gsub!('&', '&')
|
|
413
|
+
t.gsub!('"', '"')
|
|
414
|
+
t.gsub!(''', "'")
|
|
415
|
+
t.gsub!(''', "'")
|
|
416
|
+
t.gsub!('<', '<')
|
|
417
|
+
t.gsub!('>', '>')
|
|
418
|
+
t.gsub!(' ', ' ')
|
|
419
|
+
t.gsub!('’', "'")
|
|
420
|
+
t.gsub!('“', '"')
|
|
421
|
+
t.gsub!('”', '"')
|
|
422
|
+
t
|
|
423
|
+
end
|
|
424
|
+
end
|
|
425
|
+
end
|