gsc-cli 2.1.0 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +410 -414
- data/bin/gsc +28067 -5661
- data/dist/gsc +29121 -5046
- data/lib/gsc/aio_hunter.rb +343 -0
- data/lib/gsc/answer_synthesizer.rb +157 -0
- data/lib/gsc/api.rb +53 -1
- data/lib/gsc/auth.rb +26 -0
- data/lib/gsc/brand_segmenter.rb +140 -0
- data/lib/gsc/cache_manager.rb +806 -0
- data/lib/gsc/cannibalization_analyzer.rb +141 -0
- data/lib/gsc/canonical_chains.rb +367 -0
- data/lib/gsc/citation_simulator.rb +339 -0
- data/lib/gsc/cli/aio_hunter.rb +154 -0
- data/lib/gsc/cli/analytics.rb +788 -0
- data/lib/gsc/cli/audit.rb +1976 -0
- data/lib/gsc/cli/base.rb +384 -0
- data/lib/gsc/cli/cache.rb +266 -0
- data/lib/gsc/cli/canonical.rb +223 -0
- data/lib/gsc/cli/citation_simulator.rb +152 -0
- data/lib/gsc/cli/dashboard.rb +354 -0
- data/lib/gsc/cli/doctor.rb +129 -0
- data/lib/gsc/cli/eeat.rb +125 -0
- data/lib/gsc/cli/ga4.rb +852 -0
- data/lib/gsc/cli/growth.rb +650 -0
- data/lib/gsc/cli/hreflang.rb +164 -0
- data/lib/gsc/cli/image_seo.rb +162 -0
- data/lib/gsc/cli/indexing.rb +458 -0
- data/lib/gsc/cli/intent_shift.rb +125 -0
- data/lib/gsc/cli/keyword_value.rb +134 -0
- data/lib/gsc/cli/keywords.rb +795 -0
- data/lib/gsc/cli/landing_roi.rb +308 -0
- data/lib/gsc/cli/low_ctr.rb +213 -0
- data/lib/gsc/cli/mobile_parity.rb +150 -0
- data/lib/gsc/cli/report.rb +100 -0
- data/lib/gsc/cli/rich_results.rb +172 -0
- data/lib/gsc/cli/schema_generate.rb +149 -0
- data/lib/gsc/cli/seasonal.rb +232 -0
- data/lib/gsc/cli/security.rb +153 -0
- data/lib/gsc/cli/setup.rb +1291 -0
- data/lib/gsc/cli/sitemap_tree.rb +143 -0
- data/lib/gsc/cli/skill_pack.rb +62 -0
- data/lib/gsc/cli/soft_404.rb +199 -0
- data/lib/gsc/cli/sparkline.rb +227 -0
- data/lib/gsc/cli/watchdog.rb +150 -0
- data/lib/gsc/cli/zombie_purger.rb +208 -0
- data/lib/gsc/cli.rb +707 -5265
- data/lib/gsc/cli_advanced.rb +987 -44
- data/lib/gsc/client.rb +17 -2
- data/lib/gsc/color.rb +16 -1
- data/lib/gsc/command_registry.rb +47 -9
- data/lib/gsc/config.rb +2 -2
- data/lib/gsc/ctr_curve.rb +115 -0
- data/lib/gsc/decay_predictor.rb +322 -0
- data/lib/gsc/doctor.rb +434 -0
- data/lib/gsc/eeat_auditor.rb +428 -0
- data/lib/gsc/entity_auditor.rb +229 -0
- data/lib/gsc/firewall_scanner.rb +733 -0
- data/lib/gsc/geo_auditor.rb +368 -0
- data/lib/gsc/google_trends.rb +8 -1
- data/lib/gsc/heading_validator.rb +283 -0
- data/lib/gsc/hreflang_validator.rb +412 -0
- data/lib/gsc/image_seo.rb +286 -0
- data/lib/gsc/indexing_queue.rb +179 -0
- data/lib/gsc/indexnow.rb +93 -0
- data/lib/gsc/intent_shift.rb +188 -0
- data/lib/gsc/internal_links.rb +153 -36
- data/lib/gsc/keyword_value.rb +191 -0
- data/lib/gsc/landing_roi.rb +195 -0
- data/lib/gsc/llms_generator.rb +343 -22
- data/lib/gsc/low_ctr_rewriter.rb +408 -0
- data/lib/gsc/mobile_parity.rb +222 -0
- data/lib/gsc/network_tracer.rb +8 -1
- data/lib/gsc/page_analyzer.rb +47 -7
- data/lib/gsc/prompts.rb +38 -29
- data/lib/gsc/questions_harvester.rb +178 -0
- data/lib/gsc/report_generator.rb +461 -0
- data/lib/gsc/rich_results.rb +388 -0
- data/lib/gsc/robots_checker.rb +46 -15
- data/lib/gsc/schema_generator.rb +788 -0
- data/lib/gsc/schema_validator.rb +36 -38
- data/lib/gsc/seasonal_predictor.rb +381 -0
- data/lib/gsc/security_scanner.rb +496 -0
- data/lib/gsc/serp_feature_detector.rb +359 -0
- data/lib/gsc/serp_preview.rb +108 -22
- data/lib/gsc/site_crawler.rb +113 -21
- data/lib/gsc/sitemap_loader.rb +15 -4
- data/lib/gsc/sitemap_tree.rb +301 -0
- data/lib/gsc/skill_pack.rb +195 -0
- data/lib/gsc/soft_404_analyzer.rb +385 -0
- data/lib/gsc/sparkline.rb +171 -0
- data/lib/gsc/speed_correlator.rb +416 -0
- data/lib/gsc/striking_playbook.rb +190 -0
- data/lib/gsc/title_optimizer.rb +420 -0
- data/lib/gsc/vault.rb +260 -0
- data/lib/gsc/version.rb +1 -1
- data/lib/gsc/watchdog.rb +235 -0
- data/lib/gsc/zombie_purger.rb +366 -0
- data/lib/gsc.rb +118 -0
- metadata +75 -1
data/lib/gsc/llms_generator.rb
CHANGED
|
@@ -2,23 +2,28 @@
|
|
|
2
2
|
# frozen_string_literal: true
|
|
3
3
|
|
|
4
4
|
require 'uri'
|
|
5
|
+
require 'net/http'
|
|
6
|
+
require 'time'
|
|
7
|
+
require 'fileutils'
|
|
8
|
+
require_relative 'sitemap_loader'
|
|
9
|
+
require_relative 'page_analyzer'
|
|
5
10
|
|
|
6
11
|
module GSC
|
|
7
12
|
class LlmsGenerator
|
|
8
|
-
attr_reader :base_url, :
|
|
13
|
+
attr_reader :base_url, :options
|
|
9
14
|
|
|
10
|
-
def initialize(base_url)
|
|
15
|
+
def initialize(base_url, options = {})
|
|
11
16
|
@base_url = base_url.to_s.strip
|
|
12
17
|
@base_url = "https://#{@base_url}" unless @base_url =~ %r{^https?://}
|
|
18
|
+
@options = options || {}
|
|
13
19
|
end
|
|
14
20
|
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
urls = [@base_url] if urls.empty?
|
|
21
|
+
# Generate standard /llms.txt index file per specification
|
|
22
|
+
def generate_llms_txt(title: nil, summary: nil, limit: 25)
|
|
23
|
+
pages_meta = collect_pages_metadata(limit: limit)
|
|
19
24
|
|
|
20
25
|
site_title = title || URI.parse(@base_url).host.sub(/^www\./, '').capitalize
|
|
21
|
-
site_summary = summary || "Official documentation and
|
|
26
|
+
site_summary = summary || "Official developer documentation, guides, and knowledge base for #{site_title}."
|
|
22
27
|
|
|
23
28
|
out = []
|
|
24
29
|
out << "# #{site_title}"
|
|
@@ -28,36 +33,211 @@ module GSC
|
|
|
28
33
|
out << "## Core Documentation"
|
|
29
34
|
out << ""
|
|
30
35
|
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
dom = pa.analyze_dom rescue next
|
|
36
|
-
|
|
37
|
-
page_title = dom.dig(:title, :text) || url
|
|
38
|
-
page_desc = dom.dig(:meta_description, :text) || "Documentation page."
|
|
36
|
+
core_pages = pages_meta.first(12)
|
|
37
|
+
core_pages.each do |pm|
|
|
38
|
+
out << "- [#{pm[:title]}](#{pm[:url]}): #{pm[:description]}"
|
|
39
|
+
end
|
|
39
40
|
|
|
40
|
-
|
|
41
|
+
optional_pages = pages_meta[12..-1] || []
|
|
42
|
+
if optional_pages.any?
|
|
43
|
+
out << ""
|
|
44
|
+
out << "## Extended Guides & Articles"
|
|
45
|
+
out << ""
|
|
46
|
+
optional_pages.each do |pm|
|
|
47
|
+
out << "- [#{pm[:title]}](#{pm[:url]}): #{pm[:description]}"
|
|
48
|
+
end
|
|
41
49
|
end
|
|
42
50
|
|
|
43
51
|
out << ""
|
|
44
52
|
out << "## Optional"
|
|
45
53
|
out << ""
|
|
46
|
-
out << "- [Full
|
|
54
|
+
out << "- [Full Consolidated Knowledge Base](#{@base_url}/llms-full.txt): Complete multi-page markdown knowledge bundle for LLM context ingestion."
|
|
47
55
|
out << ""
|
|
48
56
|
|
|
49
57
|
out.join("\n")
|
|
50
58
|
end
|
|
51
59
|
|
|
60
|
+
# Generate complete /llms-full.txt multi-page consolidated knowledge bundle
|
|
61
|
+
def generate_llms_full_txt(title: nil, summary: nil, limit: 25, gsc_rows: nil)
|
|
62
|
+
pages_meta = collect_pages_metadata(limit: limit)
|
|
63
|
+
site_title = title || URI.parse(@base_url).host.sub(/^www\./, '').capitalize
|
|
64
|
+
site_summary = summary || "Comprehensive knowledge base and technical specifications for #{site_title}."
|
|
65
|
+
|
|
66
|
+
toc = []
|
|
67
|
+
docs_content = []
|
|
68
|
+
|
|
69
|
+
pages_meta.each_with_index do |pm, idx|
|
|
70
|
+
section_id = "section-#{idx + 1}"
|
|
71
|
+
toc << "#{idx + 1}. [#{pm[:title]}](##{section_id})"
|
|
72
|
+
|
|
73
|
+
html = fetch_html(pm[:url])
|
|
74
|
+
md_body = html_to_clean_markdown(html, pm[:url])
|
|
75
|
+
|
|
76
|
+
# FAQ injection from GSC queries if available
|
|
77
|
+
faq_block = generate_gsc_faq_block(pm[:url], gsc_rows)
|
|
78
|
+
|
|
79
|
+
doc_section = []
|
|
80
|
+
doc_section << "<a name=\"#{section_id}\"></a>"
|
|
81
|
+
doc_section << "# #{pm[:title]}"
|
|
82
|
+
doc_section << ""
|
|
83
|
+
doc_section << "**Source URL**: #{pm[:url]}"
|
|
84
|
+
doc_section << "**Description**: #{pm[:description]}"
|
|
85
|
+
doc_section << ""
|
|
86
|
+
doc_section << md_body
|
|
87
|
+
doc_section << ""
|
|
88
|
+
doc_section << faq_block if faq_block
|
|
89
|
+
doc_section << ""
|
|
90
|
+
doc_section << "---"
|
|
91
|
+
doc_section << ""
|
|
92
|
+
|
|
93
|
+
docs_content << doc_section.join("\n")
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
out = []
|
|
97
|
+
out << "# #{site_title} — Full Knowledge Base (`/llms-full.txt`)"
|
|
98
|
+
out << ""
|
|
99
|
+
out << "> #{site_summary}"
|
|
100
|
+
out << "> Generated automatically by Google Search Console CLI (`gsc llms --full`) on #{Time.now.strftime('%Y-%m-%d')}."
|
|
101
|
+
out << ""
|
|
102
|
+
out << "## Table of Contents"
|
|
103
|
+
out << ""
|
|
104
|
+
out << toc.join("\n")
|
|
105
|
+
out << ""
|
|
106
|
+
out << "---"
|
|
107
|
+
out << ""
|
|
108
|
+
out << docs_content.join("\n")
|
|
109
|
+
|
|
110
|
+
out.join("\n")
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
# Package complete bundle (both llms.txt and llms-full.txt) with token analytics
|
|
114
|
+
def package_agent_bundle(output_dir: nil, limit: 25, gsc_rows: nil)
|
|
115
|
+
llms_txt = generate_llms_txt(limit: limit)
|
|
116
|
+
llms_full_txt = generate_llms_full_txt(limit: limit, gsc_rows: gsc_rows)
|
|
117
|
+
|
|
118
|
+
# Token estimation (standard 4 chars per token)
|
|
119
|
+
txt_tokens = (llms_txt.length / 4.0).ceil
|
|
120
|
+
full_tokens = (llms_full_txt.length / 4.0).ceil
|
|
121
|
+
|
|
122
|
+
saved_files = []
|
|
123
|
+
if output_dir
|
|
124
|
+
FileUtils.mkdir_p(output_dir)
|
|
125
|
+
txt_path = File.join(output_dir, "llms.txt")
|
|
126
|
+
full_path = File.join(output_dir, "llms-full.txt")
|
|
127
|
+
|
|
128
|
+
File.write(txt_path, llms_txt, encoding: 'UTF-8')
|
|
129
|
+
File.write(full_path, llms_full_txt, encoding: 'UTF-8')
|
|
130
|
+
|
|
131
|
+
saved_files << { path: txt_path, size_bytes: File.size(txt_path) }
|
|
132
|
+
saved_files << { path: full_path, size_bytes: File.size(full_path) }
|
|
133
|
+
end
|
|
134
|
+
|
|
135
|
+
context_windows = {
|
|
136
|
+
claude_3_5_sonnet_200k: (full_tokens <= 200_000),
|
|
137
|
+
gpt_4o_128k: (full_tokens <= 128_000),
|
|
138
|
+
gemini_1_5_pro_1m: (full_tokens <= 1_000_000)
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
{
|
|
142
|
+
base_url: @base_url,
|
|
143
|
+
total_pages_packaged: limit,
|
|
144
|
+
llms_txt: {
|
|
145
|
+
chars: llms_txt.length,
|
|
146
|
+
tokens: txt_tokens,
|
|
147
|
+
content: llms_txt
|
|
148
|
+
},
|
|
149
|
+
llms_full_txt: {
|
|
150
|
+
chars: llms_full_txt.length,
|
|
151
|
+
tokens: full_tokens,
|
|
152
|
+
content: llms_full_txt
|
|
153
|
+
},
|
|
154
|
+
context_window_fit: context_windows,
|
|
155
|
+
saved_files: saved_files
|
|
156
|
+
}
|
|
157
|
+
end
|
|
158
|
+
|
|
159
|
+
# HTML to Clean Markdown Converter (Zero Gem Dependencies)
|
|
160
|
+
def html_to_clean_markdown(html, base_url = '')
|
|
161
|
+
return "" if html.nil? || html.empty?
|
|
162
|
+
|
|
163
|
+
# 1. Strip comments, scripts, styles, nav, footer, header, svg, noscript
|
|
164
|
+
clean = html.to_s.dup.force_encoding('UTF-8').scrub
|
|
165
|
+
clean.gsub!(/<!--.*?-->/m, '')
|
|
166
|
+
clean.gsub!(/<script\b[^>]*>.*?<\/script>/mi, '')
|
|
167
|
+
clean.gsub!(/<style\b[^>]*>.*?<\/style>/mi, '')
|
|
168
|
+
clean.gsub!(/<noscript\b[^>]*>.*?<\/noscript>/mi, '')
|
|
169
|
+
clean.gsub!(/<svg\b[^>]*>.*?<\/svg>/mi, '')
|
|
170
|
+
clean.gsub!(/<iframe\b[^>]*>.*?<\/iframe>/mi, '')
|
|
171
|
+
clean.gsub!(/<nav\b[^>]*>.*?<\/nav>/mi, '')
|
|
172
|
+
clean.gsub!(/<footer\b[^>]*>.*?<\/footer>/mi, '')
|
|
173
|
+
clean.gsub!(/<header\b[^>]*>.*?<\/header>/mi, '')
|
|
174
|
+
clean.gsub!(/<aside\b[^>]*>.*?<\/aside>/mi, '')
|
|
175
|
+
|
|
176
|
+
# 2. Convert Headings (h1 - h6)
|
|
177
|
+
(1..6).each do |level|
|
|
178
|
+
clean.gsub!(/<h#{level}\b[^>]*>(.*?)<\/h#{level}>/mi) do
|
|
179
|
+
heading_text = strip_tags($1).strip
|
|
180
|
+
"\n\n#{'#' * level} #{heading_text}\n\n"
|
|
181
|
+
end
|
|
182
|
+
end
|
|
183
|
+
|
|
184
|
+
# 3. Convert Tables to GFM Markdown
|
|
185
|
+
clean.gsub!(/<table\b[^>]*>(.*?)<\/table>/mi) do
|
|
186
|
+
table_html = $1
|
|
187
|
+
convert_table_to_markdown(table_html)
|
|
188
|
+
end
|
|
189
|
+
|
|
190
|
+
# 4. Convert Lists
|
|
191
|
+
clean.gsub!(/<li\b[^>]*>(.*?)<\/li>/mi) do
|
|
192
|
+
li_text = strip_tags($1).strip
|
|
193
|
+
"\n- #{li_text}"
|
|
194
|
+
end
|
|
195
|
+
clean.gsub!(/<\/?(ul|ol)\b[^>]*>/mi, "\n")
|
|
196
|
+
|
|
197
|
+
# 5. Convert Code blocks
|
|
198
|
+
clean.gsub!(/<pre\b[^>]*><code\b[^>]*>(.*?)<\/code><\/pre>/mi) do
|
|
199
|
+
code = decode_html_entities($1).strip
|
|
200
|
+
"\n```\n#{code}\n```\n"
|
|
201
|
+
end
|
|
202
|
+
clean.gsub!(/<code\b[^>]*>(.*?)<\/code>/mi) do
|
|
203
|
+
code = decode_html_entities($1).strip
|
|
204
|
+
"`#{code}`"
|
|
205
|
+
end
|
|
206
|
+
|
|
207
|
+
# 6. Convert Bold and Italic
|
|
208
|
+
clean.gsub!(/<(b|strong)\b[^>]*>(.*?)<\/\1>/mi) { "**#{strip_tags($2).strip}**" }
|
|
209
|
+
clean.gsub!(/<(i|em)\b[^>]*>(.*?)<\/\1>/mi) { "*#{strip_tags($2).strip}*" }
|
|
210
|
+
|
|
211
|
+
# 7. Convert Links
|
|
212
|
+
clean.gsub!(/<a\b[^>]*href=["']([^"']+)["'][^>]*>(.*?)<\/a>/mi) do
|
|
213
|
+
href = $1.strip
|
|
214
|
+
text = strip_tags($2).strip
|
|
215
|
+
next text if text.empty?
|
|
216
|
+
resolved_href = URI.join(base_url, href).to_s rescue href
|
|
217
|
+
safe_text = text.gsub('[', '\[').gsub(']', '\]')
|
|
218
|
+
"[#{safe_text}](#{resolved_href})"
|
|
219
|
+
end
|
|
220
|
+
|
|
221
|
+
# 8. Convert Paragraphs and line breaks
|
|
222
|
+
clean.gsub!(/<br\s*\/?>/mi, " \n")
|
|
223
|
+
clean.gsub!(/<p\b[^>]*>(.*?)<\/p>/mi) { "\n\n#{strip_tags($1).strip}\n\n" }
|
|
224
|
+
|
|
225
|
+
# 9. Strip any remaining HTML tags
|
|
226
|
+
clean = strip_tags(clean)
|
|
227
|
+
|
|
228
|
+
# 10. Decode HTML Entities
|
|
229
|
+
clean = decode_html_entities(clean)
|
|
230
|
+
|
|
231
|
+
# 11. Normalize excessive whitespace
|
|
232
|
+
clean.gsub!(/\r\n/, "\n")
|
|
233
|
+
clean.gsub!(/\n{3,}/, "\n\n")
|
|
234
|
+
clean.strip
|
|
235
|
+
end
|
|
236
|
+
|
|
52
237
|
def audit_ai_readability(url)
|
|
53
238
|
pa = GSC::PageAnalyzer.new(url)
|
|
54
239
|
data = pa.fetch_and_analyze
|
|
55
240
|
|
|
56
|
-
# Check criteria:
|
|
57
|
-
# 1. Clear H1 presence
|
|
58
|
-
# 2. Table presence (LLMs love tables)
|
|
59
|
-
# 3. Schema presence (structured data)
|
|
60
|
-
# 4. Definition / Bullet presence
|
|
61
241
|
html = pa.html || ''
|
|
62
242
|
has_tables = html.include?('<table')
|
|
63
243
|
has_lists = html.include?('<ul') || html.include?('<ol')
|
|
@@ -100,5 +280,146 @@ module GSC
|
|
|
100
280
|
}
|
|
101
281
|
}
|
|
102
282
|
end
|
|
283
|
+
|
|
284
|
+
private
|
|
285
|
+
|
|
286
|
+
def collect_pages_metadata(limit: 25)
|
|
287
|
+
sitemap_target = "#{@base_url.chomp('/')}/sitemap.xml"
|
|
288
|
+
urls = (GSC::SitemapLoader.resolve_urls(sitemap_target, @base_url, quiet: true) rescue [@base_url]) || [@base_url]
|
|
289
|
+
urls = [@base_url] if urls.empty?
|
|
290
|
+
|
|
291
|
+
# Also add standard routes if sitemap is small
|
|
292
|
+
default_routes = ["/about", "/features", "/pricing", "/faq", "/docs", "/contact"].map do |r|
|
|
293
|
+
URI.join(@base_url, r).to_s rescue nil
|
|
294
|
+
end.compact
|
|
295
|
+
|
|
296
|
+
candidate_urls = (urls + default_routes).uniq.first(limit)
|
|
297
|
+
metadata = []
|
|
298
|
+
|
|
299
|
+
candidate_urls.each do |url|
|
|
300
|
+
html = fetch_html(url)
|
|
301
|
+
next unless html && !html.empty?
|
|
302
|
+
|
|
303
|
+
title = extract_title(html) || url
|
|
304
|
+
desc = extract_description(html) || "Guide and specifications for #{title}."
|
|
305
|
+
|
|
306
|
+
metadata << {
|
|
307
|
+
url: url,
|
|
308
|
+
title: title,
|
|
309
|
+
description: desc
|
|
310
|
+
}
|
|
311
|
+
end
|
|
312
|
+
|
|
313
|
+
metadata.empty? ? [{ url: @base_url, title: "Home", description: "Home page" }] : metadata
|
|
314
|
+
end
|
|
315
|
+
|
|
316
|
+
def fetch_html(url)
|
|
317
|
+
uri = URI.parse(url) rescue nil
|
|
318
|
+
return "" unless uri && uri.host
|
|
319
|
+
|
|
320
|
+
begin
|
|
321
|
+
Net::HTTP.start(uri.hostname, uri.port, use_ssl: (uri.scheme == 'https'), open_timeout: 5, read_timeout: 8) do |http|
|
|
322
|
+
req = Net::HTTP::Get.new(uri.request_uri)
|
|
323
|
+
req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) gsc-cli/2.1'
|
|
324
|
+
req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
|
|
325
|
+
res = http.request(req)
|
|
326
|
+
res.body.to_s[0..60_000] if res.body
|
|
327
|
+
end
|
|
328
|
+
rescue StandardError
|
|
329
|
+
""
|
|
330
|
+
end
|
|
331
|
+
end
|
|
332
|
+
|
|
333
|
+
def extract_title(html)
|
|
334
|
+
if html =~ /<title\b[^>]*>(.*?)<\/title>/mi
|
|
335
|
+
strip_tags(decode_html_entities($1)).strip
|
|
336
|
+
end
|
|
337
|
+
end
|
|
338
|
+
|
|
339
|
+
def extract_description(html)
|
|
340
|
+
if html =~ /<meta\b[^>]*name=["']description["'][^>]*content=["']([^"']+)["']/mi
|
|
341
|
+
strip_tags(decode_html_entities($1)).strip
|
|
342
|
+
elsif html =~ /<meta\b[^>]*content=["']([^"']+)["'][^>]*name=["']description["']/mi
|
|
343
|
+
strip_tags(decode_html_entities($1)).strip
|
|
344
|
+
end
|
|
345
|
+
end
|
|
346
|
+
|
|
347
|
+
def convert_table_to_markdown(table_html)
|
|
348
|
+
rows = []
|
|
349
|
+
table_html.scan(/<tr\b[^>]*>(.*?)<\/tr>/mi).each do |tr_match|
|
|
350
|
+
row_content = tr_match.first
|
|
351
|
+
cells = []
|
|
352
|
+
row_content.scan(/<(th|td)\b[^>]*>(.*?)<\/\1>/mi).each do |_tag, content|
|
|
353
|
+
cells << strip_tags(content).gsub('|', '\\|').strip
|
|
354
|
+
end
|
|
355
|
+
rows << cells unless cells.empty?
|
|
356
|
+
end
|
|
357
|
+
|
|
358
|
+
return "" if rows.empty?
|
|
359
|
+
|
|
360
|
+
col_count = rows.map(&:size).max
|
|
361
|
+
rows.each { |r| r.fill('', r.size...col_count) }
|
|
362
|
+
|
|
363
|
+
md = []
|
|
364
|
+
# Header
|
|
365
|
+
header = rows.first
|
|
366
|
+
md << "| #{header.join(' | ')} |"
|
|
367
|
+
md << "| #{Array.new(col_count, '---').join(' | ')} |"
|
|
368
|
+
|
|
369
|
+
# Body
|
|
370
|
+
rows[1..-1].each do |r|
|
|
371
|
+
md << "| #{r.join(' | ')} |"
|
|
372
|
+
end
|
|
373
|
+
|
|
374
|
+
"\n\n" + md.join("\n") + "\n\n"
|
|
375
|
+
end
|
|
376
|
+
|
|
377
|
+
def generate_gsc_faq_block(url, gsc_rows)
|
|
378
|
+
return nil unless gsc_rows && gsc_rows.is_a?(Array)
|
|
379
|
+
|
|
380
|
+
clean_target = url.downcase.sub(%r{/$}, '')
|
|
381
|
+
matching_queries = []
|
|
382
|
+
|
|
383
|
+
gsc_rows.each do |r|
|
|
384
|
+
keys = r['keys'] || []
|
|
385
|
+
page = keys.find { |k| k =~ %r{^https?://} }
|
|
386
|
+
next unless page && page.downcase.sub(%r{/$}, '') == clean_target
|
|
387
|
+
|
|
388
|
+
q = keys.find { |k| k !~ %r{^https?://} }
|
|
389
|
+
clicks = (r['clicks'] || 0).to_i
|
|
390
|
+
matching_queries << { query: q, clicks: clicks } if q
|
|
391
|
+
end
|
|
392
|
+
|
|
393
|
+
return nil if matching_queries.empty?
|
|
394
|
+
|
|
395
|
+
top_q = matching_queries.sort_by { |item| -item[:clicks] }.first(5)
|
|
396
|
+
faq = []
|
|
397
|
+
faq << "### Search Questions & High-Intent Queries (Verified from Google Search Console)"
|
|
398
|
+
faq << ""
|
|
399
|
+
top_q.each do |item|
|
|
400
|
+
faq << "- **Q: How does #{item[:query]} work?**"
|
|
401
|
+
faq << " *A: Detailed in the core specifications above.*"
|
|
402
|
+
end
|
|
403
|
+
faq.join("\n")
|
|
404
|
+
end
|
|
405
|
+
|
|
406
|
+
def strip_tags(text)
|
|
407
|
+
text.to_s.gsub(/<[^>]+>/, ' ')
|
|
408
|
+
end
|
|
409
|
+
|
|
410
|
+
def decode_html_entities(text)
|
|
411
|
+
t = text.to_s.dup
|
|
412
|
+
t.gsub!('&', '&')
|
|
413
|
+
t.gsub!('"', '"')
|
|
414
|
+
t.gsub!(''', "'")
|
|
415
|
+
t.gsub!(''', "'")
|
|
416
|
+
t.gsub!('<', '<')
|
|
417
|
+
t.gsub!('>', '>')
|
|
418
|
+
t.gsub!(' ', ' ')
|
|
419
|
+
t.gsub!('’', "'")
|
|
420
|
+
t.gsub!('“', '"')
|
|
421
|
+
t.gsub!('”', '"')
|
|
422
|
+
t
|
|
423
|
+
end
|
|
103
424
|
end
|
|
104
425
|
end
|