gsc-cli 2.1.0 â 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +410 -414
- data/bin/gsc +28067 -5661
- data/dist/gsc +29121 -5046
- data/lib/gsc/aio_hunter.rb +343 -0
- data/lib/gsc/answer_synthesizer.rb +157 -0
- data/lib/gsc/api.rb +53 -1
- data/lib/gsc/auth.rb +26 -0
- data/lib/gsc/brand_segmenter.rb +140 -0
- data/lib/gsc/cache_manager.rb +806 -0
- data/lib/gsc/cannibalization_analyzer.rb +141 -0
- data/lib/gsc/canonical_chains.rb +367 -0
- data/lib/gsc/citation_simulator.rb +339 -0
- data/lib/gsc/cli/aio_hunter.rb +154 -0
- data/lib/gsc/cli/analytics.rb +788 -0
- data/lib/gsc/cli/audit.rb +1976 -0
- data/lib/gsc/cli/base.rb +384 -0
- data/lib/gsc/cli/cache.rb +266 -0
- data/lib/gsc/cli/canonical.rb +223 -0
- data/lib/gsc/cli/citation_simulator.rb +152 -0
- data/lib/gsc/cli/dashboard.rb +354 -0
- data/lib/gsc/cli/doctor.rb +129 -0
- data/lib/gsc/cli/eeat.rb +125 -0
- data/lib/gsc/cli/ga4.rb +852 -0
- data/lib/gsc/cli/growth.rb +650 -0
- data/lib/gsc/cli/hreflang.rb +164 -0
- data/lib/gsc/cli/image_seo.rb +162 -0
- data/lib/gsc/cli/indexing.rb +458 -0
- data/lib/gsc/cli/intent_shift.rb +125 -0
- data/lib/gsc/cli/keyword_value.rb +134 -0
- data/lib/gsc/cli/keywords.rb +795 -0
- data/lib/gsc/cli/landing_roi.rb +308 -0
- data/lib/gsc/cli/low_ctr.rb +213 -0
- data/lib/gsc/cli/mobile_parity.rb +150 -0
- data/lib/gsc/cli/report.rb +100 -0
- data/lib/gsc/cli/rich_results.rb +172 -0
- data/lib/gsc/cli/schema_generate.rb +149 -0
- data/lib/gsc/cli/seasonal.rb +232 -0
- data/lib/gsc/cli/security.rb +153 -0
- data/lib/gsc/cli/setup.rb +1291 -0
- data/lib/gsc/cli/sitemap_tree.rb +143 -0
- data/lib/gsc/cli/skill_pack.rb +62 -0
- data/lib/gsc/cli/soft_404.rb +199 -0
- data/lib/gsc/cli/sparkline.rb +227 -0
- data/lib/gsc/cli/watchdog.rb +150 -0
- data/lib/gsc/cli/zombie_purger.rb +208 -0
- data/lib/gsc/cli.rb +707 -5265
- data/lib/gsc/cli_advanced.rb +987 -44
- data/lib/gsc/client.rb +17 -2
- data/lib/gsc/color.rb +16 -1
- data/lib/gsc/command_registry.rb +47 -9
- data/lib/gsc/config.rb +2 -2
- data/lib/gsc/ctr_curve.rb +115 -0
- data/lib/gsc/decay_predictor.rb +322 -0
- data/lib/gsc/doctor.rb +434 -0
- data/lib/gsc/eeat_auditor.rb +428 -0
- data/lib/gsc/entity_auditor.rb +229 -0
- data/lib/gsc/firewall_scanner.rb +733 -0
- data/lib/gsc/geo_auditor.rb +368 -0
- data/lib/gsc/google_trends.rb +8 -1
- data/lib/gsc/heading_validator.rb +283 -0
- data/lib/gsc/hreflang_validator.rb +412 -0
- data/lib/gsc/image_seo.rb +286 -0
- data/lib/gsc/indexing_queue.rb +179 -0
- data/lib/gsc/indexnow.rb +93 -0
- data/lib/gsc/intent_shift.rb +188 -0
- data/lib/gsc/internal_links.rb +153 -36
- data/lib/gsc/keyword_value.rb +191 -0
- data/lib/gsc/landing_roi.rb +195 -0
- data/lib/gsc/llms_generator.rb +343 -22
- data/lib/gsc/low_ctr_rewriter.rb +408 -0
- data/lib/gsc/mobile_parity.rb +222 -0
- data/lib/gsc/network_tracer.rb +8 -1
- data/lib/gsc/page_analyzer.rb +47 -7
- data/lib/gsc/prompts.rb +38 -29
- data/lib/gsc/questions_harvester.rb +178 -0
- data/lib/gsc/report_generator.rb +461 -0
- data/lib/gsc/rich_results.rb +388 -0
- data/lib/gsc/robots_checker.rb +46 -15
- data/lib/gsc/schema_generator.rb +788 -0
- data/lib/gsc/schema_validator.rb +36 -38
- data/lib/gsc/seasonal_predictor.rb +381 -0
- data/lib/gsc/security_scanner.rb +496 -0
- data/lib/gsc/serp_feature_detector.rb +359 -0
- data/lib/gsc/serp_preview.rb +108 -22
- data/lib/gsc/site_crawler.rb +113 -21
- data/lib/gsc/sitemap_loader.rb +15 -4
- data/lib/gsc/sitemap_tree.rb +301 -0
- data/lib/gsc/skill_pack.rb +195 -0
- data/lib/gsc/soft_404_analyzer.rb +385 -0
- data/lib/gsc/sparkline.rb +171 -0
- data/lib/gsc/speed_correlator.rb +416 -0
- data/lib/gsc/striking_playbook.rb +190 -0
- data/lib/gsc/title_optimizer.rb +420 -0
- data/lib/gsc/vault.rb +260 -0
- data/lib/gsc/version.rb +1 -1
- data/lib/gsc/watchdog.rb +235 -0
- data/lib/gsc/zombie_purger.rb +366 -0
- data/lib/gsc.rb +118 -0
- metadata +75 -1
|
@@ -0,0 +1,339 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'net/http'
|
|
5
|
+
require 'uri'
|
|
6
|
+
require 'json'
|
|
7
|
+
|
|
8
|
+
module GSC
|
|
9
|
+
class CitationSimulator
|
|
10
|
+
MODELS = %w[perplexity chatgpt claude aio].freeze
|
|
11
|
+
DEFAULT_MODEL = 'perplexity'
|
|
12
|
+
|
|
13
|
+
STOPWORDS = %w[
|
|
14
|
+
a about above after again against all am an and any are aren't as at be because
|
|
15
|
+
been before being below between both but by can't cannot could couldn't did didn't
|
|
16
|
+
do does doesn't doing don't down during each few for from further had hadn't has
|
|
17
|
+
hasn't have haven't having he he'd he'll he's her here here's hers herself him
|
|
18
|
+
himself his how how's i i'd i'll i'm i've if in into is isn't it it's its itself
|
|
19
|
+
let's me more most mustn't my myself no nor not of off on once only or other ought
|
|
20
|
+
our ours ourselves out over own same shan't she she'd she'll she's should shouldn't
|
|
21
|
+
so some such than that that's the their theirs them themselves then there there's
|
|
22
|
+
these they they'd they'll they're they've this those through to too under until
|
|
23
|
+
up very was wasn't we we'd we'll we're we've were weren't what what's when when's
|
|
24
|
+
where where's which while who who's whom why why's with won't would wouldn't
|
|
25
|
+
you you'd you'll you're you've your yours yourself yourselves
|
|
26
|
+
].freeze
|
|
27
|
+
|
|
28
|
+
attr_reader :options, :model, :query
|
|
29
|
+
|
|
30
|
+
def initialize(options = {})
|
|
31
|
+
@options = options
|
|
32
|
+
@model = (options[:model] || DEFAULT_MODEL).to_s.downcase
|
|
33
|
+
@model = DEFAULT_MODEL unless MODELS.include?(@model)
|
|
34
|
+
@query = options[:query]&.to_s&.strip
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
def self.simulate(html_or_url, query = nil, options = {})
|
|
38
|
+
opts = options.merge(query: query || options[:query])
|
|
39
|
+
new(opts).simulate(html_or_url)
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
def simulate(target)
|
|
43
|
+
url, html, http_status = load_content(target)
|
|
44
|
+
inferred_query = @query.to_s.empty? ? infer_query_from_html(html) : @query
|
|
45
|
+
|
|
46
|
+
chunks = extract_content_chunks(html)
|
|
47
|
+
scored_chunks = score_chunks(chunks, inferred_query)
|
|
48
|
+
|
|
49
|
+
best_chunk = scored_chunks.first || default_chunk(inferred_query)
|
|
50
|
+
extracted_quotes = extract_verifiable_quotes(best_chunk, inferred_query)
|
|
51
|
+
|
|
52
|
+
cls_score = compute_citation_likelihood(best_chunk, chunks, html, inferred_query)
|
|
53
|
+
grade = compute_grade(cls_score)
|
|
54
|
+
|
|
55
|
+
emulated_answer = synthesize_emulated_answer(best_chunk, inferred_query, extracted_quotes, url)
|
|
56
|
+
prescriptions = generate_prescriptions(best_chunk, cls_score, html)
|
|
57
|
+
|
|
58
|
+
{
|
|
59
|
+
url: url,
|
|
60
|
+
http_status: http_status,
|
|
61
|
+
target_query: inferred_query,
|
|
62
|
+
model_profile: @model,
|
|
63
|
+
citation_likelihood_score: cls_score,
|
|
64
|
+
grade: grade,
|
|
65
|
+
status: cls_score >= 70 ? 'HIGH CITABILITY' : (cls_score >= 45 ? 'MODERATE CITABILITY' : 'LOW CITABILITY'),
|
|
66
|
+
emulated_ai_response: emulated_answer,
|
|
67
|
+
extracted_quotes: extracted_quotes,
|
|
68
|
+
top_cited_chunk: {
|
|
69
|
+
heading: best_chunk[:heading],
|
|
70
|
+
text: best_chunk[:text],
|
|
71
|
+
word_count: best_chunk[:word_count],
|
|
72
|
+
facts_count: best_chunk[:facts_count],
|
|
73
|
+
fact_density: best_chunk[:fact_density],
|
|
74
|
+
relevance_score: best_chunk[:relevance_score]
|
|
75
|
+
},
|
|
76
|
+
signals: {
|
|
77
|
+
total_chunks_analyzed: chunks.size,
|
|
78
|
+
facts_detected: best_chunk[:facts_count],
|
|
79
|
+
has_schema: html.include?('application/ld+json'),
|
|
80
|
+
has_author: html.match?(/author|written by|byline/i),
|
|
81
|
+
has_published_date: html.match?(/\d{4}-\d{2}-\d{2}|published|updated/i),
|
|
82
|
+
has_comparative_table: html.include?('<table')
|
|
83
|
+
},
|
|
84
|
+
prescriptions: prescriptions
|
|
85
|
+
}
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
private
|
|
89
|
+
|
|
90
|
+
def load_content(target)
|
|
91
|
+
target_str = target.to_s.strip
|
|
92
|
+
if target_str.match?(%r{^https?://})
|
|
93
|
+
uri = URI.parse(target_str)
|
|
94
|
+
req = Net::HTTP::Get.new(uri)
|
|
95
|
+
req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36'
|
|
96
|
+
req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
|
|
97
|
+
|
|
98
|
+
res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 8, read_timeout: 10) do |http|
|
|
99
|
+
http.request(req)
|
|
100
|
+
end
|
|
101
|
+
[target_str, res.body.to_s.dup.force_encoding('UTF-8').scrub, res.code.to_i]
|
|
102
|
+
elsif File.exist?(target_str)
|
|
103
|
+
[target_str, File.read(target_str, encoding: 'UTF-8'), 200]
|
|
104
|
+
else
|
|
105
|
+
# Raw HTML or string passed directly
|
|
106
|
+
[target_str.start_with?('http') ? target_str : 'local-document', target_str, 200]
|
|
107
|
+
end
|
|
108
|
+
rescue StandardError => e
|
|
109
|
+
[target_str.to_s, "<html><body><h1>Error loading content: #{e.message}</h1></body></html>", 0]
|
|
110
|
+
end
|
|
111
|
+
|
|
112
|
+
def infer_query_from_html(html)
|
|
113
|
+
h1_match = (html =~ %r{<h1[^>]*>(.*?)</h1>}i) ? clean_text($1) : ''
|
|
114
|
+
return h1_match unless h1_match.empty?
|
|
115
|
+
|
|
116
|
+
if html =~ %r{<title[^>]*>(.*?)</title>}i
|
|
117
|
+
title_part = (clean_text($1).split(/[|\-â]/).first || '').strip
|
|
118
|
+
return title_part unless title_part.empty?
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
'core product benefits and comparison'
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
def extract_content_chunks(html)
|
|
125
|
+
# Strip non-content blocks
|
|
126
|
+
cleaned = html.gsub(%r{<(script|style|nav|footer|header|aside|svg|noscript)[^>]*>.*?</\1>}im, ' ')
|
|
127
|
+
cleaned = cleaned.gsub(/<!--.*?-->/m, ' ')
|
|
128
|
+
|
|
129
|
+
chunks = []
|
|
130
|
+
current_heading = 'Main Content'
|
|
131
|
+
|
|
132
|
+
# Match sections, headings, paragraphs, and list blocks
|
|
133
|
+
cleaned.scan(%r{<(h[1-4]|p|li|blockquote|table)[^>]*>(.*?)</\1>}im) do |tag, content|
|
|
134
|
+
raw_text = clean_text(content)
|
|
135
|
+
next if raw_text.length < 25
|
|
136
|
+
|
|
137
|
+
if tag.start_with?('h')
|
|
138
|
+
current_heading = raw_text
|
|
139
|
+
else
|
|
140
|
+
words = raw_text.split(/\s+/)
|
|
141
|
+
next if words.size < 6
|
|
142
|
+
|
|
143
|
+
facts_count = count_facts(raw_text)
|
|
144
|
+
fact_density = (facts_count.to_f / words.size * 100.0).round(1)
|
|
145
|
+
|
|
146
|
+
chunks << {
|
|
147
|
+
tag: tag,
|
|
148
|
+
heading: current_heading,
|
|
149
|
+
text: raw_text,
|
|
150
|
+
word_count: words.size,
|
|
151
|
+
facts_count: facts_count,
|
|
152
|
+
fact_density: fact_density,
|
|
153
|
+
relevance_score: 0.0
|
|
154
|
+
}
|
|
155
|
+
end
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
chunks.empty? ? [default_chunk('general content')] : chunks
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
def score_chunks(chunks, query_str)
|
|
162
|
+
query_tokens = tokenize(query_str)
|
|
163
|
+
return chunks if query_tokens.empty?
|
|
164
|
+
|
|
165
|
+
chunks.each do |chunk|
|
|
166
|
+
chunk_tokens = tokenize("#{chunk[:heading]} #{chunk[:text]}")
|
|
167
|
+
overlap = query_tokens & chunk_tokens
|
|
168
|
+
|
|
169
|
+
# Base token match ratio
|
|
170
|
+
base_score = (overlap.size.to_f / [query_tokens.size, 1].max) * 40.0
|
|
171
|
+
|
|
172
|
+
# Exact phrase bonus
|
|
173
|
+
exact_bonus = chunk[:text].downcase.include?(query_str.downcase) ? 25.0 : 0.0
|
|
174
|
+
|
|
175
|
+
# Fact density bonus (up to 20 points)
|
|
176
|
+
fact_bonus = [chunk[:fact_density] * 2.0, 20.0].min
|
|
177
|
+
|
|
178
|
+
# Optimal word count sweet spot (40-80 words)
|
|
179
|
+
length_score = case chunk[:word_count]
|
|
180
|
+
when 40..80 then 15.0
|
|
181
|
+
when 25..39, 81..120 then 10.0
|
|
182
|
+
when 15..24, 121..180 then 5.0
|
|
183
|
+
else 2.0
|
|
184
|
+
end
|
|
185
|
+
|
|
186
|
+
# Model specific bias
|
|
187
|
+
model_bias = case @model
|
|
188
|
+
when 'perplexity' then chunk[:tag] == 'table' || chunk[:facts_count] >= 2 ? 10.0 : 0.0
|
|
189
|
+
when 'chatgpt' then chunk[:heading].match?(/how|steps|guide/i) ? 8.0 : 0.0
|
|
190
|
+
when 'claude' then chunk[:text].include?('because') || chunk[:text].include?('defined as') ? 8.0 : 0.0
|
|
191
|
+
when 'aio' then chunk[:word_count].between?(45, 65) ? 10.0 : 0.0
|
|
192
|
+
else 0.0
|
|
193
|
+
end
|
|
194
|
+
|
|
195
|
+
chunk[:relevance_score] = (base_score + exact_bonus + fact_bonus + length_score + model_bias).round(1)
|
|
196
|
+
end
|
|
197
|
+
|
|
198
|
+
chunks.sort_by { |c| -c[:relevance_score] }
|
|
199
|
+
end
|
|
200
|
+
|
|
201
|
+
def extract_verifiable_quotes(chunk, query_str)
|
|
202
|
+
sentences = chunk[:text].split(/(?<=[.!?])\s+/).map(&:strip).reject(&:empty?)
|
|
203
|
+
query_tokens = tokenize(query_str)
|
|
204
|
+
|
|
205
|
+
scored_sentences = sentences.map do |sent|
|
|
206
|
+
s_tokens = tokenize(sent)
|
|
207
|
+
overlap = query_tokens & s_tokens
|
|
208
|
+
facts = count_facts(sent)
|
|
209
|
+
score = (overlap.size * 3) + (facts * 4) + (sent.length.between?(50, 160) ? 5 : 0)
|
|
210
|
+
{ text: sent, score: score, facts: facts }
|
|
211
|
+
end
|
|
212
|
+
|
|
213
|
+
top = scored_sentences.sort_by { |s| -s[:score] }.first(2)
|
|
214
|
+
top.map { |s| s[:text] }
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
def compute_citation_likelihood(best_chunk, all_chunks, html, query_str)
|
|
218
|
+
score = 0.0
|
|
219
|
+
|
|
220
|
+
# 1. Best chunk relevance (max 35)
|
|
221
|
+
score += [best_chunk[:relevance_score] * 0.45, 35.0].min
|
|
222
|
+
|
|
223
|
+
# 2. Fact and statistical density (max 25)
|
|
224
|
+
fact_pts = [best_chunk[:facts_count] * 6.0, 25.0].min
|
|
225
|
+
score += fact_pts
|
|
226
|
+
|
|
227
|
+
# 3. Structural Clarity (max 15)
|
|
228
|
+
score += 5.0 if best_chunk[:heading] && best_chunk[:heading] != 'Main Content'
|
|
229
|
+
score += 5.0 if html.include?('<table') || html.include?('<ul>') || html.include?('<ol>')
|
|
230
|
+
score += 5.0 if html.include?('application/ld+json')
|
|
231
|
+
|
|
232
|
+
# 4. E-E-A-T & Trust verification markers (max 15)
|
|
233
|
+
score += 5.0 if html.match?(/author|written by|reviewed by/i)
|
|
234
|
+
score += 5.0 if html.match?(/\d{4}-\d{2}-\d{2}|published|updated/i)
|
|
235
|
+
score += 5.0 if html.match?(/sources|references|citations|study|data/i)
|
|
236
|
+
|
|
237
|
+
# 5. Length & Conciseness suitability (max 10)
|
|
238
|
+
if best_chunk[:word_count].between?(35, 85)
|
|
239
|
+
score += 10.0
|
|
240
|
+
elsif best_chunk[:word_count].between?(20, 120)
|
|
241
|
+
score += 5.0
|
|
242
|
+
end
|
|
243
|
+
|
|
244
|
+
# Evasive / AI fluff penalty (-10)
|
|
245
|
+
if best_chunk[:text].match?(/in today's (fast-paced|digital) world|it is important to remember|as mentioned earlier/i)
|
|
246
|
+
score -= 10.0
|
|
247
|
+
end
|
|
248
|
+
|
|
249
|
+
[[score.round(1), 100.0].min, 5.0].max
|
|
250
|
+
end
|
|
251
|
+
|
|
252
|
+
def synthesize_emulated_answer(best_chunk, query_str, quotes, url)
|
|
253
|
+
quote_str = quotes.first || best_chunk[:text]
|
|
254
|
+
host = (URI.parse(url).host rescue nil) || (url.start_with?('http') ? url : 'document')
|
|
255
|
+
|
|
256
|
+
case @model
|
|
257
|
+
when 'perplexity'
|
|
258
|
+
"Based on #{host}, #{quote_str} [1]\n\n[1] #{url} â \"#{quote_str}\""
|
|
259
|
+
when 'chatgpt'
|
|
260
|
+
"According to #{host}'s documentation, #{quote_str}.\n\nSource: #{url}"
|
|
261
|
+
when 'claude'
|
|
262
|
+
"#{quote_str}\n\nKey citation verified from #{host} (#{url})."
|
|
263
|
+
when 'aio'
|
|
264
|
+
"Here is what you need to know: #{quote_str}\n\n(Cited from #{url})"
|
|
265
|
+
end
|
|
266
|
+
end
|
|
267
|
+
|
|
268
|
+
def generate_prescriptions(best_chunk, score, html)
|
|
269
|
+
steps = []
|
|
270
|
+
|
|
271
|
+
if best_chunk[:facts_count] < 2
|
|
272
|
+
steps << "Embed at least 2 explicit numerical metrics, percentages, or benchmark statistics in the lead sentence."
|
|
273
|
+
end
|
|
274
|
+
|
|
275
|
+
if !best_chunk[:word_count].between?(40, 75)
|
|
276
|
+
steps << "Refactor the core answer block to 45â65 words. Currently at #{best_chunk[:word_count]} words (LLMs heavily favor concise 50w propositions)."
|
|
277
|
+
end
|
|
278
|
+
|
|
279
|
+
unless html.include?('application/ld+json')
|
|
280
|
+
steps << "Add Schema.org JSON-LD (FAQPage, Article, or TechArticle) to provide machine-readable ground truth."
|
|
281
|
+
end
|
|
282
|
+
|
|
283
|
+
unless html.match?(/author|written by|reviewed by/i)
|
|
284
|
+
steps << "Add an explicit author byline with credentials (E-E-A-T trust signals are heavily weighted by RAG retrievers)."
|
|
285
|
+
end
|
|
286
|
+
|
|
287
|
+
unless html.include?('<table')
|
|
288
|
+
steps << "Convert comparison points into an HTML <table>. Perplexity and AI Overviews preferentially cite tabular data."
|
|
289
|
+
end
|
|
290
|
+
|
|
291
|
+
steps << "Format the section header as a direct question matching user search intent (e.g. 'How does X work?')." if steps.size < 3
|
|
292
|
+
steps
|
|
293
|
+
end
|
|
294
|
+
|
|
295
|
+
def compute_grade(score)
|
|
296
|
+
case score
|
|
297
|
+
when 90.0..100.0 then 'A+'
|
|
298
|
+
when 80.0...90.0 then 'A'
|
|
299
|
+
when 70.0...80.0 then 'B'
|
|
300
|
+
when 55.0...70.0 then 'C'
|
|
301
|
+
when 40.0...55.0 then 'D'
|
|
302
|
+
else 'F'
|
|
303
|
+
end
|
|
304
|
+
end
|
|
305
|
+
|
|
306
|
+
def count_facts(text)
|
|
307
|
+
count = 0
|
|
308
|
+
# Percentages and currencies ($199, 45%, 3.5x, âŦ50, ÂŖ10)
|
|
309
|
+
count += text.scan(/\b(?:\$|âŦ|ÂŖ)\d+(?:\.\d+)?|\b\d+(?:\.\d+)?%|\b\d+(?:\.\d+)?x\b/i).size
|
|
310
|
+
# Numerical values > 1900 or decimals or speeds (e.g. 250ms, 4.2s, 10,000, 2026)
|
|
311
|
+
count += text.scan(/\b\d{1,3}(?:,\d{3})+\b|\b\d+(?:\.\d+)?(?:ms|s|gb|mb|kb|km|mph|users|hours|days)\b/i).size
|
|
312
|
+
# Proper technical capitalized words / acronyms (e.g. HTTP/2, REST, API, JSON-LD, LCP, CLS)
|
|
313
|
+
count += text.scan(/\b[A-Z]{2,6}\b/).size
|
|
314
|
+
count
|
|
315
|
+
end
|
|
316
|
+
|
|
317
|
+
def tokenize(str)
|
|
318
|
+
str.to_s.downcase.gsub(/[^a-z0-9\s]/, ' ').split(/\s+/).reject do |t|
|
|
319
|
+
t.empty? || STOPWORDS.include?(t)
|
|
320
|
+
end.uniq
|
|
321
|
+
end
|
|
322
|
+
|
|
323
|
+
def clean_text(html_str)
|
|
324
|
+
html_str.gsub(/<[^>]+>/, ' ').gsub(/\s+/, ' ').strip
|
|
325
|
+
end
|
|
326
|
+
|
|
327
|
+
def default_chunk(query_str)
|
|
328
|
+
{
|
|
329
|
+
tag: 'p',
|
|
330
|
+
heading: 'Overview',
|
|
331
|
+
text: "Direct authoritative summary answering #{query_str}.",
|
|
332
|
+
word_count: 7,
|
|
333
|
+
facts_count: 1,
|
|
334
|
+
fact_density: 14.3,
|
|
335
|
+
relevance_score: 50.0
|
|
336
|
+
}
|
|
337
|
+
end
|
|
338
|
+
end
|
|
339
|
+
end
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'json'
|
|
5
|
+
require_relative 'base'
|
|
6
|
+
require_relative '../aio_hunter'
|
|
7
|
+
require_relative '../color'
|
|
8
|
+
|
|
9
|
+
module GSC
|
|
10
|
+
class CLI
|
|
11
|
+
module AioHunter
|
|
12
|
+
module_function
|
|
13
|
+
|
|
14
|
+
def run(command, target, extra, options, api = nil, site_url = nil, hostname = nil)
|
|
15
|
+
query_or_domain = target || (extra && extra.first)
|
|
16
|
+
|
|
17
|
+
puts "đ Hunting Google AI Overview (AIO) opportunities and citation sources..." unless options[:json]
|
|
18
|
+
|
|
19
|
+
res = GSC::AioHunter.analyze(query_or_domain, options, api, site_url)
|
|
20
|
+
|
|
21
|
+
if options[:json]
|
|
22
|
+
puts JSON.pretty_generate(res)
|
|
23
|
+
return
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
if options[:csv]
|
|
27
|
+
export_csv(res)
|
|
28
|
+
return
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
if res[:mode] == :query
|
|
32
|
+
render_query_terminal(res)
|
|
33
|
+
else
|
|
34
|
+
render_portfolio_terminal(res)
|
|
35
|
+
end
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
def render_query_terminal(res)
|
|
39
|
+
q = res[:query]
|
|
40
|
+
aio = res[:aio_presence]
|
|
41
|
+
cit = res[:citation_analysis]
|
|
42
|
+
rec = res[:capture_recipe]
|
|
43
|
+
|
|
44
|
+
puts "\n#{Color.bold("âââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââ")}"
|
|
45
|
+
puts "#{Color.bold("đ¤ GOOGLE AI OVERVIEW (AIO) OPPORTUNITY & CITATION HUNTER")}"
|
|
46
|
+
puts "#{Color.bold("âââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââ")}"
|
|
47
|
+
|
|
48
|
+
puts " âĸ Search Query: #{Color.bold(Color.cyan("\"#{q}\""))}"
|
|
49
|
+
puts " âĸ Evaluated Domain: #{Color.bold(res[:target_domain])}" if res[:target_domain]
|
|
50
|
+
puts " âĸ Query Intent: #{Color.bold(res[:intent].to_s.upcase.tr('_', ' '))}"
|
|
51
|
+
|
|
52
|
+
prob_color = aio[:probability_percent] >= 75 ? Color.red("#{aio[:probability_percent]}% - #{aio[:status]}") : Color.yellow("#{aio[:probability_percent]}% - #{aio[:status]}")
|
|
53
|
+
puts " âĸ AIO SERP Probability: [ #{Color.bold(prob_color)} ]"
|
|
54
|
+
puts " âĸ Organic CTR Drag: #{Color.red("-#{aio[:ctr_suppression_estimate]} suppression")} of blue links"
|
|
55
|
+
|
|
56
|
+
if cit[:domain_evaluated]
|
|
57
|
+
cited_str = cit[:top_3_prime_candidate] ? Color.green("â
PRIME CITATION CANDIDATE (Ranks Pos #{cit[:ranking_position]})") : Color.yellow("â ī¸ #{cit[:eligibility_status]} (Citation Gap: #{cit[:citation_gap_index]}%)")
|
|
58
|
+
else
|
|
59
|
+
cited_str = Color.yellow("âšī¸ Unspecified Domain (Pass --domain <domain> to evaluate GSC rank)")
|
|
60
|
+
end
|
|
61
|
+
puts " âĸ Domain Citation State: #{Color.bold(cited_str)}"
|
|
62
|
+
puts " âĸ AIO Opportunity Score: #{Color.bold(Color.green("#{res[:opportunity_score]} / 100"))}"
|
|
63
|
+
puts "#{Color.bold("âââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââ")}\n"
|
|
64
|
+
|
|
65
|
+
# Citation Criteria
|
|
66
|
+
puts "#{Color.bold("GOOGLE GEMINI AIO CITATION & GROUNDING CRITERIA:")}"
|
|
67
|
+
puts " âĸ Grounding Model: Google Gemini prioritizes citations from Top 3 organic ranking URLs"
|
|
68
|
+
puts " âĸ Direct Answer Match: Requires 35â50 word direct definition answering \"#{q}\""
|
|
69
|
+
puts " âĸ Structured Schema: #{Color.yellow(rec[:recommended_schema])}"
|
|
70
|
+
puts " âĸ Required Structure: #{rec[:content_elements].first}"
|
|
71
|
+
|
|
72
|
+
# Capture Playbook
|
|
73
|
+
puts "\n#{Color.bold("âââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââ")}"
|
|
74
|
+
puts "#{Color.bold("ACTIONABLE AIO CITATION CAPTURE PLAYBOOK:")}"
|
|
75
|
+
puts " âĸ Strategy: #{Color.bold(rec[:strategy])}"
|
|
76
|
+
puts " âĸ Recommended Heading: #{Color.cyan(rec[:recommended_heading])}"
|
|
77
|
+
puts " âĸ Recommended Schema: #{Color.yellow(rec[:recommended_schema])}"
|
|
78
|
+
puts " âĸ Required Elements: #{rec[:content_elements].join(' | ')}"
|
|
79
|
+
puts "\n #{Color.bold("Ready-to-Paste Direct Answer Snippet:")}"
|
|
80
|
+
puts Color.green(" \"#{rec[:direct_answer_draft]}\"")
|
|
81
|
+
puts "#{Color.bold("âââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââ")}\n"
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
def render_portfolio_terminal(res)
|
|
85
|
+
sum = res[:summary]
|
|
86
|
+
|
|
87
|
+
puts "\n#{Color.bold("âââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââ")}"
|
|
88
|
+
puts "#{Color.bold("đ¤ GOOGLE AI OVERVIEW (AIO) PORTFOLIO RADAR: #{res[:domain].upcase}")}"
|
|
89
|
+
puts "#{Color.bold("âââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââ")}"
|
|
90
|
+
|
|
91
|
+
puts " âĸ Total Queries Audited: #{Color.bold(res[:total_queries_audited].to_s)} keywords"
|
|
92
|
+
puts " âĸ AIO SERP Penetration: #{Color.bold(Color.yellow("#{sum[:aio_penetration_rate]}%"))} (#{sum[:aio_triggering_queries]} queries triggering AIO)"
|
|
93
|
+
puts " âĸ Top 3 Prime Candidates: #{Color.bold(Color.green(sum[:currently_cited_count].to_s))}"
|
|
94
|
+
puts " âĸ High-Threat Uncited: #{Color.bold(Color.red("#{sum[:high_threat_uncited_count]} keywords"))} (Click hemorrhaging to AIO)"
|
|
95
|
+
puts " âĸ AIO Vulnerability Index: #{Color.bold(Color.red("#{sum[:portfolio_aio_vulnerability_index]}%"))}"
|
|
96
|
+
if res[:opportunities].empty?
|
|
97
|
+
puts " âšī¸ No Search Console query data found for #{res[:domain]}."
|
|
98
|
+
puts " Connect Google Search Console credentials via `gsc setup` to audit your portfolio,"
|
|
99
|
+
puts " or analyze a specific keyword with `gsc aio \"<query>\"`."
|
|
100
|
+
puts "\n#{Color.bold("âââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââ")}\n"
|
|
101
|
+
return
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
puts format("%-32s | %-6s | %-6s | %-8s | %-7s | %-8s | %-24s",
|
|
105
|
+
"Query", "Imp", "Pos", "AIO Prob", "Top 3?", "Opp Score", "Recommended Fix")
|
|
106
|
+
puts "#{Color.bold("â" * 105)}"
|
|
107
|
+
|
|
108
|
+
res[:opportunities].each do |item|
|
|
109
|
+
q_trunc = item[:query].length > 30 ? "#{item[:query][0..27]}..." : item[:query]
|
|
110
|
+
prob_s = "#{item[:aio_probability]}%"
|
|
111
|
+
cited_s = item[:top_3_prime] ? Color.green("YES") : Color.gray("NO")
|
|
112
|
+
opp_s = "#{item[:opportunity_score]}/100"
|
|
113
|
+
|
|
114
|
+
puts format("%-32s | %-6d | %-6.1f | %-8s | %-16s | %-9s | %-24s",
|
|
115
|
+
q_trunc, item[:impressions], item[:position], prob_s, cited_s, opp_s, item[:recipe_summary])
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
puts "#{Color.bold("âââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââ")}\n"
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
def export_csv(res)
|
|
122
|
+
require 'csv'
|
|
123
|
+
|
|
124
|
+
if res[:mode] == :query
|
|
125
|
+
puts "query,intent,aio_probability,domain_cited,opportunity_score,strategy"
|
|
126
|
+
puts [
|
|
127
|
+
res[:query],
|
|
128
|
+
res[:intent],
|
|
129
|
+
res[:aio_presence][:probability_percent],
|
|
130
|
+
res[:citation_analysis][:domain_cited],
|
|
131
|
+
res[:opportunity_score],
|
|
132
|
+
res[:capture_recipe][:strategy]
|
|
133
|
+
].to_csv
|
|
134
|
+
else
|
|
135
|
+
puts "query,impressions,clicks,position,ctr,intent,aio_probability,domain_cited,opportunity_score,recipe"
|
|
136
|
+
res[:opportunities].each do |item|
|
|
137
|
+
puts [
|
|
138
|
+
item[:query],
|
|
139
|
+
item[:impressions],
|
|
140
|
+
item[:clicks],
|
|
141
|
+
item[:position],
|
|
142
|
+
item[:ctr],
|
|
143
|
+
item[:intent],
|
|
144
|
+
item[:aio_probability],
|
|
145
|
+
item[:domain_cited],
|
|
146
|
+
item[:opportunity_score],
|
|
147
|
+
item[:recipe_summary]
|
|
148
|
+
].to_csv
|
|
149
|
+
end
|
|
150
|
+
end
|
|
151
|
+
end
|
|
152
|
+
end
|
|
153
|
+
end
|
|
154
|
+
end
|