gsc-cli 2.1.0 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +410 -414
- data/bin/gsc +28067 -5661
- data/dist/gsc +29121 -5046
- data/lib/gsc/aio_hunter.rb +343 -0
- data/lib/gsc/answer_synthesizer.rb +157 -0
- data/lib/gsc/api.rb +53 -1
- data/lib/gsc/auth.rb +26 -0
- data/lib/gsc/brand_segmenter.rb +140 -0
- data/lib/gsc/cache_manager.rb +806 -0
- data/lib/gsc/cannibalization_analyzer.rb +141 -0
- data/lib/gsc/canonical_chains.rb +367 -0
- data/lib/gsc/citation_simulator.rb +339 -0
- data/lib/gsc/cli/aio_hunter.rb +154 -0
- data/lib/gsc/cli/analytics.rb +788 -0
- data/lib/gsc/cli/audit.rb +1976 -0
- data/lib/gsc/cli/base.rb +384 -0
- data/lib/gsc/cli/cache.rb +266 -0
- data/lib/gsc/cli/canonical.rb +223 -0
- data/lib/gsc/cli/citation_simulator.rb +152 -0
- data/lib/gsc/cli/dashboard.rb +354 -0
- data/lib/gsc/cli/doctor.rb +129 -0
- data/lib/gsc/cli/eeat.rb +125 -0
- data/lib/gsc/cli/ga4.rb +852 -0
- data/lib/gsc/cli/growth.rb +650 -0
- data/lib/gsc/cli/hreflang.rb +164 -0
- data/lib/gsc/cli/image_seo.rb +162 -0
- data/lib/gsc/cli/indexing.rb +458 -0
- data/lib/gsc/cli/intent_shift.rb +125 -0
- data/lib/gsc/cli/keyword_value.rb +134 -0
- data/lib/gsc/cli/keywords.rb +795 -0
- data/lib/gsc/cli/landing_roi.rb +308 -0
- data/lib/gsc/cli/low_ctr.rb +213 -0
- data/lib/gsc/cli/mobile_parity.rb +150 -0
- data/lib/gsc/cli/report.rb +100 -0
- data/lib/gsc/cli/rich_results.rb +172 -0
- data/lib/gsc/cli/schema_generate.rb +149 -0
- data/lib/gsc/cli/seasonal.rb +232 -0
- data/lib/gsc/cli/security.rb +153 -0
- data/lib/gsc/cli/setup.rb +1291 -0
- data/lib/gsc/cli/sitemap_tree.rb +143 -0
- data/lib/gsc/cli/skill_pack.rb +62 -0
- data/lib/gsc/cli/soft_404.rb +199 -0
- data/lib/gsc/cli/sparkline.rb +227 -0
- data/lib/gsc/cli/watchdog.rb +150 -0
- data/lib/gsc/cli/zombie_purger.rb +208 -0
- data/lib/gsc/cli.rb +707 -5265
- data/lib/gsc/cli_advanced.rb +987 -44
- data/lib/gsc/client.rb +17 -2
- data/lib/gsc/color.rb +16 -1
- data/lib/gsc/command_registry.rb +47 -9
- data/lib/gsc/config.rb +2 -2
- data/lib/gsc/ctr_curve.rb +115 -0
- data/lib/gsc/decay_predictor.rb +322 -0
- data/lib/gsc/doctor.rb +434 -0
- data/lib/gsc/eeat_auditor.rb +428 -0
- data/lib/gsc/entity_auditor.rb +229 -0
- data/lib/gsc/firewall_scanner.rb +733 -0
- data/lib/gsc/geo_auditor.rb +368 -0
- data/lib/gsc/google_trends.rb +8 -1
- data/lib/gsc/heading_validator.rb +283 -0
- data/lib/gsc/hreflang_validator.rb +412 -0
- data/lib/gsc/image_seo.rb +286 -0
- data/lib/gsc/indexing_queue.rb +179 -0
- data/lib/gsc/indexnow.rb +93 -0
- data/lib/gsc/intent_shift.rb +188 -0
- data/lib/gsc/internal_links.rb +153 -36
- data/lib/gsc/keyword_value.rb +191 -0
- data/lib/gsc/landing_roi.rb +195 -0
- data/lib/gsc/llms_generator.rb +343 -22
- data/lib/gsc/low_ctr_rewriter.rb +408 -0
- data/lib/gsc/mobile_parity.rb +222 -0
- data/lib/gsc/network_tracer.rb +8 -1
- data/lib/gsc/page_analyzer.rb +47 -7
- data/lib/gsc/prompts.rb +38 -29
- data/lib/gsc/questions_harvester.rb +178 -0
- data/lib/gsc/report_generator.rb +461 -0
- data/lib/gsc/rich_results.rb +388 -0
- data/lib/gsc/robots_checker.rb +46 -15
- data/lib/gsc/schema_generator.rb +788 -0
- data/lib/gsc/schema_validator.rb +36 -38
- data/lib/gsc/seasonal_predictor.rb +381 -0
- data/lib/gsc/security_scanner.rb +496 -0
- data/lib/gsc/serp_feature_detector.rb +359 -0
- data/lib/gsc/serp_preview.rb +108 -22
- data/lib/gsc/site_crawler.rb +113 -21
- data/lib/gsc/sitemap_loader.rb +15 -4
- data/lib/gsc/sitemap_tree.rb +301 -0
- data/lib/gsc/skill_pack.rb +195 -0
- data/lib/gsc/soft_404_analyzer.rb +385 -0
- data/lib/gsc/sparkline.rb +171 -0
- data/lib/gsc/speed_correlator.rb +416 -0
- data/lib/gsc/striking_playbook.rb +190 -0
- data/lib/gsc/title_optimizer.rb +420 -0
- data/lib/gsc/vault.rb +260 -0
- data/lib/gsc/version.rb +1 -1
- data/lib/gsc/watchdog.rb +235 -0
- data/lib/gsc/zombie_purger.rb +366 -0
- data/lib/gsc.rb +118 -0
- metadata +75 -1
data/lib/gsc/sitemap_loader.rb
CHANGED
|
@@ -4,9 +4,18 @@ require 'net/http'
|
|
|
4
4
|
require 'uri'
|
|
5
5
|
require 'zlib'
|
|
6
6
|
require 'stringio'
|
|
7
|
+
require 'cgi'
|
|
7
8
|
|
|
8
9
|
module GSC
|
|
9
10
|
class SitemapLoader
|
|
11
|
+
def initialize(path_or_url = nil)
|
|
12
|
+
@path_or_url = path_or_url
|
|
13
|
+
end
|
|
14
|
+
|
|
15
|
+
def load(default_origin = nil)
|
|
16
|
+
self.class.resolve_urls(@path_or_url, default_origin, quiet: true)
|
|
17
|
+
end
|
|
18
|
+
|
|
10
19
|
def self.fetch_content(path_or_url)
|
|
11
20
|
if path_or_url.start_with?('http://', 'https://')
|
|
12
21
|
uri = URI(path_or_url)
|
|
@@ -61,14 +70,15 @@ module GSC
|
|
|
61
70
|
end
|
|
62
71
|
|
|
63
72
|
xml = fetch_content(target_input)
|
|
73
|
+
clean_xml = xml.to_s.gsub(/<!\[CDATA\[(.*?)\]\]>/m, '\1')
|
|
64
74
|
urls = []
|
|
65
75
|
|
|
66
76
|
# Parse child sitemaps if this is a sitemap index
|
|
67
|
-
sitemap_locs =
|
|
77
|
+
sitemap_locs = clean_xml.scan(/<sitemap>\s*<loc>([^<]+)<\/loc>/m).flatten
|
|
68
78
|
if sitemap_locs.any?
|
|
69
79
|
puts "📑 Found #{sitemap_locs.size} nested sitemaps in index..." unless quiet
|
|
70
80
|
sitemap_locs.each do |child_url|
|
|
71
|
-
child_url = child_url.strip
|
|
81
|
+
child_url = CGI.unescapeHTML(child_url.strip)
|
|
72
82
|
unless child_url.start_with?('http://', 'https://')
|
|
73
83
|
puts Color.c(" ⚠️ Skipping non-HTTP child sitemap location: #{child_url}", Color::YELLOW) unless quiet
|
|
74
84
|
next
|
|
@@ -76,13 +86,14 @@ module GSC
|
|
|
76
86
|
puts " ↳ Loading child sitemap: #{child_url}" unless quiet
|
|
77
87
|
begin
|
|
78
88
|
child_xml = fetch_content(child_url)
|
|
79
|
-
|
|
89
|
+
clean_child = child_xml.to_s.gsub(/<!\[CDATA\[(.*?)\]\]>/m, '\1')
|
|
90
|
+
urls.concat(clean_child.scan(/<url>\s*<loc>([^<]+)<\/loc>/m).flatten.map { |u| CGI.unescapeHTML(u.strip) })
|
|
80
91
|
rescue StandardError => e
|
|
81
92
|
puts Color.c(" ⚠️ Warning: Could not load child sitemap #{child_url}: #{e.message}", Color::YELLOW) unless quiet
|
|
82
93
|
end
|
|
83
94
|
end
|
|
84
95
|
else
|
|
85
|
-
urls.concat(
|
|
96
|
+
urls.concat(clean_xml.scan(/<loc>([^<]+)<\/loc>/m).flatten.map { |u| CGI.unescapeHTML(u.strip) })
|
|
86
97
|
end
|
|
87
98
|
|
|
88
99
|
urls.uniq
|
|
@@ -0,0 +1,301 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'net/http'
|
|
5
|
+
require 'uri'
|
|
6
|
+
require 'zlib'
|
|
7
|
+
require 'stringio'
|
|
8
|
+
require 'time'
|
|
9
|
+
require 'date'
|
|
10
|
+
require 'cgi'
|
|
11
|
+
|
|
12
|
+
module GSC
|
|
13
|
+
class SitemapTree
|
|
14
|
+
MAX_URLS_PER_SITEMAP = 50_000
|
|
15
|
+
MAX_BYTES_PER_SITEMAP = 50 * 1024 * 1024 # 50 MB
|
|
16
|
+
STALE_DAYS_THRESHOLD = 180
|
|
17
|
+
|
|
18
|
+
attr_reader :root_source, :options, :tree, :all_urls, :violations
|
|
19
|
+
|
|
20
|
+
def initialize(root_source, options = {})
|
|
21
|
+
@root_source = root_source.to_s.strip
|
|
22
|
+
@options = options
|
|
23
|
+
@tree = {}
|
|
24
|
+
@all_urls = []
|
|
25
|
+
@violations = []
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def audit(gsc_pages = [])
|
|
29
|
+
content, size_bytes = fetch_raw_content(@root_source)
|
|
30
|
+
is_index = content.include?('<sitemapindex')
|
|
31
|
+
|
|
32
|
+
@tree = if is_index
|
|
33
|
+
parse_sitemap_index(@root_source, content, size_bytes)
|
|
34
|
+
else
|
|
35
|
+
parse_single_urlset(@root_source, content, size_bytes)
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
# Collect all unique URLs from tree
|
|
39
|
+
@all_urls = collect_urls_from_tree(@tree).uniq
|
|
40
|
+
|
|
41
|
+
# Coverage analysis against GSC pages if provided
|
|
42
|
+
coverage = analyze_coverage(gsc_pages, @all_urls)
|
|
43
|
+
|
|
44
|
+
# Global compliance health score (0-100)
|
|
45
|
+
health_score = calculate_health_score
|
|
46
|
+
|
|
47
|
+
{
|
|
48
|
+
root: @root_source,
|
|
49
|
+
is_index: is_index,
|
|
50
|
+
total_sitemaps: count_sitemaps(@tree),
|
|
51
|
+
total_urls: @all_urls.size,
|
|
52
|
+
total_size_bytes: sum_sizes(@tree),
|
|
53
|
+
tree: @tree,
|
|
54
|
+
all_urls: @all_urls,
|
|
55
|
+
violations: @violations,
|
|
56
|
+
coverage: coverage,
|
|
57
|
+
health_score: health_score
|
|
58
|
+
}
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
private
|
|
62
|
+
|
|
63
|
+
def parse_sitemap_index(source_url, xml, size_bytes)
|
|
64
|
+
sub_sitemaps = []
|
|
65
|
+
now = Time.now
|
|
66
|
+
|
|
67
|
+
# Extract <sitemap> blocks
|
|
68
|
+
xml.scan(/<sitemap\b[^>]*>(.*?)<\/sitemap>/im).each do |match|
|
|
69
|
+
block = match[0]
|
|
70
|
+
loc = block.match(/<loc\b[^>]*>(.*?)<\/loc>/i)&.[](1)&.strip
|
|
71
|
+
next unless loc
|
|
72
|
+
loc = loc.sub(/^<!\[CDATA\[/i, '').sub(/\]\]>$/i, '').strip
|
|
73
|
+
loc = CGI.unescapeHTML(loc)
|
|
74
|
+
|
|
75
|
+
lastmod_str = block.match(/<lastmod\b[^>]*>(.*?)<\/lastmod>/i)&.[](1)&.strip
|
|
76
|
+
lastmod_time, lastmod_valid = parse_and_validate_timestamp(lastmod_str, loc)
|
|
77
|
+
|
|
78
|
+
# Recursively fetch child sitemap if requested or level 1
|
|
79
|
+
child_node = begin
|
|
80
|
+
child_content, child_size = fetch_raw_content(loc)
|
|
81
|
+
if child_content.include?('<sitemapindex')
|
|
82
|
+
parse_sitemap_index(loc, child_content, child_size)
|
|
83
|
+
else
|
|
84
|
+
parse_single_urlset(loc, child_content, child_size)
|
|
85
|
+
end
|
|
86
|
+
rescue => e
|
|
87
|
+
@violations << { type: :fetch_error, sitemap: loc, message: e.message }
|
|
88
|
+
{
|
|
89
|
+
url: loc,
|
|
90
|
+
type: :error,
|
|
91
|
+
error: e.message,
|
|
92
|
+
url_count: 0,
|
|
93
|
+
urls: []
|
|
94
|
+
}
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
child_node[:declared_lastmod] = lastmod_str
|
|
98
|
+
child_node[:lastmod_time] = lastmod_time
|
|
99
|
+
child_node[:lastmod_valid] = lastmod_valid
|
|
100
|
+
|
|
101
|
+
sub_sitemaps << child_node
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
{
|
|
105
|
+
url: source_url,
|
|
106
|
+
type: :sitemapindex,
|
|
107
|
+
size_bytes: size_bytes,
|
|
108
|
+
children: sub_sitemaps,
|
|
109
|
+
url_count: sub_sitemaps.sum { |s| s[:url_count] || 0 }
|
|
110
|
+
}
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
def parse_single_urlset(source_url, xml, size_bytes)
|
|
114
|
+
urls = []
|
|
115
|
+
now = Time.now
|
|
116
|
+
latest_lastmod = nil
|
|
117
|
+
|
|
118
|
+
# Check file size limit (50MB)
|
|
119
|
+
if size_bytes > MAX_BYTES_PER_SITEMAP
|
|
120
|
+
@violations << {
|
|
121
|
+
type: :size_overflow,
|
|
122
|
+
sitemap: source_url,
|
|
123
|
+
size_bytes: size_bytes,
|
|
124
|
+
message: "Exceeds Google 50MB limit (#{format('%.2f', size_bytes / (1024.0 * 1024.0))} MB)"
|
|
125
|
+
}
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
xml.scan(/<url\b[^>]*>(.*?)<\/url>/im).each do |match|
|
|
129
|
+
block = match[0]
|
|
130
|
+
loc = block.match(/<loc\b[^>]*>(.*?)<\/loc>/i)&.[](1)&.strip
|
|
131
|
+
next unless loc
|
|
132
|
+
loc = loc.sub(/^<!\[CDATA\[/i, '').sub(/\]\]>$/i, '').strip
|
|
133
|
+
loc = CGI.unescapeHTML(loc)
|
|
134
|
+
|
|
135
|
+
lastmod_str = block.match(/<lastmod\b[^>]*>(.*?)<\/lastmod>/i)&.[](1)&.strip
|
|
136
|
+
changefreq = block.match(/<changefreq\b[^>]*>(.*?)<\/changefreq>/i)&.[](1)&.strip
|
|
137
|
+
priority = block.match(/<priority\b[^>]*>(.*?)<\/priority>/i)&.[](1)&.strip
|
|
138
|
+
|
|
139
|
+
lastmod_time, lastmod_valid = parse_and_validate_timestamp(lastmod_str, source_url)
|
|
140
|
+
if lastmod_time && (latest_lastmod.nil? || lastmod_time > latest_lastmod)
|
|
141
|
+
latest_lastmod = lastmod_time
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
urls << {
|
|
145
|
+
loc: loc,
|
|
146
|
+
lastmod: lastmod_str,
|
|
147
|
+
changefreq: changefreq,
|
|
148
|
+
priority: priority
|
|
149
|
+
}
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
# Check URL count limit (50,000)
|
|
153
|
+
if urls.size > MAX_URLS_PER_SITEMAP
|
|
154
|
+
@violations << {
|
|
155
|
+
type: :url_overflow,
|
|
156
|
+
sitemap: source_url,
|
|
157
|
+
count: urls.size,
|
|
158
|
+
message: "Exceeds Google 50,000 URLs per sitemap limit (#{urls.size} URLs)"
|
|
159
|
+
}
|
|
160
|
+
end
|
|
161
|
+
|
|
162
|
+
# Check date drift / staleness
|
|
163
|
+
drift_days = latest_lastmod ? ((now - latest_lastmod) / 86400.0).round : nil
|
|
164
|
+
is_stale = drift_days && drift_days > STALE_DAYS_THRESHOLD
|
|
165
|
+
|
|
166
|
+
{
|
|
167
|
+
url: source_url,
|
|
168
|
+
type: :urlset,
|
|
169
|
+
size_bytes: size_bytes,
|
|
170
|
+
url_count: urls.size,
|
|
171
|
+
latest_lastmod: latest_lastmod ? latest_lastmod.strftime('%Y-%m-%d') : nil,
|
|
172
|
+
drift_days: drift_days,
|
|
173
|
+
is_stale: is_stale,
|
|
174
|
+
urls: urls
|
|
175
|
+
}
|
|
176
|
+
end
|
|
177
|
+
|
|
178
|
+
def parse_and_validate_timestamp(ts_str, sitemap_url)
|
|
179
|
+
return [nil, true] if ts_str.nil? || ts_str.empty?
|
|
180
|
+
|
|
181
|
+
# ISO-8601 regex test
|
|
182
|
+
is_iso = ts_str =~ /^\d{4}-\d{2}-\d{2}(?:T\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}:?\d{2})?)?$/
|
|
183
|
+
unless is_iso
|
|
184
|
+
@violations << {
|
|
185
|
+
type: :invalid_timestamp,
|
|
186
|
+
sitemap: sitemap_url,
|
|
187
|
+
timestamp: ts_str,
|
|
188
|
+
message: "Invalid non-ISO-8601 timestamp: '#{ts_str}'"
|
|
189
|
+
}
|
|
190
|
+
return [nil, false]
|
|
191
|
+
end
|
|
192
|
+
|
|
193
|
+
t = Time.parse(ts_str) rescue nil
|
|
194
|
+
if t && t > (Time.now + 86400) # Allow 1 day clock drift
|
|
195
|
+
@violations << {
|
|
196
|
+
type: :future_timestamp,
|
|
197
|
+
sitemap: sitemap_url,
|
|
198
|
+
timestamp: ts_str,
|
|
199
|
+
message: "Future timestamp detected: #{ts_str}"
|
|
200
|
+
}
|
|
201
|
+
return [t, false]
|
|
202
|
+
end
|
|
203
|
+
|
|
204
|
+
[t, true]
|
|
205
|
+
end
|
|
206
|
+
|
|
207
|
+
def collect_urls_from_tree(node)
|
|
208
|
+
if node[:type] == :sitemapindex && node[:children]
|
|
209
|
+
node[:children].flat_map { |c| collect_urls_from_tree(c) }
|
|
210
|
+
elsif node[:type] == :urlset && node[:urls]
|
|
211
|
+
node[:urls].map { |u| u[:loc] }
|
|
212
|
+
else
|
|
213
|
+
[]
|
|
214
|
+
end
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
def count_sitemaps(node)
|
|
218
|
+
if node[:type] == :sitemapindex && node[:children]
|
|
219
|
+
1 + node[:children].sum { |c| count_sitemaps(c) }
|
|
220
|
+
else
|
|
221
|
+
1
|
|
222
|
+
end
|
|
223
|
+
end
|
|
224
|
+
|
|
225
|
+
def sum_sizes(node)
|
|
226
|
+
size = node[:size_bytes] || 0
|
|
227
|
+
if node[:type] == :sitemapindex && node[:children]
|
|
228
|
+
size += node[:children].sum { |c| sum_sizes(c) }
|
|
229
|
+
end
|
|
230
|
+
size
|
|
231
|
+
end
|
|
232
|
+
|
|
233
|
+
def analyze_coverage(gsc_pages, sitemap_urls)
|
|
234
|
+
return { status: :no_gsc_data } if gsc_pages.nil? || gsc_pages.empty?
|
|
235
|
+
|
|
236
|
+
sitemap_set = sitemap_urls.map { |u| normalize_url(u) }.to_set
|
|
237
|
+
gsc_urls = gsc_pages.map { |p| normalize_url(p[:url] || p['url'] || p) }
|
|
238
|
+
|
|
239
|
+
included = gsc_urls.select { |u| sitemap_set.include?(u) }
|
|
240
|
+
missing = gsc_urls.reject { |u| sitemap_set.include?(u) }
|
|
241
|
+
|
|
242
|
+
coverage_pct = gsc_urls.any? ? ((included.size.to_f / gsc_urls.size) * 100.0).round(1) : 100.0
|
|
243
|
+
|
|
244
|
+
{
|
|
245
|
+
total_gsc_pages: gsc_urls.size,
|
|
246
|
+
included_in_sitemap: included.size,
|
|
247
|
+
missing_from_sitemap: missing.size,
|
|
248
|
+
coverage_pct: coverage_pct,
|
|
249
|
+
missing_sample: missing.first(10)
|
|
250
|
+
}
|
|
251
|
+
end
|
|
252
|
+
|
|
253
|
+
def calculate_health_score
|
|
254
|
+
score = 100
|
|
255
|
+
@violations.each do |v|
|
|
256
|
+
case v[:type]
|
|
257
|
+
when :fetch_error then score -= 25
|
|
258
|
+
when :size_overflow, :url_overflow then score -= 20
|
|
259
|
+
when :future_timestamp then score -= 15
|
|
260
|
+
when :invalid_timestamp then score -= 10
|
|
261
|
+
end
|
|
262
|
+
end
|
|
263
|
+
[0, score].max
|
|
264
|
+
end
|
|
265
|
+
|
|
266
|
+
def fetch_raw_content(source)
|
|
267
|
+
if source =~ %r{^https?://}
|
|
268
|
+
uri = URI.parse(source)
|
|
269
|
+
req = Net::HTTP::Get.new(uri)
|
|
270
|
+
req['User-Agent'] = 'Mozilla/5.0 (compatible; GSC-SitemapTree-Auditor/1.0)'
|
|
271
|
+
req['Accept-Encoding'] = 'gzip'
|
|
272
|
+
|
|
273
|
+
http = Net::HTTP.new(uri.host, uri.port)
|
|
274
|
+
http.use_ssl = (uri.scheme == 'https')
|
|
275
|
+
http.open_timeout = 8
|
|
276
|
+
http.read_timeout = 15
|
|
277
|
+
|
|
278
|
+
res = http.request(req)
|
|
279
|
+
raise "HTTP #{res.code} on #{source}" unless res.is_a?(Net::HTTPSuccess)
|
|
280
|
+
|
|
281
|
+
raw = res.body || ''
|
|
282
|
+
decompressed = if (res['content-encoding'] =~ /gzip/i || source.end_with?('.gz')) && !raw.empty?
|
|
283
|
+
Zlib::GzipReader.new(StringIO.new(raw)).read
|
|
284
|
+
else
|
|
285
|
+
raw
|
|
286
|
+
end
|
|
287
|
+
content = decompressed.to_s.dup.force_encoding('UTF-8').scrub
|
|
288
|
+
[content, content.bytesize]
|
|
289
|
+
else
|
|
290
|
+
raise "Local sitemap file not found: #{source}" unless File.exist?(source)
|
|
291
|
+
content = File.read(source, encoding: 'UTF-8').scrub
|
|
292
|
+
[content, content.bytesize]
|
|
293
|
+
end
|
|
294
|
+
end
|
|
295
|
+
|
|
296
|
+
def normalize_url(url)
|
|
297
|
+
u = url.to_s.strip.downcase.chomp('/')
|
|
298
|
+
u.sub(%r{^https?://(www\.)?}, '')
|
|
299
|
+
end
|
|
300
|
+
end
|
|
301
|
+
end
|
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'json'
|
|
5
|
+
require 'fileutils'
|
|
6
|
+
|
|
7
|
+
module GSC
|
|
8
|
+
class SkillPack
|
|
9
|
+
DEFAULT_SKILL_NAME = 'gsc'
|
|
10
|
+
|
|
11
|
+
attr_reader :options, :target_dir
|
|
12
|
+
|
|
13
|
+
def initialize(options = {}, target_dir = nil)
|
|
14
|
+
@options = options
|
|
15
|
+
@target_dir = target_dir || options[:dir]
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
def self.package(options = {}, target_dir = nil)
|
|
19
|
+
new(options, target_dir).package
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
def package
|
|
23
|
+
content_md = generate_skill_markdown
|
|
24
|
+
recipes = generate_recipes_json
|
|
25
|
+
|
|
26
|
+
destinations = resolve_destinations
|
|
27
|
+
|
|
28
|
+
installed_paths = []
|
|
29
|
+
|
|
30
|
+
unless @options[:dry_run]
|
|
31
|
+
destinations.each do |dest|
|
|
32
|
+
FileUtils.mkdir_p(dest[:dir])
|
|
33
|
+
skill_path = File.join(dest[:dir], 'SKILL.md')
|
|
34
|
+
recipe_path = File.join(dest[:dir], 'recipes.json')
|
|
35
|
+
|
|
36
|
+
File.write(skill_path, content_md, encoding: 'UTF-8')
|
|
37
|
+
File.write(recipe_path, JSON.pretty_generate(recipes), encoding: 'UTF-8')
|
|
38
|
+
|
|
39
|
+
installed_paths << {
|
|
40
|
+
platform: dest[:platform],
|
|
41
|
+
skill_path: skill_path,
|
|
42
|
+
recipe_path: recipe_path
|
|
43
|
+
}
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
# Agent verification check
|
|
48
|
+
verification = verify_agent_environment
|
|
49
|
+
|
|
50
|
+
{
|
|
51
|
+
skill_name: DEFAULT_SKILL_NAME,
|
|
52
|
+
version: GSC::VERSION,
|
|
53
|
+
dry_run: !!@options[:dry_run],
|
|
54
|
+
destinations_count: destinations.size,
|
|
55
|
+
installed_locations: installed_paths,
|
|
56
|
+
verification: verification,
|
|
57
|
+
recipes_count: recipes[:recipes].size,
|
|
58
|
+
skill_preview: content_md[0..400] + "..."
|
|
59
|
+
}
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
def generate_skill_markdown
|
|
63
|
+
<<~MARKDOWN
|
|
64
|
+
---
|
|
65
|
+
name: gsc
|
|
66
|
+
description: Automates Google Search Console, Google Indexing API, live URL index inspection, sitemap batch submission, search ranking analytics, E-E-A-T audits, image optimization, schema testing, and conversion-weighted revenue opportunity matrices.
|
|
67
|
+
---
|
|
68
|
+
|
|
69
|
+
# Google Search Console (GSC) CLI Agent Skill
|
|
70
|
+
|
|
71
|
+
This skill provides full-featured, zero-dependency autonomous search engine intelligence directly inside AI developer coding environments (Google Antigravity, Claude Code, Cursor, Windsurf).
|
|
72
|
+
|
|
73
|
+
## Core Capabilities & Quick Reference
|
|
74
|
+
|
|
75
|
+
### 1. Organic Search Performance & Analytics
|
|
76
|
+
- `gsc performance [domain] --json`: Overview clicks, impressions, CTR, average position.
|
|
77
|
+
- `gsc top-queries [domain] --limit 50 --json`: Top ranking search queries.
|
|
78
|
+
- `gsc top-pages [domain] --limit 50 --json`: Top ranking landing pages.
|
|
79
|
+
- `gsc opportunities [domain] --json`: High-opportunity queries in striking distance (positions 4–20).
|
|
80
|
+
- `gsc strike [domain] --json`: Tactical striking-distance keyword playbook.
|
|
81
|
+
- `gsc brand [domain] --json`: Brand vs non-brand organic search segmentation.
|
|
82
|
+
|
|
83
|
+
### 2. Live URL Indexing & Googlebot Inspection
|
|
84
|
+
- `gsc inspect <url> --json`: Real-time Googlebot crawl verdict and indexing status.
|
|
85
|
+
- `gsc index <url> --json`: Submit URL to Google Indexing API for rapid re-crawl.
|
|
86
|
+
- `gsc index-batch [sitemap] --json`: Bulk submit URLs from sitemap to Indexing API.
|
|
87
|
+
|
|
88
|
+
### 3. SEO Diagnostics & Technical Audits
|
|
89
|
+
- `gsc audit [domain] --json`: Comprehensive 360-degree organic health audit.
|
|
90
|
+
- `gsc report [domain] --html --json`: Generate single-file executive audit dashboard.
|
|
91
|
+
- `gsc image-seo <url|file> --json`: Audit WebP/AVIF formats, missing alt text, and CLS image dimensions.
|
|
92
|
+
- `gsc hreflang-check <url|file> --json`: International hreflang reciprocity and ISO syntax verification.
|
|
93
|
+
- `gsc eeat <url|file> --json`: Author E-E-A-T, credentials, reviewer attribution, and Knowledge Graph signals.
|
|
94
|
+
- `gsc rich-results <url|file> --json`: Test JSON-LD against Google Rich Results API rules.
|
|
95
|
+
- `gsc mobile-parity [domain] --json`: Cross-device Desktop vs Mobile position gap and responsive penalty audit.
|
|
96
|
+
|
|
97
|
+
### 4. Growth, Revenue & Strategy
|
|
98
|
+
- `gsc kw-value [domain] --json`: Conversion-weighted keyword revenue matrix ($ upside to Top 3).
|
|
99
|
+
- `gsc intent-shift [domain] --json`: Search intent drift & landing page mismatch detection.
|
|
100
|
+
- `gsc zombies [sitemap] --json`: Identify dead crawl-waste pages and generate 410/301 rules.
|
|
101
|
+
- `gsc watch [domain] --once --json`: Background rank drop and CTR crash monitoring watchdog.
|
|
102
|
+
- `gsc cite-sim <url> <query> --json`: AI Search Overviews citation probability simulator.
|
|
103
|
+
|
|
104
|
+
## Autonomous Execution Recipes for AI Agents
|
|
105
|
+
|
|
106
|
+
### Recipe 1: Diagnose Sudden Organic Traffic Drop
|
|
107
|
+
1. Run `gsc watch [domain] --once --json` to detect queries with severe rank drops (>= 2 positions).
|
|
108
|
+
2. Run `gsc mobile-parity [domain] --json` to check if drops are concentrated on mobile devices.
|
|
109
|
+
3. For dropped URLs, run `gsc inspect <url> --json` to check for Googlebot indexing failures.
|
|
110
|
+
|
|
111
|
+
### Recipe 2: High-ROI Striking Distance Acceleration
|
|
112
|
+
1. Run `gsc kw-value [domain] --json` to locate queries in positions 4–15 with highest monthly $ upside.
|
|
113
|
+
2. Run `gsc intent-shift [domain] --json` to verify searcher intent matches the landing page type.
|
|
114
|
+
3. Run `gsc rich-results <url> --json` to check if adding FAQ or Product schema can capture SERP real estate.
|
|
115
|
+
MARKDOWN
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
def generate_recipes_json
|
|
119
|
+
{
|
|
120
|
+
skill: DEFAULT_SKILL_NAME,
|
|
121
|
+
version: GSC::VERSION,
|
|
122
|
+
recipes: [
|
|
123
|
+
{
|
|
124
|
+
name: "diagnose_traffic_drop",
|
|
125
|
+
trigger: "User reports traffic drop or lost search rankings",
|
|
126
|
+
steps: [
|
|
127
|
+
{ command: "gsc watch {{domain}} --once --json", extract: "alerts" },
|
|
128
|
+
{ command: "gsc mobile-parity {{domain}} --json", extract: "disparities" }
|
|
129
|
+
]
|
|
130
|
+
},
|
|
131
|
+
{
|
|
132
|
+
name: "optimize_striking_distance",
|
|
133
|
+
trigger: "User asks how to increase revenue or capture high-ROI search keywords",
|
|
134
|
+
steps: [
|
|
135
|
+
{ command: "gsc kw-value {{domain}} --json", extract: "keywords" },
|
|
136
|
+
{ command: "gsc intent-shift {{domain}} --json", extract: "shifts" }
|
|
137
|
+
]
|
|
138
|
+
},
|
|
139
|
+
{
|
|
140
|
+
name: "technical_page_audit",
|
|
141
|
+
trigger: "User asks to audit an individual page or template",
|
|
142
|
+
steps: [
|
|
143
|
+
{ command: "gsc rich-results {{url}} --json", extract: "schemas" },
|
|
144
|
+
{ command: "gsc image-seo {{url}} --json", extract: "images" },
|
|
145
|
+
{ command: "gsc eeat {{url}} --json", extract: "signals" }
|
|
146
|
+
]
|
|
147
|
+
}
|
|
148
|
+
]
|
|
149
|
+
}
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
private
|
|
153
|
+
|
|
154
|
+
def resolve_destinations
|
|
155
|
+
destinations = []
|
|
156
|
+
|
|
157
|
+
# If explicit target directory provided
|
|
158
|
+
if @target_dir && !@target_dir.empty?
|
|
159
|
+
destinations << { platform: 'Custom Directory', dir: File.expand_path(@target_dir) }
|
|
160
|
+
return destinations
|
|
161
|
+
end
|
|
162
|
+
|
|
163
|
+
target = (@options[:target] || 'all').to_s.downcase
|
|
164
|
+
|
|
165
|
+
if target == 'all' || target == 'antigravity' || target == 'gemini'
|
|
166
|
+
gemini_dir = File.expand_path('~/.gemini/config/skills/gsc')
|
|
167
|
+
destinations << { platform: 'Google Antigravity / Gemini', dir: gemini_dir }
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
if target == 'all' || target == 'claude'
|
|
171
|
+
claude_dir = File.expand_path('~/.claude/skills/gsc')
|
|
172
|
+
destinations << { platform: 'Claude Code', dir: claude_dir }
|
|
173
|
+
end
|
|
174
|
+
|
|
175
|
+
if target == 'all' || target == 'workspace' || target == 'local'
|
|
176
|
+
local_dir = File.expand_path('.agents/skills/gsc', Dir.pwd)
|
|
177
|
+
destinations << { platform: 'Universal Workspace (.agents)', dir: local_dir }
|
|
178
|
+
end
|
|
179
|
+
|
|
180
|
+
destinations
|
|
181
|
+
end
|
|
182
|
+
|
|
183
|
+
def verify_agent_environment
|
|
184
|
+
bin = ENV['GSC_BIN_PATH'] || File.expand_path('~/.local/bin/gsc')
|
|
185
|
+
bin_exists = File.exist?(bin) && File.executable?(bin)
|
|
186
|
+
|
|
187
|
+
{
|
|
188
|
+
binary_path: bin,
|
|
189
|
+
binary_available: bin_exists,
|
|
190
|
+
ruby_version: RUBY_VERSION,
|
|
191
|
+
status: bin_exists ? 'READY_FOR_AI_AGENTS' : 'BINARY_NEEDS_INSTALL'
|
|
192
|
+
}
|
|
193
|
+
end
|
|
194
|
+
end
|
|
195
|
+
end
|