gsc-cli 2.0.2 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/AUTH.md +205 -0
- data/FUNDING.md +120 -0
- data/README.md +463 -299
- data/bin/gsc +29158 -4921
- data/dist/gsc +29158 -4921
- data/lib/gsc/aio_hunter.rb +343 -0
- data/lib/gsc/answer_synthesizer.rb +157 -0
- data/lib/gsc/api.rb +53 -1
- data/lib/gsc/auth.rb +26 -0
- data/lib/gsc/backlinks_manager.rb +96 -0
- data/lib/gsc/brand_segmenter.rb +140 -0
- data/lib/gsc/cache_manager.rb +806 -0
- data/lib/gsc/cannibalization_analyzer.rb +141 -0
- data/lib/gsc/canonical_chains.rb +367 -0
- data/lib/gsc/citation_simulator.rb +339 -0
- data/lib/gsc/cli/aio_hunter.rb +154 -0
- data/lib/gsc/cli/analytics.rb +788 -0
- data/lib/gsc/cli/audit.rb +1976 -0
- data/lib/gsc/cli/base.rb +384 -0
- data/lib/gsc/cli/cache.rb +266 -0
- data/lib/gsc/cli/canonical.rb +223 -0
- data/lib/gsc/cli/citation_simulator.rb +152 -0
- data/lib/gsc/cli/dashboard.rb +354 -0
- data/lib/gsc/cli/doctor.rb +129 -0
- data/lib/gsc/cli/eeat.rb +125 -0
- data/lib/gsc/cli/ga4.rb +852 -0
- data/lib/gsc/cli/growth.rb +650 -0
- data/lib/gsc/cli/hreflang.rb +164 -0
- data/lib/gsc/cli/image_seo.rb +162 -0
- data/lib/gsc/cli/indexing.rb +458 -0
- data/lib/gsc/cli/intent_shift.rb +125 -0
- data/lib/gsc/cli/keyword_value.rb +134 -0
- data/lib/gsc/cli/keywords.rb +795 -0
- data/lib/gsc/cli/landing_roi.rb +308 -0
- data/lib/gsc/cli/low_ctr.rb +213 -0
- data/lib/gsc/cli/mobile_parity.rb +150 -0
- data/lib/gsc/cli/report.rb +100 -0
- data/lib/gsc/cli/rich_results.rb +172 -0
- data/lib/gsc/cli/schema_generate.rb +149 -0
- data/lib/gsc/cli/seasonal.rb +232 -0
- data/lib/gsc/cli/security.rb +153 -0
- data/lib/gsc/cli/setup.rb +1291 -0
- data/lib/gsc/cli/sitemap_tree.rb +143 -0
- data/lib/gsc/cli/skill_pack.rb +62 -0
- data/lib/gsc/cli/soft_404.rb +199 -0
- data/lib/gsc/cli/sparkline.rb +227 -0
- data/lib/gsc/cli/watchdog.rb +150 -0
- data/lib/gsc/cli/zombie_purger.rb +208 -0
- data/lib/gsc/cli.rb +706 -5111
- data/lib/gsc/cli_advanced.rb +1513 -0
- data/lib/gsc/client.rb +17 -2
- data/lib/gsc/color.rb +16 -1
- data/lib/gsc/command_registry.rb +47 -9
- data/lib/gsc/config.rb +11 -2
- data/lib/gsc/content_gap.rb +112 -0
- data/lib/gsc/ctr_curve.rb +115 -0
- data/lib/gsc/decay_predictor.rb +322 -0
- data/lib/gsc/doctor.rb +434 -0
- data/lib/gsc/eeat_auditor.rb +428 -0
- data/lib/gsc/entity_auditor.rb +229 -0
- data/lib/gsc/firewall_scanner.rb +733 -0
- data/lib/gsc/geo_auditor.rb +368 -0
- data/lib/gsc/google_suggest.rb +109 -0
- data/lib/gsc/google_trends.rb +8 -1
- data/lib/gsc/heading_validator.rb +283 -0
- data/lib/gsc/hreflang_validator.rb +412 -0
- data/lib/gsc/image_seo.rb +286 -0
- data/lib/gsc/indexing_queue.rb +179 -0
- data/lib/gsc/indexnow.rb +93 -0
- data/lib/gsc/intent_shift.rb +188 -0
- data/lib/gsc/internal_links.rb +249 -0
- data/lib/gsc/keyword_value.rb +191 -0
- data/lib/gsc/landing_roi.rb +195 -0
- data/lib/gsc/llms_generator.rb +425 -0
- data/lib/gsc/low_ctr_rewriter.rb +408 -0
- data/lib/gsc/mobile_parity.rb +222 -0
- data/lib/gsc/network_tracer.rb +93 -0
- data/lib/gsc/open_page_rank.rb +72 -0
- data/lib/gsc/page_analyzer.rb +47 -7
- data/lib/gsc/page_comparator.rb +108 -0
- data/lib/gsc/page_speed.rb +110 -0
- data/lib/gsc/prompts.rb +38 -29
- data/lib/gsc/questions_harvester.rb +178 -0
- data/lib/gsc/report_generator.rb +461 -0
- data/lib/gsc/rich_results.rb +388 -0
- data/lib/gsc/robots_checker.rb +114 -0
- data/lib/gsc/schema_generator.rb +788 -0
- data/lib/gsc/schema_validator.rb +120 -0
- data/lib/gsc/seasonal_predictor.rb +381 -0
- data/lib/gsc/security_scanner.rb +496 -0
- data/lib/gsc/serp_feature_detector.rb +359 -0
- data/lib/gsc/serp_preview.rb +152 -0
- data/lib/gsc/site_crawler.rb +113 -21
- data/lib/gsc/sitemap_loader.rb +15 -4
- data/lib/gsc/sitemap_tree.rb +301 -0
- data/lib/gsc/skill_pack.rb +195 -0
- data/lib/gsc/soft_404_analyzer.rb +385 -0
- data/lib/gsc/sparkline.rb +171 -0
- data/lib/gsc/speed_correlator.rb +416 -0
- data/lib/gsc/striking_playbook.rb +190 -0
- data/lib/gsc/title_optimizer.rb +420 -0
- data/lib/gsc/vault.rb +260 -0
- data/lib/gsc/version.rb +1 -1
- data/lib/gsc/watchdog.rb +235 -0
- data/lib/gsc/zombie_purger.rb +366 -0
- data/lib/gsc.rb +144 -0
- metadata +91 -2
|
@@ -0,0 +1,286 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'net/http'
|
|
5
|
+
require 'uri'
|
|
6
|
+
require 'json'
|
|
7
|
+
|
|
8
|
+
module GSC
|
|
9
|
+
class ImageSeo
|
|
10
|
+
MODERN_FORMATS = %w[webp avif svg].freeze
|
|
11
|
+
LEGACY_FORMATS = %w[png jpg jpeg gif bmp webp_fallback].freeze
|
|
12
|
+
|
|
13
|
+
attr_reader :options, :url, :html
|
|
14
|
+
|
|
15
|
+
def initialize(options = {})
|
|
16
|
+
@options = options
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def self.audit(target, options = {})
|
|
20
|
+
new(options).audit(target)
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
def audit(target)
|
|
24
|
+
@url, @html = load_content(target)
|
|
25
|
+
images = extract_images(@html, @url)
|
|
26
|
+
|
|
27
|
+
# Optionally check file size via HTTP HEAD
|
|
28
|
+
if @options[:check_size]
|
|
29
|
+
audit_image_sizes(images)
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
evaluate_health(images)
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
private
|
|
36
|
+
|
|
37
|
+
def load_content(target)
|
|
38
|
+
target_str = target.to_s.strip
|
|
39
|
+
if target_str.match?(%r{^https?://})
|
|
40
|
+
uri = URI.parse(target_str)
|
|
41
|
+
req = Net::HTTP::Get.new(uri)
|
|
42
|
+
req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36'
|
|
43
|
+
req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
|
|
44
|
+
|
|
45
|
+
res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 8, read_timeout: 10) do |http|
|
|
46
|
+
http.request(req)
|
|
47
|
+
end
|
|
48
|
+
[target_str, res.body.to_s.dup.force_encoding('UTF-8').scrub]
|
|
49
|
+
elsif File.exist?(target_str)
|
|
50
|
+
[target_str, File.read(target_str, encoding: 'UTF-8')]
|
|
51
|
+
else
|
|
52
|
+
[target_str.start_with?('http') ? target_str : 'local-document', target_str]
|
|
53
|
+
end
|
|
54
|
+
rescue StandardError => e
|
|
55
|
+
[target_str.to_s, "<html><body><!-- Error: #{e.message} --></body></html>"]
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
def extract_images(html_str, base_url)
|
|
59
|
+
images = []
|
|
60
|
+
index = 0
|
|
61
|
+
|
|
62
|
+
# Match all <img> tags
|
|
63
|
+
html_str.scan(/<img\b([^>]*?)>/im) do |match|
|
|
64
|
+
tag_attrs = match.first
|
|
65
|
+
src = extract_attr(tag_attrs, 'src')
|
|
66
|
+
next if src.nil? || src.empty?
|
|
67
|
+
|
|
68
|
+
alt = extract_attr(tag_attrs, 'alt')
|
|
69
|
+
width = extract_attr(tag_attrs, 'width')
|
|
70
|
+
height = extract_attr(tag_attrs, 'height')
|
|
71
|
+
loading = extract_attr(tag_attrs, 'loading')&.downcase
|
|
72
|
+
fetchpriority = extract_attr(tag_attrs, 'fetchpriority')&.downcase
|
|
73
|
+
decoding = extract_attr(tag_attrs, 'decoding')&.downcase
|
|
74
|
+
role = extract_attr(tag_attrs, 'role')&.downcase
|
|
75
|
+
aria_hidden = extract_attr(tag_attrs, 'aria-hidden')&.downcase
|
|
76
|
+
|
|
77
|
+
is_decorative = (role == 'presentation' || role == 'none' || aria_hidden == 'true')
|
|
78
|
+
resolved_src = resolve_url(src, base_url)
|
|
79
|
+
ext = extract_extension(resolved_src)
|
|
80
|
+
|
|
81
|
+
is_hero = (index == 0)
|
|
82
|
+
|
|
83
|
+
issues = []
|
|
84
|
+
issues << :missing_alt if alt.nil? && !is_decorative
|
|
85
|
+
issues << :empty_alt if alt == '' && !is_decorative
|
|
86
|
+
issues << :long_alt if alt && alt.length > 125
|
|
87
|
+
issues << :legacy_format if LEGACY_FORMATS.include?(ext)
|
|
88
|
+
issues << :missing_dimensions if (width.nil? || height.nil?)
|
|
89
|
+
issues << :lcp_lazy_loaded if is_hero && loading == 'lazy'
|
|
90
|
+
issues << :hero_missing_priority if is_hero && fetchpriority != 'high'
|
|
91
|
+
issues << :missing_lazy if !is_hero && loading != 'lazy'
|
|
92
|
+
|
|
93
|
+
images << {
|
|
94
|
+
index: index + 1,
|
|
95
|
+
src: resolved_src,
|
|
96
|
+
raw_src: src,
|
|
97
|
+
alt: alt,
|
|
98
|
+
width: width ? width.to_i : nil,
|
|
99
|
+
height: height ? height.to_i : nil,
|
|
100
|
+
format: ext,
|
|
101
|
+
loading: loading,
|
|
102
|
+
fetchpriority: fetchpriority,
|
|
103
|
+
decoding: decoding,
|
|
104
|
+
is_decorative: is_decorative,
|
|
105
|
+
is_hero: is_hero,
|
|
106
|
+
issues: issues,
|
|
107
|
+
issue_count: issues.size,
|
|
108
|
+
picture_tag_snippet: build_picture_snippet(resolved_src, alt, width, height, is_hero),
|
|
109
|
+
nextjs_snippet: build_nextjs_snippet(resolved_src, alt, width, height, is_hero)
|
|
110
|
+
}
|
|
111
|
+
index += 1
|
|
112
|
+
end
|
|
113
|
+
|
|
114
|
+
images
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
def audit_image_sizes(images)
|
|
118
|
+
images.each do |img|
|
|
119
|
+
next unless img[:src].match?(%r{^https?://})
|
|
120
|
+
begin
|
|
121
|
+
uri = URI.parse(img[:src])
|
|
122
|
+
res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 4, read_timeout: 4) do |http|
|
|
123
|
+
http.head(uri.request_uri)
|
|
124
|
+
end
|
|
125
|
+
if res['content-length']
|
|
126
|
+
bytes = res['content-length'].to_i
|
|
127
|
+
img[:bytes] = bytes
|
|
128
|
+
img[:size_kb] = (bytes / 1024.0).round(1)
|
|
129
|
+
img[:issues] << :oversized_payload if bytes > 200 * 1024
|
|
130
|
+
img[:issues] << :critical_payload if bytes > 500 * 1024
|
|
131
|
+
end
|
|
132
|
+
rescue StandardError
|
|
133
|
+
img[:bytes] = nil
|
|
134
|
+
end
|
|
135
|
+
end
|
|
136
|
+
end
|
|
137
|
+
|
|
138
|
+
def evaluate_health(images)
|
|
139
|
+
total = images.size
|
|
140
|
+
if total.zero?
|
|
141
|
+
return {
|
|
142
|
+
url: @url,
|
|
143
|
+
total_images: 0,
|
|
144
|
+
health_score: 100.0,
|
|
145
|
+
grade: 'A+',
|
|
146
|
+
metrics: { alt_coverage_pct: 100.0, dimension_coverage_pct: 100.0, modern_format_pct: 100.0 },
|
|
147
|
+
issues_summary: {},
|
|
148
|
+
images: []
|
|
149
|
+
}
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
valid_alt_count = images.count { |img| !img[:issues].include?(:missing_alt) && !img[:issues].include?(:empty_alt) }
|
|
153
|
+
valid_dims_count = images.count { |img| !img[:issues].include?(:missing_dimensions) }
|
|
154
|
+
modern_format_count = images.count { |img| MODERN_FORMATS.include?(img[:format]) }
|
|
155
|
+
proper_lazy_count = images.count { |img| (!img[:is_hero] && img[:loading] == 'lazy') || (img[:is_hero] && img[:loading] != 'lazy') }
|
|
156
|
+
|
|
157
|
+
alt_coverage = ((valid_alt_count.to_f / total) * 100.0).round(1)
|
|
158
|
+
dim_coverage = ((valid_dims_count.to_f / total) * 100.0).round(1)
|
|
159
|
+
format_coverage = ((modern_format_count.to_f / total) * 100.0).round(1)
|
|
160
|
+
lazy_coverage = ((proper_lazy_count.to_f / total) * 100.0).round(1)
|
|
161
|
+
|
|
162
|
+
# Deductions
|
|
163
|
+
score = 100.0
|
|
164
|
+
score -= (100.0 - alt_coverage) * 0.35 # Alt text is critical (up to 35 pts)
|
|
165
|
+
score -= (100.0 - dim_coverage) * 0.25 # CLS dimensions (up to 25 pts)
|
|
166
|
+
score -= (100.0 - format_coverage) * 0.25 # Next-gen formats (up to 25 pts)
|
|
167
|
+
score -= (100.0 - lazy_coverage) * 0.15 # Lazy loading & LCP priority (up to 15 pts)
|
|
168
|
+
|
|
169
|
+
score = [[score.round(1), 100.0].min, 0.0].max
|
|
170
|
+
grade = compute_grade(score)
|
|
171
|
+
|
|
172
|
+
summary = {
|
|
173
|
+
missing_alt: images.count { |i| i[:issues].include?(:missing_alt) },
|
|
174
|
+
empty_alt: images.count { |i| i[:issues].include?(:empty_alt) },
|
|
175
|
+
missing_dimensions: images.count { |i| i[:issues].include?(:missing_dimensions) },
|
|
176
|
+
legacy_format: images.count { |i| i[:issues].include?(:legacy_format) },
|
|
177
|
+
lcp_lazy_loaded: images.count { |i| i[:issues].include?(:lcp_lazy_loaded) },
|
|
178
|
+
hero_missing_priority: images.count { |i| i[:issues].include?(:hero_missing_priority) },
|
|
179
|
+
oversized: images.count { |i| i[:issues].include?(:oversized_payload) }
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
prescriptions = generate_prescriptions(summary, images)
|
|
183
|
+
|
|
184
|
+
limit = (@options[:limit] || 20).to_i
|
|
185
|
+
displayed = images.first(limit)
|
|
186
|
+
|
|
187
|
+
{
|
|
188
|
+
url: @url,
|
|
189
|
+
total_images: total,
|
|
190
|
+
health_score: score,
|
|
191
|
+
grade: grade,
|
|
192
|
+
metrics: {
|
|
193
|
+
alt_coverage_pct: alt_coverage,
|
|
194
|
+
dimension_coverage_pct: dim_coverage,
|
|
195
|
+
modern_format_pct: format_coverage,
|
|
196
|
+
lazy_loading_pct: lazy_coverage
|
|
197
|
+
},
|
|
198
|
+
issues_summary: summary,
|
|
199
|
+
prescriptions: prescriptions,
|
|
200
|
+
images: displayed
|
|
201
|
+
}
|
|
202
|
+
end
|
|
203
|
+
|
|
204
|
+
def generate_prescriptions(sum, images)
|
|
205
|
+
recs = []
|
|
206
|
+
|
|
207
|
+
if sum[:missing_alt] > 0
|
|
208
|
+
recs << "Add descriptive, keyword-rich alt text to #{sum[:missing_alt]} images missing the alt attribute."
|
|
209
|
+
end
|
|
210
|
+
|
|
211
|
+
if sum[:missing_dimensions] > 0
|
|
212
|
+
recs << "Specify explicit width and height attributes on #{sum[:missing_dimensions]} images to eliminate Cumulative Layout Shift (CLS)."
|
|
213
|
+
end
|
|
214
|
+
|
|
215
|
+
if sum[:legacy_format] > 0
|
|
216
|
+
recs << "Convert #{sum[:legacy_format]} PNG/JPEG images to modern WebP or AVIF formats for 65–80% byte reduction."
|
|
217
|
+
end
|
|
218
|
+
|
|
219
|
+
if sum[:lcp_lazy_loaded] > 0
|
|
220
|
+
recs << "Remove loading=\"lazy\" from the primary above-the-fold hero image to accelerate Largest Contentful Paint (LCP)."
|
|
221
|
+
end
|
|
222
|
+
|
|
223
|
+
if sum[:hero_missing_priority] > 0
|
|
224
|
+
recs << "Add fetchpriority=\"high\" to the primary hero image to trigger immediate preloading by the browser."
|
|
225
|
+
end
|
|
226
|
+
|
|
227
|
+
recs
|
|
228
|
+
end
|
|
229
|
+
|
|
230
|
+
def build_picture_snippet(src, alt, width, height, is_hero)
|
|
231
|
+
base = src.sub(/\.[^.]+$/, '')
|
|
232
|
+
w_attr = width ? " width=\"#{width}\"" : ""
|
|
233
|
+
h_attr = height ? " height=\"#{height}\"" : ""
|
|
234
|
+
loading_attr = is_hero ? " fetchpriority=\"high\"" : " loading=\"lazy\" decoding=\"async\""
|
|
235
|
+
|
|
236
|
+
<<~HTML.strip
|
|
237
|
+
<picture>
|
|
238
|
+
<source srcset="#{base}.avif" type="image/avif">
|
|
239
|
+
<source srcset="#{base}.webp" type="image/webp">
|
|
240
|
+
<img src="#{src}" alt="#{alt || 'Descriptive image text'}"#{w_attr}#{h_attr}#{loading_attr}>
|
|
241
|
+
</picture>
|
|
242
|
+
HTML
|
|
243
|
+
end
|
|
244
|
+
|
|
245
|
+
def build_nextjs_snippet(src, alt, width, height, is_hero)
|
|
246
|
+
w = width || 800
|
|
247
|
+
h = height || 600
|
|
248
|
+
priority_attr = is_hero ? " priority" : ""
|
|
249
|
+
|
|
250
|
+
"<Image src=\"#{src}\" alt=\"#{alt || 'Descriptive image text'}\" width={#{w}} height={#{h}}#{priority_attr} />"
|
|
251
|
+
end
|
|
252
|
+
|
|
253
|
+
def compute_grade(score)
|
|
254
|
+
case score
|
|
255
|
+
when 90.0..100.0 then 'A+'
|
|
256
|
+
when 80.0...90.0 then 'A'
|
|
257
|
+
when 70.0...80.0 then 'B'
|
|
258
|
+
when 55.0...70.0 then 'C'
|
|
259
|
+
when 40.0...55.0 then 'D'
|
|
260
|
+
else 'F'
|
|
261
|
+
end
|
|
262
|
+
end
|
|
263
|
+
|
|
264
|
+
def extract_attr(tag_str, attr_name)
|
|
265
|
+
if tag_str =~ /\b#{attr_name}\s*=\s*(['"])(.*?)\1/i
|
|
266
|
+
$2.strip
|
|
267
|
+
elsif tag_str =~ /\b#{attr_name}\s*=\s*([^\s>]+)/i
|
|
268
|
+
$1.strip
|
|
269
|
+
else
|
|
270
|
+
nil
|
|
271
|
+
end
|
|
272
|
+
end
|
|
273
|
+
|
|
274
|
+
def resolve_url(src, base)
|
|
275
|
+
return src if src.match?(%r{^https?://})
|
|
276
|
+
URI.join(base, src).to_s
|
|
277
|
+
rescue StandardError
|
|
278
|
+
src
|
|
279
|
+
end
|
|
280
|
+
|
|
281
|
+
def extract_extension(url_str)
|
|
282
|
+
clean = url_str.split('?').first.split('#').first
|
|
283
|
+
File.extname(clean).sub(/^\./, '').downcase
|
|
284
|
+
end
|
|
285
|
+
end
|
|
286
|
+
end
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'json'
|
|
4
|
+
require 'fileutils'
|
|
5
|
+
require 'date'
|
|
6
|
+
require 'time'
|
|
7
|
+
|
|
8
|
+
module GSC
|
|
9
|
+
class IndexingQueue
|
|
10
|
+
QUEUE_FILE = File.join(Config::CONFIG_DIR, 'indexing_queue.json')
|
|
11
|
+
DAILY_LIMIT = 200
|
|
12
|
+
|
|
13
|
+
attr_reader :state
|
|
14
|
+
|
|
15
|
+
def initialize(file_path = QUEUE_FILE)
|
|
16
|
+
@file_path = file_path
|
|
17
|
+
@state = load_state
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
def add_urls(urls)
|
|
21
|
+
normalized = Array(urls).map(&:to_s).map(&:strip).reject(&:empty?).uniq
|
|
22
|
+
# Only keep valid http(s) URLs
|
|
23
|
+
valid_urls = normalized.select { |u| u.start_with?('http://', 'https://') }
|
|
24
|
+
|
|
25
|
+
existing_pending = @state['pending'] || []
|
|
26
|
+
new_urls = valid_urls - existing_pending
|
|
27
|
+
|
|
28
|
+
@state['pending'] = existing_pending + new_urls
|
|
29
|
+
save_state
|
|
30
|
+
new_urls.size
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
def status
|
|
34
|
+
check_quota_reset!
|
|
35
|
+
used = @state['daily_quota_used'] || 0
|
|
36
|
+
remaining = [DAILY_LIMIT - used, 0].max
|
|
37
|
+
|
|
38
|
+
{
|
|
39
|
+
pending_count: (@state['pending'] || []).size,
|
|
40
|
+
submitted_count: (@state['submitted'] || []).size,
|
|
41
|
+
failed_count: (@state['failed'] || []).size,
|
|
42
|
+
daily_quota_limit: DAILY_LIMIT,
|
|
43
|
+
daily_quota_used: used,
|
|
44
|
+
daily_quota_remaining: remaining,
|
|
45
|
+
last_reset_date: @state['last_reset_date']
|
|
46
|
+
}
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
def clear(scope = :all)
|
|
50
|
+
if scope == :pending
|
|
51
|
+
@state['pending'] = []
|
|
52
|
+
else
|
|
53
|
+
@state['pending'] = []
|
|
54
|
+
@state['submitted'] = []
|
|
55
|
+
@state['failed'] = []
|
|
56
|
+
end
|
|
57
|
+
save_state
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
def process_batch(api, batch_size: 50, dry_run: false, delay_sec: 0.15)
|
|
61
|
+
check_quota_reset!
|
|
62
|
+
used = @state['daily_quota_used'] || 0
|
|
63
|
+
remaining_quota = [DAILY_LIMIT - used, 0].max
|
|
64
|
+
|
|
65
|
+
if remaining_quota <= 0 && !dry_run
|
|
66
|
+
return {
|
|
67
|
+
status: :quota_exhausted,
|
|
68
|
+
message: "Daily quota of #{DAILY_LIMIT} requests reached for today (#{@state['last_reset_date']}). Next reset at 00:00 UTC.",
|
|
69
|
+
processed: 0,
|
|
70
|
+
remaining_in_queue: (@state['pending'] || []).size
|
|
71
|
+
}
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
to_process_count = [batch_size.to_i, remaining_quota].min
|
|
75
|
+
urls = (@state['pending'] || []).shift(to_process_count)
|
|
76
|
+
|
|
77
|
+
if urls.empty?
|
|
78
|
+
return {
|
|
79
|
+
status: :queue_empty,
|
|
80
|
+
message: 'Indexing queue is empty. Use `gsc index-batch add <url|sitemap>` to enqueue URLs.',
|
|
81
|
+
processed: 0,
|
|
82
|
+
remaining_in_queue: 0
|
|
83
|
+
}
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
results = []
|
|
87
|
+
successful = 0
|
|
88
|
+
failed = 0
|
|
89
|
+
|
|
90
|
+
urls.each_with_index do |url, idx|
|
|
91
|
+
if dry_run
|
|
92
|
+
results << { url: url, status: 'DRY_RUN', ok: true }
|
|
93
|
+
successful += 1
|
|
94
|
+
next
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
res = api.publish_url(url, 'URL_UPDATED')
|
|
98
|
+
if res[:ok]
|
|
99
|
+
successful += 1
|
|
100
|
+
@state['submitted'] ||= []
|
|
101
|
+
@state['submitted'] << {
|
|
102
|
+
url: url,
|
|
103
|
+
status: res[:status],
|
|
104
|
+
submitted_at: Time.now.utc.iso8601
|
|
105
|
+
}
|
|
106
|
+
@state['daily_quota_used'] = (@state['daily_quota_used'] || 0) + 1
|
|
107
|
+
results << { url: url, status: 'SUCCESS', http_code: res[:status], ok: true }
|
|
108
|
+
else
|
|
109
|
+
failed += 1
|
|
110
|
+
@state['failed'] ||= []
|
|
111
|
+
@state['failed'] << {
|
|
112
|
+
url: url,
|
|
113
|
+
error: res[:data],
|
|
114
|
+
failed_at: Time.now.utc.iso8601
|
|
115
|
+
}
|
|
116
|
+
results << { url: url, status: 'FAILED', error: res[:data], ok: false }
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
sleep(delay_sec) if delay_sec > 0 && idx < urls.size - 1
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
# In dry_run, restore pending list so URLs aren't lost
|
|
123
|
+
if dry_run
|
|
124
|
+
@state['pending'] = urls + (@state['pending'] || [])
|
|
125
|
+
else
|
|
126
|
+
save_state
|
|
127
|
+
end
|
|
128
|
+
|
|
129
|
+
{
|
|
130
|
+
status: :success,
|
|
131
|
+
processed: urls.size,
|
|
132
|
+
successful: successful,
|
|
133
|
+
failed: failed,
|
|
134
|
+
dry_run: dry_run,
|
|
135
|
+
remaining_in_queue: (@state['pending'] || []).size,
|
|
136
|
+
daily_quota_remaining: dry_run ? remaining_quota : [DAILY_LIMIT - @state['daily_quota_used'], 0].max,
|
|
137
|
+
results: results
|
|
138
|
+
}
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
private
|
|
142
|
+
|
|
143
|
+
def check_quota_reset!
|
|
144
|
+
today = Date.today.to_s
|
|
145
|
+
if @state['last_reset_date'] != today
|
|
146
|
+
@state['last_reset_date'] = today
|
|
147
|
+
@state['daily_quota_used'] = 0
|
|
148
|
+
save_state
|
|
149
|
+
end
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
def load_state
|
|
153
|
+
return default_state unless File.exist?(@file_path)
|
|
154
|
+
|
|
155
|
+
data = JSON.parse(File.read(@file_path))
|
|
156
|
+
data.is_a?(Hash) ? data : default_state
|
|
157
|
+
rescue StandardError
|
|
158
|
+
default_state
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
def save_state
|
|
162
|
+
FileUtils.mkdir_p(File.dirname(@file_path))
|
|
163
|
+
File.write(@file_path, JSON.pretty_generate(@state))
|
|
164
|
+
rescue StandardError => e
|
|
165
|
+
# Silently handle disk write errors
|
|
166
|
+
end
|
|
167
|
+
|
|
168
|
+
def default_state
|
|
169
|
+
{
|
|
170
|
+
'daily_quota_limit' => DAILY_LIMIT,
|
|
171
|
+
'daily_quota_used' => 0,
|
|
172
|
+
'last_reset_date' => Date.today.to_s,
|
|
173
|
+
'pending' => [],
|
|
174
|
+
'submitted' => [],
|
|
175
|
+
'failed' => []
|
|
176
|
+
}
|
|
177
|
+
end
|
|
178
|
+
end
|
|
179
|
+
end
|
data/lib/gsc/indexnow.rb
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'net/http'
|
|
4
|
+
require 'uri'
|
|
5
|
+
require 'json'
|
|
6
|
+
require 'openssl'
|
|
7
|
+
|
|
8
|
+
module GSC
|
|
9
|
+
class IndexNow
|
|
10
|
+
ENDPOINT = 'https://api.indexnow.org/indexnow'
|
|
11
|
+
|
|
12
|
+
def self.generate_key
|
|
13
|
+
OpenSSL::Random.random_bytes(16).unpack1('H*')
|
|
14
|
+
end
|
|
15
|
+
|
|
16
|
+
def self.get_or_create_key
|
|
17
|
+
existing = Config.get('indexnow_key') || ENV['INDEXNOW_KEY']
|
|
18
|
+
return existing if existing && !existing.strip.empty?
|
|
19
|
+
|
|
20
|
+
new_key = generate_key
|
|
21
|
+
Config.set('indexnow_key', new_key)
|
|
22
|
+
new_key
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def self.set_key(key)
|
|
26
|
+
clean = key.to_s.strip
|
|
27
|
+
Config.set('indexnow_key', clean)
|
|
28
|
+
clean
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
def self.submit(urls, key: nil, host: nil)
|
|
32
|
+
urls = Array(urls).map(&:to_s).map(&:strip).reject(&:empty?)
|
|
33
|
+
raise 'No URLs provided for IndexNow submission' if urls.empty?
|
|
34
|
+
|
|
35
|
+
first_uri = URI.parse(urls.first) rescue nil
|
|
36
|
+
detected_host = host || (first_uri ? first_uri.host : Config.default_domain)
|
|
37
|
+
raise 'Could not determine host for IndexNow submission. Please provide full URLs (e.g. https://example.com/page)' unless detected_host
|
|
38
|
+
|
|
39
|
+
detected_host = detected_host.sub(%r{^https?://}, '').sub(/^sc-domain:/, '').chomp('/')
|
|
40
|
+
|
|
41
|
+
active_key = key || get_or_create_key
|
|
42
|
+
|
|
43
|
+
payload = {
|
|
44
|
+
host: detected_host,
|
|
45
|
+
key: active_key,
|
|
46
|
+
keyLocation: "https://#{detected_host}/#{active_key}.txt",
|
|
47
|
+
urlList: urls
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
uri = URI(ENDPOINT)
|
|
51
|
+
http = Net::HTTP.new(uri.host, uri.port)
|
|
52
|
+
http.use_ssl = true
|
|
53
|
+
http.open_timeout = 10
|
|
54
|
+
http.read_timeout = 20
|
|
55
|
+
|
|
56
|
+
req = Net::HTTP::Post.new(uri.request_uri)
|
|
57
|
+
req['Content-Type'] = 'application/json; charset=utf-8'
|
|
58
|
+
req['User-Agent'] = 'gsc-cli IndexNow/1.0'
|
|
59
|
+
req.body = JSON.generate(payload)
|
|
60
|
+
|
|
61
|
+
res = http.request(req)
|
|
62
|
+
|
|
63
|
+
status_msg = case res.code.to_i
|
|
64
|
+
when 200 then 'OK (URLs submitted successfully)'
|
|
65
|
+
when 202 then 'Accepted (Key pending verification)'
|
|
66
|
+
when 400 then 'Bad Request (Invalid JSON or URL format)'
|
|
67
|
+
when 403 then "Forbidden (Key invalid or https://#{detected_host}/#{active_key}.txt missing)"
|
|
68
|
+
when 422 then 'Unprocessable Entity (URLs do not match host)'
|
|
69
|
+
when 429 then 'Too Many Requests'
|
|
70
|
+
else "HTTP #{res.code}"
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
{
|
|
74
|
+
success: [200, 202].include?(res.code.to_i),
|
|
75
|
+
http_code: res.code.to_i,
|
|
76
|
+
message: status_msg,
|
|
77
|
+
host: detected_host,
|
|
78
|
+
key: active_key,
|
|
79
|
+
key_location: "https://#{detected_host}/#{active_key}.txt",
|
|
80
|
+
submitted_urls: urls.size,
|
|
81
|
+
urls: urls
|
|
82
|
+
}
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
def self.submit_sitemap(sitemap_path_or_url, key: nil, limit: nil)
|
|
86
|
+
urls = SitemapLoader.resolve_urls(sitemap_path_or_url, quiet: true)
|
|
87
|
+
raise "No URLs found in sitemap: #{sitemap_path_or_url}" if urls.empty?
|
|
88
|
+
|
|
89
|
+
urls = urls.first(limit) if limit && limit.positive?
|
|
90
|
+
submit(urls, key: key)
|
|
91
|
+
end
|
|
92
|
+
end
|
|
93
|
+
end
|