gsc-cli 2.1.0 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +410 -414
- data/bin/gsc +28067 -5661
- data/dist/gsc +29121 -5046
- data/lib/gsc/aio_hunter.rb +343 -0
- data/lib/gsc/answer_synthesizer.rb +157 -0
- data/lib/gsc/api.rb +53 -1
- data/lib/gsc/auth.rb +26 -0
- data/lib/gsc/brand_segmenter.rb +140 -0
- data/lib/gsc/cache_manager.rb +806 -0
- data/lib/gsc/cannibalization_analyzer.rb +141 -0
- data/lib/gsc/canonical_chains.rb +367 -0
- data/lib/gsc/citation_simulator.rb +339 -0
- data/lib/gsc/cli/aio_hunter.rb +154 -0
- data/lib/gsc/cli/analytics.rb +788 -0
- data/lib/gsc/cli/audit.rb +1976 -0
- data/lib/gsc/cli/base.rb +384 -0
- data/lib/gsc/cli/cache.rb +266 -0
- data/lib/gsc/cli/canonical.rb +223 -0
- data/lib/gsc/cli/citation_simulator.rb +152 -0
- data/lib/gsc/cli/dashboard.rb +354 -0
- data/lib/gsc/cli/doctor.rb +129 -0
- data/lib/gsc/cli/eeat.rb +125 -0
- data/lib/gsc/cli/ga4.rb +852 -0
- data/lib/gsc/cli/growth.rb +650 -0
- data/lib/gsc/cli/hreflang.rb +164 -0
- data/lib/gsc/cli/image_seo.rb +162 -0
- data/lib/gsc/cli/indexing.rb +458 -0
- data/lib/gsc/cli/intent_shift.rb +125 -0
- data/lib/gsc/cli/keyword_value.rb +134 -0
- data/lib/gsc/cli/keywords.rb +795 -0
- data/lib/gsc/cli/landing_roi.rb +308 -0
- data/lib/gsc/cli/low_ctr.rb +213 -0
- data/lib/gsc/cli/mobile_parity.rb +150 -0
- data/lib/gsc/cli/report.rb +100 -0
- data/lib/gsc/cli/rich_results.rb +172 -0
- data/lib/gsc/cli/schema_generate.rb +149 -0
- data/lib/gsc/cli/seasonal.rb +232 -0
- data/lib/gsc/cli/security.rb +153 -0
- data/lib/gsc/cli/setup.rb +1291 -0
- data/lib/gsc/cli/sitemap_tree.rb +143 -0
- data/lib/gsc/cli/skill_pack.rb +62 -0
- data/lib/gsc/cli/soft_404.rb +199 -0
- data/lib/gsc/cli/sparkline.rb +227 -0
- data/lib/gsc/cli/watchdog.rb +150 -0
- data/lib/gsc/cli/zombie_purger.rb +208 -0
- data/lib/gsc/cli.rb +707 -5265
- data/lib/gsc/cli_advanced.rb +987 -44
- data/lib/gsc/client.rb +17 -2
- data/lib/gsc/color.rb +16 -1
- data/lib/gsc/command_registry.rb +47 -9
- data/lib/gsc/config.rb +2 -2
- data/lib/gsc/ctr_curve.rb +115 -0
- data/lib/gsc/decay_predictor.rb +322 -0
- data/lib/gsc/doctor.rb +434 -0
- data/lib/gsc/eeat_auditor.rb +428 -0
- data/lib/gsc/entity_auditor.rb +229 -0
- data/lib/gsc/firewall_scanner.rb +733 -0
- data/lib/gsc/geo_auditor.rb +368 -0
- data/lib/gsc/google_trends.rb +8 -1
- data/lib/gsc/heading_validator.rb +283 -0
- data/lib/gsc/hreflang_validator.rb +412 -0
- data/lib/gsc/image_seo.rb +286 -0
- data/lib/gsc/indexing_queue.rb +179 -0
- data/lib/gsc/indexnow.rb +93 -0
- data/lib/gsc/intent_shift.rb +188 -0
- data/lib/gsc/internal_links.rb +153 -36
- data/lib/gsc/keyword_value.rb +191 -0
- data/lib/gsc/landing_roi.rb +195 -0
- data/lib/gsc/llms_generator.rb +343 -22
- data/lib/gsc/low_ctr_rewriter.rb +408 -0
- data/lib/gsc/mobile_parity.rb +222 -0
- data/lib/gsc/network_tracer.rb +8 -1
- data/lib/gsc/page_analyzer.rb +47 -7
- data/lib/gsc/prompts.rb +38 -29
- data/lib/gsc/questions_harvester.rb +178 -0
- data/lib/gsc/report_generator.rb +461 -0
- data/lib/gsc/rich_results.rb +388 -0
- data/lib/gsc/robots_checker.rb +46 -15
- data/lib/gsc/schema_generator.rb +788 -0
- data/lib/gsc/schema_validator.rb +36 -38
- data/lib/gsc/seasonal_predictor.rb +381 -0
- data/lib/gsc/security_scanner.rb +496 -0
- data/lib/gsc/serp_feature_detector.rb +359 -0
- data/lib/gsc/serp_preview.rb +108 -22
- data/lib/gsc/site_crawler.rb +113 -21
- data/lib/gsc/sitemap_loader.rb +15 -4
- data/lib/gsc/sitemap_tree.rb +301 -0
- data/lib/gsc/skill_pack.rb +195 -0
- data/lib/gsc/soft_404_analyzer.rb +385 -0
- data/lib/gsc/sparkline.rb +171 -0
- data/lib/gsc/speed_correlator.rb +416 -0
- data/lib/gsc/striking_playbook.rb +190 -0
- data/lib/gsc/title_optimizer.rb +420 -0
- data/lib/gsc/vault.rb +260 -0
- data/lib/gsc/version.rb +1 -1
- data/lib/gsc/watchdog.rb +235 -0
- data/lib/gsc/zombie_purger.rb +366 -0
- data/lib/gsc.rb +118 -0
- metadata +75 -1
|
@@ -0,0 +1,788 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'json'
|
|
5
|
+
require 'uri'
|
|
6
|
+
require 'net/http'
|
|
7
|
+
require 'time'
|
|
8
|
+
|
|
9
|
+
module GSC
|
|
10
|
+
class SchemaGenerator
|
|
11
|
+
attr_reader :options
|
|
12
|
+
|
|
13
|
+
def initialize(options = {})
|
|
14
|
+
@options = options
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
# 1. Generate schema programmatically from input options
|
|
18
|
+
def generate(type, params = {})
|
|
19
|
+
normalized_type = type.to_s.downcase.gsub(/[^a-z]/, '')
|
|
20
|
+
|
|
21
|
+
schema = case normalized_type
|
|
22
|
+
when 'product'
|
|
23
|
+
build_product(params)
|
|
24
|
+
when 'faq', 'faqpage'
|
|
25
|
+
build_faq(params)
|
|
26
|
+
when 'howto', 'how-to'
|
|
27
|
+
build_howto(params)
|
|
28
|
+
when 'article', 'blog', 'blogposting', 'newsarticle'
|
|
29
|
+
build_article(params)
|
|
30
|
+
when 'software', 'softwareapplication', 'app', 'webapplication', 'mobileapplication'
|
|
31
|
+
build_software(params)
|
|
32
|
+
when 'localbusiness', 'store', 'restaurant', 'business'
|
|
33
|
+
build_local_business(params)
|
|
34
|
+
when 'organization', 'org', 'corp'
|
|
35
|
+
build_organization(params)
|
|
36
|
+
when 'breadcrumb', 'breadcrumbs', 'breadcrumblist'
|
|
37
|
+
build_breadcrumbs(params)
|
|
38
|
+
when 'course'
|
|
39
|
+
build_course(params)
|
|
40
|
+
when 'job', 'jobposting', 'careers'
|
|
41
|
+
build_job_posting(params)
|
|
42
|
+
when 'event'
|
|
43
|
+
build_event(params)
|
|
44
|
+
when 'video', 'videoobject'
|
|
45
|
+
build_video(params)
|
|
46
|
+
when 'recipe'
|
|
47
|
+
build_recipe(params)
|
|
48
|
+
else
|
|
49
|
+
build_article(params)
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
validation = validate_schema(schema)
|
|
53
|
+
|
|
54
|
+
{
|
|
55
|
+
schema: schema,
|
|
56
|
+
type: schema['@type'],
|
|
57
|
+
validation: validation,
|
|
58
|
+
snippets: format_snippets(schema)
|
|
59
|
+
}
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
# 2. Extract content from a live URL or raw HTML and auto-synthesize optimal Schema
|
|
63
|
+
def extract_from_url(url, explicit_type = nil)
|
|
64
|
+
uri = URI.parse(url)
|
|
65
|
+
req = Net::HTTP::Get.new(uri)
|
|
66
|
+
req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36'
|
|
67
|
+
|
|
68
|
+
http = Net::HTTP.new(uri.host, uri.port)
|
|
69
|
+
http.use_ssl = (uri.scheme == 'https')
|
|
70
|
+
http.open_timeout = 8
|
|
71
|
+
http.read_timeout = 10
|
|
72
|
+
|
|
73
|
+
res = http.request(req)
|
|
74
|
+
html = res.body.to_s.force_encoding('UTF-8').scrub
|
|
75
|
+
|
|
76
|
+
extract_from_html(html, url, explicit_type)
|
|
77
|
+
rescue => e
|
|
78
|
+
path_slug = URI.parse(url).path.split('/').reject(&:empty?).last || 'home' rescue 'home'
|
|
79
|
+
headline = path_slug.gsub(/[-_]/, ' ').split.map(&:capitalize).join(' ')
|
|
80
|
+
type_to_use = explicit_type || options[:type] || 'article'
|
|
81
|
+
generate(type_to_use, {
|
|
82
|
+
headline: headline,
|
|
83
|
+
name: headline,
|
|
84
|
+
url: url,
|
|
85
|
+
description: "Comprehensive guide and overview for #{headline}."
|
|
86
|
+
})
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
def extract_from_html(html, url = nil, explicit_type = nil)
|
|
90
|
+
title = extract_tag_content(html, 'title') || extract_meta_content(html, 'og:title') || 'Untitled Page'
|
|
91
|
+
desc = extract_meta_content(html, 'description') || extract_meta_content(html, 'og:description') || ''
|
|
92
|
+
image = extract_meta_content(html, 'og:image') || extract_first_image(html, url)
|
|
93
|
+
h1 = extract_first_h1(html)
|
|
94
|
+
primary_name = (h1 && !h1.empty?) ? h1 : title
|
|
95
|
+
|
|
96
|
+
# Price extraction heuristic
|
|
97
|
+
price_match = html.match(/(?:\$|USD|EUR|GBP|\u00A3|\u20AC)\s*([0-9]+(?:\.[0-9]{2})?)/i) ||
|
|
98
|
+
html.match(/"price"\s*:\s*"?([0-9]+(?:\.[0-9]{2})?)"?/i) ||
|
|
99
|
+
html.match(/itemprop=["']price["'][^>]*content=["']([0-9.]+)["']/i)
|
|
100
|
+
|
|
101
|
+
# FAQ extraction heuristic
|
|
102
|
+
q_matches = html.scan(/<(?:h[234]|summary|dt)[^>]*>([^<]*\?[\s\S]*?)<\/(?:h[234]|summary|dt)>/i).flatten.map { |q| strip_html(q) }.reject(&:empty?)
|
|
103
|
+
|
|
104
|
+
# Breadcrumb extraction heuristic
|
|
105
|
+
bc_matches = html.scan(/<(?:a|span)[^>]*class=["'][^"']*(?:breadcrumb|crumb)[^"']*["'][^>]*>(.*?)<\/(?:a|span)>/i).flatten.map { |b| strip_html(b) }.reject(&:empty?)
|
|
106
|
+
|
|
107
|
+
auto_detected = if price_match || html =~ /class=["'][^"']*(?:product|ecommerce|price|add-to-cart)/i
|
|
108
|
+
'product'
|
|
109
|
+
elsif q_matches.size >= 2
|
|
110
|
+
'faq'
|
|
111
|
+
elsif html =~ /class=["'][^"']*(?:software|app|download|pricing-table)/i
|
|
112
|
+
'software'
|
|
113
|
+
elsif bc_matches.size >= 2
|
|
114
|
+
'breadcrumb'
|
|
115
|
+
else
|
|
116
|
+
'article'
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
target_type = explicit_type || options[:type] || auto_detected
|
|
120
|
+
brand_extracted = URI.parse(url).host.to_s.sub(/^www\./, '').split('.').first.capitalize rescue 'Brand'
|
|
121
|
+
|
|
122
|
+
params = {
|
|
123
|
+
url: options[:url] || url,
|
|
124
|
+
name: options[:name] || primary_name,
|
|
125
|
+
headline: options[:headline] || options[:name] || primary_name,
|
|
126
|
+
description: options[:description] || (desc.empty? ? "#{primary_name} overview and details." : desc),
|
|
127
|
+
image: options[:image] || image,
|
|
128
|
+
brand: options[:brand] || brand_extracted,
|
|
129
|
+
price: options[:price] || (price_match ? price_match[1].to_f : 0.0),
|
|
130
|
+
currency: options[:currency] || 'USD',
|
|
131
|
+
rating: options[:rating],
|
|
132
|
+
reviews: options[:reviews],
|
|
133
|
+
author: options[:author],
|
|
134
|
+
operating_system: options[:os] || options[:operating_system] || 'Web, macOS, Windows, iOS, Android',
|
|
135
|
+
category: options[:category] || 'UtilitiesApplication'
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
if q_matches.size >= 2
|
|
139
|
+
params[:questions] = q_matches.first(5).map do |clean_q|
|
|
140
|
+
{ q: clean_q, a: "Detailed answer and explanation regarding #{clean_q.sub(/\?*$/, '')}." }
|
|
141
|
+
end
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
if bc_matches.size >= 2
|
|
145
|
+
params[:breadcrumbs] = bc_matches.map { |b| { name: b, url: url } }
|
|
146
|
+
end
|
|
147
|
+
|
|
148
|
+
existing_schemas = extract_existing_schemas(html)
|
|
149
|
+
result = generate(target_type, params)
|
|
150
|
+
|
|
151
|
+
target_class = result[:schema]['@type'].to_s.downcase
|
|
152
|
+
matching_existing = existing_schemas.find do |s|
|
|
153
|
+
s_type = s['@type'].to_s.downcase
|
|
154
|
+
s_type == target_class ||
|
|
155
|
+
(target_class == 'softwareapplication' && s_type =~ /software|app/) ||
|
|
156
|
+
(target_class == 'faqpage' && s_type =~ /faq/) ||
|
|
157
|
+
(target_class == 'article' && s_type =~ /article|blog/)
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
existing_types = existing_schemas.map { |s| s['@type'] }.compact.uniq
|
|
161
|
+
|
|
162
|
+
result[:existing_analysis] = {
|
|
163
|
+
existing_count: existing_schemas.size,
|
|
164
|
+
existing_types: existing_types,
|
|
165
|
+
duplicate_detected: !matching_existing.nil?,
|
|
166
|
+
matching_type: matching_existing ? matching_existing['@type'] : nil,
|
|
167
|
+
matching_schema: matching_existing
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
result
|
|
171
|
+
end
|
|
172
|
+
|
|
173
|
+
def extract_existing_schemas(html)
|
|
174
|
+
schemas = []
|
|
175
|
+
return schemas if html.nil? || html.empty?
|
|
176
|
+
|
|
177
|
+
html.scan(%r{<script[^>]*type=["']application/ld\+json["'][^>]*>(.*?)</script>}im) do |match|
|
|
178
|
+
json_str = match[0].to_s.strip
|
|
179
|
+
next if json_str.empty?
|
|
180
|
+
|
|
181
|
+
begin
|
|
182
|
+
parsed = JSON.parse(json_str)
|
|
183
|
+
schemas.concat(flatten_existing(parsed))
|
|
184
|
+
rescue JSON::ParserError
|
|
185
|
+
# skip corrupted
|
|
186
|
+
end
|
|
187
|
+
end
|
|
188
|
+
schemas
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
def flatten_existing(data)
|
|
192
|
+
case data
|
|
193
|
+
when Array
|
|
194
|
+
data.flat_map { |item| flatten_existing(item) }
|
|
195
|
+
when Hash
|
|
196
|
+
if data['@graph'].is_a?(Array)
|
|
197
|
+
data['@graph'].flat_map { |item| flatten_existing(item) }
|
|
198
|
+
else
|
|
199
|
+
[data]
|
|
200
|
+
end
|
|
201
|
+
else
|
|
202
|
+
[]
|
|
203
|
+
end
|
|
204
|
+
end
|
|
205
|
+
|
|
206
|
+
# 3. Validation against Google Rich Result Guidelines
|
|
207
|
+
def validate_schema(schema)
|
|
208
|
+
if schema.is_a?(Hash) && schema['@graph'].is_a?(Array)
|
|
209
|
+
aggregated_errors = []
|
|
210
|
+
aggregated_warnings = []
|
|
211
|
+
schema['@graph'].each do |item|
|
|
212
|
+
sub_val = validate_schema(item)
|
|
213
|
+
aggregated_errors.concat(sub_val[:errors])
|
|
214
|
+
aggregated_warnings.concat(sub_val[:warnings])
|
|
215
|
+
end
|
|
216
|
+
return {
|
|
217
|
+
valid: aggregated_errors.empty?,
|
|
218
|
+
errors: aggregated_errors,
|
|
219
|
+
warnings: aggregated_warnings
|
|
220
|
+
}
|
|
221
|
+
end
|
|
222
|
+
|
|
223
|
+
type = schema['@type']
|
|
224
|
+
errors = []
|
|
225
|
+
warnings = []
|
|
226
|
+
|
|
227
|
+
case type
|
|
228
|
+
when 'Product'
|
|
229
|
+
errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
|
|
230
|
+
errors << 'Missing "offers" or "aggregateRating"' unless schema['offers'] || schema['aggregateRating']
|
|
231
|
+
warnings << 'Missing "image" (strongly recommended for Google Shopping rich results)' unless schema['image']
|
|
232
|
+
warnings << 'Missing "brand"' unless schema['brand']
|
|
233
|
+
|
|
234
|
+
if schema['offers']
|
|
235
|
+
offers = schema['offers']
|
|
236
|
+
errors << 'Offers missing "price"' unless offers['price']
|
|
237
|
+
errors << 'Offers missing "priceCurrency"' unless offers['priceCurrency']
|
|
238
|
+
end
|
|
239
|
+
|
|
240
|
+
when 'FAQPage'
|
|
241
|
+
main_entity = schema['mainEntity']
|
|
242
|
+
if !main_entity.is_a?(Array) || main_entity.empty?
|
|
243
|
+
errors << 'FAQPage must contain a non-empty "mainEntity" array'
|
|
244
|
+
else
|
|
245
|
+
main_entity.each_with_index do |q, idx|
|
|
246
|
+
errors << "Question ##{idx + 1} missing 'name'" if q['name'].to_s.strip.empty?
|
|
247
|
+
errors << "Question ##{idx + 1} missing 'acceptedAnswer'" unless q['acceptedAnswer'] && q['acceptedAnswer']['text']
|
|
248
|
+
end
|
|
249
|
+
end
|
|
250
|
+
|
|
251
|
+
when 'HowTo'
|
|
252
|
+
errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
|
|
253
|
+
errors << 'Missing "step" array' unless schema['step'].is_a?(Array) && !schema['step'].empty?
|
|
254
|
+
warnings << 'Missing "totalTime" or "estimatedCost"' unless schema['totalTime'] || schema['estimatedCost']
|
|
255
|
+
|
|
256
|
+
when 'Article', 'BlogPosting', 'NewsArticle'
|
|
257
|
+
errors << 'Missing "headline"' if schema['headline'].to_s.strip.empty?
|
|
258
|
+
errors << 'Missing "author"' unless schema['author']
|
|
259
|
+
warnings << 'Missing "datePublished"' unless schema['datePublished']
|
|
260
|
+
warnings << 'Missing "image" (Google Top Stories requires image >= 1200px)' unless schema['image']
|
|
261
|
+
|
|
262
|
+
when 'SoftwareApplication', 'WebApplication', 'MobileApplication'
|
|
263
|
+
errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
|
|
264
|
+
warnings << 'Missing "operatingSystem"' unless schema['operatingSystem']
|
|
265
|
+
warnings << 'Missing "applicationCategory"' unless schema['applicationCategory']
|
|
266
|
+
|
|
267
|
+
when 'LocalBusiness', 'Store', 'Restaurant'
|
|
268
|
+
errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
|
|
269
|
+
errors << 'Missing "address"' unless schema['address']
|
|
270
|
+
warnings << 'Missing "telephone"' unless schema['telephone']
|
|
271
|
+
warnings << 'Missing "openingHoursSpecification"' unless schema['openingHoursSpecification']
|
|
272
|
+
|
|
273
|
+
when 'BreadcrumbList'
|
|
274
|
+
items = schema['itemListElement']
|
|
275
|
+
if !items.is_a?(Array) || items.empty?
|
|
276
|
+
errors << 'BreadcrumbList must contain "itemListElement" array'
|
|
277
|
+
else
|
|
278
|
+
items.each_with_index do |it, idx|
|
|
279
|
+
errors << "Breadcrumb ##{idx + 1} missing 'position'" unless it['position']
|
|
280
|
+
errors << "Breadcrumb ##{idx + 1} missing 'name'" unless it['name']
|
|
281
|
+
errors << "Breadcrumb ##{idx + 1} missing 'item'" unless it['item']
|
|
282
|
+
end
|
|
283
|
+
end
|
|
284
|
+
|
|
285
|
+
when 'Course'
|
|
286
|
+
errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
|
|
287
|
+
errors << 'Missing "description"' if schema['description'].to_s.strip.empty?
|
|
288
|
+
errors << 'Missing "provider"' unless schema['provider']
|
|
289
|
+
|
|
290
|
+
when 'JobPosting'
|
|
291
|
+
errors << 'Missing "title"' if schema['title'].to_s.strip.empty?
|
|
292
|
+
errors << 'Missing "description"' if schema['description'].to_s.strip.empty?
|
|
293
|
+
errors << 'Missing "datePosted"' unless schema['datePosted']
|
|
294
|
+
errors << 'Missing "hiringOrganization"' unless schema['hiringOrganization']
|
|
295
|
+
|
|
296
|
+
when 'Event'
|
|
297
|
+
errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
|
|
298
|
+
errors << 'Missing "startDate"' unless schema['startDate']
|
|
299
|
+
errors << 'Missing "location"' unless schema['location']
|
|
300
|
+
|
|
301
|
+
when 'VideoObject'
|
|
302
|
+
errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
|
|
303
|
+
errors << 'Missing "description"' if schema['description'].to_s.strip.empty?
|
|
304
|
+
errors << 'Missing "thumbnailUrl"' unless schema['thumbnailUrl']
|
|
305
|
+
errors << 'Missing "uploadDate"' unless schema['uploadDate']
|
|
306
|
+
|
|
307
|
+
when 'Recipe'
|
|
308
|
+
errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
|
|
309
|
+
errors << 'Missing "image"' unless schema['image']
|
|
310
|
+
errors << 'Missing "recipeIngredient"' unless schema['recipeIngredient']
|
|
311
|
+
errors << 'Missing "recipeInstructions"' unless schema['recipeInstructions']
|
|
312
|
+
|
|
313
|
+
when 'Organization'
|
|
314
|
+
errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
|
|
315
|
+
warnings << 'Missing "url"' unless schema['url']
|
|
316
|
+
warnings << 'Missing "logo"' unless schema['logo']
|
|
317
|
+
end
|
|
318
|
+
|
|
319
|
+
{
|
|
320
|
+
valid: errors.empty?,
|
|
321
|
+
errors: errors,
|
|
322
|
+
warnings: warnings,
|
|
323
|
+
score: calculate_schema_score(errors, warnings)
|
|
324
|
+
}
|
|
325
|
+
end
|
|
326
|
+
|
|
327
|
+
def format_snippets(schema)
|
|
328
|
+
json_str = JSON.pretty_generate(schema)
|
|
329
|
+
|
|
330
|
+
html_script = "<script type=\"application/ld+json\">\n#{json_str}\n</script>"
|
|
331
|
+
|
|
332
|
+
nextjs_script = "<script\n type=\"application/ld+json\"\n dangerouslySetInnerHTML={{ __html: JSON.stringify(#{json_str.lines.map { |l| ' ' + l }.join.strip}) }}\n/>"
|
|
333
|
+
|
|
334
|
+
shopify_snippet = "{% comment %} Schema.org #{schema['@type']} JSON-LD (Generated by gsc-cli) {% endcomment %}\n<script type=\"application/ld+json\">\n#{json_str}\n</script>"
|
|
335
|
+
|
|
336
|
+
{
|
|
337
|
+
html: html_script,
|
|
338
|
+
nextjs: nextjs_script,
|
|
339
|
+
shopify: shopify_snippet,
|
|
340
|
+
raw_json: json_str
|
|
341
|
+
}
|
|
342
|
+
end
|
|
343
|
+
|
|
344
|
+
def strip_html(str)
|
|
345
|
+
return '' unless str
|
|
346
|
+
s = str.to_s.dup
|
|
347
|
+
s = s.gsub(/<script\b[^<]*(?:(?!<\/script>)<[^<]*)*<\/script>/im, '')
|
|
348
|
+
s = s.gsub(/<style\b[^<]*(?:(?!<\/style>)<[^<]*)*<\/style>/im, '')
|
|
349
|
+
s = s.gsub(/<br\s*\/?>/i, ' ')
|
|
350
|
+
s = s.gsub(/<[^>]+>/, ' ')
|
|
351
|
+
s = s.gsub(/ /i, ' ')
|
|
352
|
+
.gsub(/&/i, '&')
|
|
353
|
+
.gsub(/"/i, '"')
|
|
354
|
+
.gsub(/'/i, "'")
|
|
355
|
+
.gsub(/</i, '<')
|
|
356
|
+
.gsub(/>/i, '>')
|
|
357
|
+
s.gsub(/\s+/, ' ').strip
|
|
358
|
+
end
|
|
359
|
+
|
|
360
|
+
private
|
|
361
|
+
|
|
362
|
+
def build_product(p)
|
|
363
|
+
name = p[:name] || p['name'] || 'Featured Product'
|
|
364
|
+
desc = p[:description] || p['description'] || "#{name} description."
|
|
365
|
+
price = (p[:price] || p['price']) ? (p[:price] || p['price']).to_f.round(2) : nil
|
|
366
|
+
currency = p[:currency] || p['currency'] || 'USD'
|
|
367
|
+
image = p[:image] || p['image']
|
|
368
|
+
brand_name = p[:brand] || p['brand'] || 'Brand'
|
|
369
|
+
sku = p[:sku] || p['sku']
|
|
370
|
+
rating_val = (p[:rating] || p['rating']) ? (p[:rating] || p['rating']).to_f.round(1) : nil
|
|
371
|
+
review_count = (p[:reviews] || p['reviews']) ? (p[:reviews] || p['reviews']).to_i : nil
|
|
372
|
+
|
|
373
|
+
schema = {
|
|
374
|
+
'@context' => 'https://schema.org',
|
|
375
|
+
'@type' => 'Product',
|
|
376
|
+
'name' => name,
|
|
377
|
+
'description' => desc,
|
|
378
|
+
'brand' => {
|
|
379
|
+
'@type' => 'Brand',
|
|
380
|
+
'name' => brand_name
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
|
|
384
|
+
schema['image'] = [image] if image
|
|
385
|
+
schema['sku'] = sku if sku
|
|
386
|
+
|
|
387
|
+
if price
|
|
388
|
+
schema['offers'] = {
|
|
389
|
+
'@type' => 'Offer',
|
|
390
|
+
'url' => p[:url] || p['url'] || '',
|
|
391
|
+
'priceCurrency' => currency,
|
|
392
|
+
'price' => price.to_s,
|
|
393
|
+
'priceValidUntil' => (Time.now + (365 * 24 * 3600)).strftime('%Y-%m-%d'),
|
|
394
|
+
'itemCondition' => 'https://schema.org/NewCondition',
|
|
395
|
+
'availability' => 'https://schema.org/InStock'
|
|
396
|
+
}
|
|
397
|
+
end
|
|
398
|
+
|
|
399
|
+
if rating_val && review_count
|
|
400
|
+
schema['aggregateRating'] = {
|
|
401
|
+
'@type' => 'AggregateRating',
|
|
402
|
+
'ratingValue' => rating_val.to_s,
|
|
403
|
+
'reviewCount' => review_count.to_s
|
|
404
|
+
}
|
|
405
|
+
end
|
|
406
|
+
|
|
407
|
+
schema
|
|
408
|
+
end
|
|
409
|
+
|
|
410
|
+
def build_faq(p)
|
|
411
|
+
questions = p[:questions] || p['questions'] || []
|
|
412
|
+
if questions.empty?
|
|
413
|
+
# Fallback default questions
|
|
414
|
+
q_text = p[:q] || p['q'] || p[:question] || p['question'] || 'How does this work?'
|
|
415
|
+
a_text = p[:a] || p['a'] || p[:answer] || p['answer'] || 'It operates automatically using advanced heuristics.'
|
|
416
|
+
questions = [{ q: q_text, a: a_text }]
|
|
417
|
+
end
|
|
418
|
+
|
|
419
|
+
entities = questions.map do |item|
|
|
420
|
+
q_str = item[:q] || item['q'] || item[:question] || item['question'] || 'Frequently Asked Question'
|
|
421
|
+
a_str = item[:a] || item['a'] || item[:answer] || item['answer'] || 'Answer description.'
|
|
422
|
+
{
|
|
423
|
+
'@type' => 'Question',
|
|
424
|
+
'name' => q_str,
|
|
425
|
+
'acceptedAnswer' => {
|
|
426
|
+
'@type' => 'Answer',
|
|
427
|
+
'text' => a_str
|
|
428
|
+
}
|
|
429
|
+
}
|
|
430
|
+
end
|
|
431
|
+
|
|
432
|
+
{
|
|
433
|
+
'@context' => 'https://schema.org',
|
|
434
|
+
'@type' => 'FAQPage',
|
|
435
|
+
'mainEntity' => entities
|
|
436
|
+
}
|
|
437
|
+
end
|
|
438
|
+
|
|
439
|
+
def build_howto(p)
|
|
440
|
+
name = p[:name] || p['name'] || p[:headline] || 'How to Optimize Performance'
|
|
441
|
+
desc = p[:description] || p['description'] || "Step-by-step actionable guide for #{name}."
|
|
442
|
+
steps = p[:steps] || p['steps'] || [
|
|
443
|
+
{ name: 'Initial Assessment', text: 'Audit current baseline metrics and identify bottlenecks.' },
|
|
444
|
+
{ name: 'Implementation', text: 'Deploy high-impact optimizations across key components.' },
|
|
445
|
+
{ name: 'Verification', text: 'Validate performance gains and ensure zero regressions.' }
|
|
446
|
+
]
|
|
447
|
+
|
|
448
|
+
step_entities = steps.each_with_index.map do |s, idx|
|
|
449
|
+
{
|
|
450
|
+
'@type' => 'HowToStep',
|
|
451
|
+
'position' => (idx + 1).to_s,
|
|
452
|
+
'name' => s[:name] || s['name'] || "Step #{idx + 1}",
|
|
453
|
+
'text' => s[:text] || s['text'] || s.to_s
|
|
454
|
+
}
|
|
455
|
+
end
|
|
456
|
+
|
|
457
|
+
{
|
|
458
|
+
'@context' => 'https://schema.org',
|
|
459
|
+
'@type' => 'HowTo',
|
|
460
|
+
'name' => name,
|
|
461
|
+
'description' => desc,
|
|
462
|
+
'totalTime' => p[:time] || p['time'] || 'PT15M',
|
|
463
|
+
'step' => step_entities
|
|
464
|
+
}
|
|
465
|
+
end
|
|
466
|
+
|
|
467
|
+
def build_article(p)
|
|
468
|
+
headline = p[:headline] || p['headline'] || p[:name] || 'Comprehensive Guide'
|
|
469
|
+
desc = p[:description] || p['description'] || "#{headline}: In-depth analysis and expert insights."
|
|
470
|
+
author_name = p[:author] || p['author'] || 'Editorial Team'
|
|
471
|
+
publisher_name = p[:publisher] || p['publisher'] || 'Publisher'
|
|
472
|
+
url = p[:url] || p['url'] || ''
|
|
473
|
+
image = p[:image] || p['image']
|
|
474
|
+
published = p[:date_published] || p['date_published'] || Time.now.strftime('%Y-%m-%dT%H:%M:%SZ')
|
|
475
|
+
|
|
476
|
+
schema = {
|
|
477
|
+
'@context' => 'https://schema.org',
|
|
478
|
+
'@type' => 'Article',
|
|
479
|
+
'headline' => headline,
|
|
480
|
+
'description' => desc,
|
|
481
|
+
'datePublished' => published,
|
|
482
|
+
'dateModified' => Time.now.strftime('%Y-%m-%dT%H:%M:%SZ'),
|
|
483
|
+
'author' => {
|
|
484
|
+
'@type' => 'Person',
|
|
485
|
+
'name' => author_name
|
|
486
|
+
},
|
|
487
|
+
'publisher' => {
|
|
488
|
+
'@type' => 'Organization',
|
|
489
|
+
'name' => publisher_name
|
|
490
|
+
}
|
|
491
|
+
}
|
|
492
|
+
|
|
493
|
+
schema['image'] = [image] if image
|
|
494
|
+
schema['publisher']['logo'] = { '@type' => 'ImageObject', 'url' => p[:logo] || p['logo'] } if (p[:logo] || p['logo'])
|
|
495
|
+
schema['mainEntityOfPage'] = { '@type' => 'WebPage', '@id' => url } unless url.empty?
|
|
496
|
+
|
|
497
|
+
schema
|
|
498
|
+
end
|
|
499
|
+
|
|
500
|
+
def build_software(p)
|
|
501
|
+
name = p[:name] || p['name'] || 'Software Application'
|
|
502
|
+
desc = p[:description] || p['description'] || "#{name} application."
|
|
503
|
+
os = p[:os] || p['os'] || p[:operating_system] || 'Web, macOS, Windows, Linux'
|
|
504
|
+
category = p[:category] || p['category'] || 'BusinessApplication'
|
|
505
|
+
price = (p[:price] || p['price'] || 0.0).to_f.round(2)
|
|
506
|
+
rating_val = (p[:rating] || p['rating']) ? (p[:rating] || p['rating']).to_f.round(1) : nil
|
|
507
|
+
review_count = (p[:reviews] || p['reviews']) ? (p[:reviews] || p['reviews']).to_i : nil
|
|
508
|
+
|
|
509
|
+
schema = {
|
|
510
|
+
'@context' => 'https://schema.org',
|
|
511
|
+
'@type' => 'SoftwareApplication',
|
|
512
|
+
'name' => name,
|
|
513
|
+
'operatingSystem' => os,
|
|
514
|
+
'applicationCategory' => category,
|
|
515
|
+
'description' => desc,
|
|
516
|
+
'offers' => {
|
|
517
|
+
'@type' => 'Offer',
|
|
518
|
+
'price' => price.to_s,
|
|
519
|
+
'priceCurrency' => p[:currency] || p['currency'] || 'USD'
|
|
520
|
+
}
|
|
521
|
+
}
|
|
522
|
+
|
|
523
|
+
if rating_val && review_count
|
|
524
|
+
schema['aggregateRating'] = {
|
|
525
|
+
'@type' => 'AggregateRating',
|
|
526
|
+
'ratingValue' => rating_val.to_s,
|
|
527
|
+
'reviewCount' => review_count.to_s
|
|
528
|
+
}
|
|
529
|
+
end
|
|
530
|
+
|
|
531
|
+
schema
|
|
532
|
+
end
|
|
533
|
+
|
|
534
|
+
def build_local_business(p)
|
|
535
|
+
name = p[:name] || p['name'] || 'Local Business'
|
|
536
|
+
url = p[:url] || p['url'] || ''
|
|
537
|
+
|
|
538
|
+
schema = {
|
|
539
|
+
'@context' => 'https://schema.org',
|
|
540
|
+
'@type' => 'LocalBusiness',
|
|
541
|
+
'name' => name,
|
|
542
|
+
'telephone' => p[:telephone] || '+1-555-0199',
|
|
543
|
+
'address' => {
|
|
544
|
+
'@type' => 'PostalAddress',
|
|
545
|
+
'streetAddress' => p[:street] || '123 Market Street',
|
|
546
|
+
'addressLocality' => p[:city] || 'San Francisco',
|
|
547
|
+
'addressRegion' => p[:state] || 'CA',
|
|
548
|
+
'postalCode' => p[:zip] || '94103',
|
|
549
|
+
'addressCountry' => 'US'
|
|
550
|
+
},
|
|
551
|
+
'openingHoursSpecification' => [
|
|
552
|
+
{
|
|
553
|
+
'@type' => 'OpeningHoursSpecification',
|
|
554
|
+
'dayOfWeek' => %w[Monday Tuesday Wednesday Thursday Friday],
|
|
555
|
+
'opens' => '09:00',
|
|
556
|
+
'closes' => '18:00'
|
|
557
|
+
}
|
|
558
|
+
]
|
|
559
|
+
}
|
|
560
|
+
|
|
561
|
+
schema['url'] = url unless url.empty?
|
|
562
|
+
schema['image'] = [p[:image]] if p[:image]
|
|
563
|
+
schema
|
|
564
|
+
end
|
|
565
|
+
|
|
566
|
+
def build_breadcrumbs(p)
|
|
567
|
+
items = p[:breadcrumbs] || p['breadcrumbs'] || []
|
|
568
|
+
if items.empty?
|
|
569
|
+
base_url = p[:url] || 'https://example.com'
|
|
570
|
+
items = [
|
|
571
|
+
{ name: 'Home', url: base_url },
|
|
572
|
+
{ name: p[:name] || 'Current Page', url: "#{base_url}/current" }
|
|
573
|
+
]
|
|
574
|
+
end
|
|
575
|
+
|
|
576
|
+
item_list = items.each_with_index.map do |it, idx|
|
|
577
|
+
{
|
|
578
|
+
'@type' => 'ListItem',
|
|
579
|
+
'position' => idx + 1,
|
|
580
|
+
'name' => it[:name] || it['name'] || "Item #{idx + 1}",
|
|
581
|
+
'item' => it[:url] || it['url'] || "#{p[:url]}/step-#{idx + 1}"
|
|
582
|
+
}
|
|
583
|
+
end
|
|
584
|
+
|
|
585
|
+
{
|
|
586
|
+
'@context' => 'https://schema.org',
|
|
587
|
+
'@type' => 'BreadcrumbList',
|
|
588
|
+
'itemListElement' => item_list
|
|
589
|
+
}
|
|
590
|
+
end
|
|
591
|
+
|
|
592
|
+
def build_course(p)
|
|
593
|
+
name = p[:name] || p['name'] || p[:headline] || 'Comprehensive Online Course'
|
|
594
|
+
desc = p[:description] || p['description'] || "Complete course on #{name}."
|
|
595
|
+
provider = p[:brand] || p[:provider] || 'Academy'
|
|
596
|
+
|
|
597
|
+
schema = {
|
|
598
|
+
'@context' => 'https://schema.org',
|
|
599
|
+
'@type' => 'Course',
|
|
600
|
+
'name' => name,
|
|
601
|
+
'description' => desc,
|
|
602
|
+
'provider' => {
|
|
603
|
+
'@type' => 'Organization',
|
|
604
|
+
'name' => provider,
|
|
605
|
+
'sameAs' => p[:url] || ''
|
|
606
|
+
}
|
|
607
|
+
}
|
|
608
|
+
|
|
609
|
+
if p[:price]
|
|
610
|
+
schema['offers'] = {
|
|
611
|
+
'@type' => 'Offer',
|
|
612
|
+
'category' => 'Paid',
|
|
613
|
+
'price' => (p[:price] || 0.0).to_f.round(2).to_s,
|
|
614
|
+
'priceCurrency' => p[:currency] || 'USD'
|
|
615
|
+
}
|
|
616
|
+
end
|
|
617
|
+
|
|
618
|
+
schema
|
|
619
|
+
end
|
|
620
|
+
|
|
621
|
+
def build_job_posting(p)
|
|
622
|
+
title = p[:title] || p[:name] || 'Technical Specialist'
|
|
623
|
+
desc = p[:description] || "Exciting opportunity for #{title}."
|
|
624
|
+
hiring_org = p[:brand] || p[:hiring_organization] || 'Company'
|
|
625
|
+
date_posted = p[:date_posted] || Time.now.strftime('%Y-%m-%d')
|
|
626
|
+
valid_through = (Time.now + (90 * 24 * 3600)).strftime('%Y-%m-%d')
|
|
627
|
+
|
|
628
|
+
schema = {
|
|
629
|
+
'@context' => 'https://schema.org',
|
|
630
|
+
'@type' => 'JobPosting',
|
|
631
|
+
'title' => title,
|
|
632
|
+
'description' => desc,
|
|
633
|
+
'datePosted' => date_posted,
|
|
634
|
+
'validThrough' => valid_through,
|
|
635
|
+
'employmentType' => p[:employment_type] || 'FULL_TIME',
|
|
636
|
+
'hiringOrganization' => {
|
|
637
|
+
'@type' => 'Organization',
|
|
638
|
+
'name' => hiring_org,
|
|
639
|
+
'sameAs' => p[:url] || ''
|
|
640
|
+
},
|
|
641
|
+
'jobLocation' => {
|
|
642
|
+
'@type' => 'Place',
|
|
643
|
+
'address' => {
|
|
644
|
+
'@type' => 'PostalAddress',
|
|
645
|
+
'addressCountry' => 'US'
|
|
646
|
+
}
|
|
647
|
+
}
|
|
648
|
+
}
|
|
649
|
+
|
|
650
|
+
if p[:price] || p[:salary]
|
|
651
|
+
schema['baseSalary'] = {
|
|
652
|
+
'@type' => 'MonetaryAmount',
|
|
653
|
+
'currency' => p[:currency] || 'USD',
|
|
654
|
+
'value' => {
|
|
655
|
+
'@type' => 'QuantitativeValue',
|
|
656
|
+
'value' => (p[:price] || p[:salary]).to_f,
|
|
657
|
+
'unitText' => 'YEAR'
|
|
658
|
+
}
|
|
659
|
+
}
|
|
660
|
+
end
|
|
661
|
+
|
|
662
|
+
schema
|
|
663
|
+
end
|
|
664
|
+
|
|
665
|
+
def build_event(p)
|
|
666
|
+
name = p[:name] || p['name'] || 'Featured Event & Workshop'
|
|
667
|
+
desc = p[:description] || p['description'] || "#{name} overview."
|
|
668
|
+
start_date = p[:start_date] || (Time.now + (14 * 24 * 3600)).strftime('%Y-%m-%dT09:00:00Z')
|
|
669
|
+
end_date = p[:end_date] || (Time.now + (14 * 24 * 3600) + 28800).strftime('%Y-%m-%dT17:00:00Z')
|
|
670
|
+
|
|
671
|
+
schema = {
|
|
672
|
+
'@context' => 'https://schema.org',
|
|
673
|
+
'@type' => 'Event',
|
|
674
|
+
'name' => name,
|
|
675
|
+
'description' => desc,
|
|
676
|
+
'startDate' => start_date,
|
|
677
|
+
'endDate' => end_date,
|
|
678
|
+
'eventAttendanceMode' => 'https://schema.org/OnlineEventAttendanceMode',
|
|
679
|
+
'eventStatus' => 'https://schema.org/EventScheduled',
|
|
680
|
+
'location' => {
|
|
681
|
+
'@type' => 'VirtualLocation',
|
|
682
|
+
'url' => p[:url] || 'https://example.com/webinar'
|
|
683
|
+
},
|
|
684
|
+
'organizer' => {
|
|
685
|
+
'@type' => 'Organization',
|
|
686
|
+
'name' => p[:brand] || 'Event Organizer',
|
|
687
|
+
'url' => p[:url] || ''
|
|
688
|
+
}
|
|
689
|
+
}
|
|
690
|
+
|
|
691
|
+
if p[:price]
|
|
692
|
+
schema['offers'] = {
|
|
693
|
+
'@type' => 'Offer',
|
|
694
|
+
'url' => p[:url] || '',
|
|
695
|
+
'price' => (p[:price] || 0.0).to_f.round(2).to_s,
|
|
696
|
+
'priceCurrency' => p[:currency] || 'USD',
|
|
697
|
+
'availability' => 'https://schema.org/InStock'
|
|
698
|
+
}
|
|
699
|
+
end
|
|
700
|
+
|
|
701
|
+
schema
|
|
702
|
+
end
|
|
703
|
+
|
|
704
|
+
def build_video(p)
|
|
705
|
+
name = p[:name] || p['name'] || p[:headline] || 'Product Demo & Walkthrough'
|
|
706
|
+
desc = p[:description] || p['description'] || "#{name} video overview."
|
|
707
|
+
thumb = p[:image] || p[:thumbnail] || 'https://example.com/thumbnail.jpg'
|
|
708
|
+
|
|
709
|
+
{
|
|
710
|
+
'@context' => 'https://schema.org',
|
|
711
|
+
'@type' => 'VideoObject',
|
|
712
|
+
'name' => name,
|
|
713
|
+
'description' => desc,
|
|
714
|
+
'thumbnailUrl' => [thumb],
|
|
715
|
+
'uploadDate' => p[:upload_date] || Time.now.strftime('%Y-%m-%d'),
|
|
716
|
+
'contentUrl' => p[:url] || 'https://example.com/video.mp4'
|
|
717
|
+
}
|
|
718
|
+
end
|
|
719
|
+
|
|
720
|
+
def build_recipe(p)
|
|
721
|
+
name = p[:name] || p['name'] || 'Classic Recipe'
|
|
722
|
+
desc = p[:description] || p['description'] || "#{name} step-by-step recipe."
|
|
723
|
+
image = p[:image] || 'https://example.com/recipe.jpg'
|
|
724
|
+
|
|
725
|
+
{
|
|
726
|
+
'@context' => 'https://schema.org',
|
|
727
|
+
'@type' => 'Recipe',
|
|
728
|
+
'name' => name,
|
|
729
|
+
'description' => desc,
|
|
730
|
+
'image' => [image],
|
|
731
|
+
'recipeIngredient' => p[:ingredients] || ['1 cup ingredient A', '2 tbsp ingredient B'],
|
|
732
|
+
'recipeInstructions' => [
|
|
733
|
+
{ '@type' => 'HowToStep', 'text' => 'Mix ingredients in a large bowl.' },
|
|
734
|
+
{ '@type' => 'HowToStep', 'text' => 'Cook for 15 minutes until golden brown.' }
|
|
735
|
+
]
|
|
736
|
+
}
|
|
737
|
+
end
|
|
738
|
+
|
|
739
|
+
def build_organization(p)
|
|
740
|
+
name = p[:name] || p['name'] || 'Organization Name'
|
|
741
|
+
url = p[:url] || p['url'] || ''
|
|
742
|
+
logo = p[:logo] || p['logo'] || (!url.empty? ? "#{url}/logo.png" : nil)
|
|
743
|
+
|
|
744
|
+
schema = {
|
|
745
|
+
'@context' => 'https://schema.org',
|
|
746
|
+
'@type' => 'Organization',
|
|
747
|
+
'name' => name,
|
|
748
|
+
'sameAs' => p[:same_as] || p['same_as'] || []
|
|
749
|
+
}
|
|
750
|
+
|
|
751
|
+
schema['url'] = url unless url.empty?
|
|
752
|
+
schema['logo'] = logo if logo
|
|
753
|
+
|
|
754
|
+
schema
|
|
755
|
+
end
|
|
756
|
+
|
|
757
|
+
def calculate_schema_score(errors, warnings)
|
|
758
|
+
score = 100
|
|
759
|
+
score -= (errors.size * 35)
|
|
760
|
+
score -= (warnings.size * 10)
|
|
761
|
+
[0, score].max
|
|
762
|
+
end
|
|
763
|
+
|
|
764
|
+
def extract_tag_content(html, tag)
|
|
765
|
+
match = html.match(/<#{tag}[^>]*>(.*?)<\/#{tag}>/im)
|
|
766
|
+
match ? strip_html(match[1]) : nil
|
|
767
|
+
end
|
|
768
|
+
|
|
769
|
+
def extract_meta_content(html, name_or_prop)
|
|
770
|
+
pattern = /<meta[^>]+(?:name|property)=["']#{Regexp.escape(name_or_prop)}["'][^>]+content=["']([^"']*)["']/i
|
|
771
|
+
alt_pattern = /<meta[^>]+content=["']([^"']*)["'][^>]+(?:name|property)=["']#{Regexp.escape(name_or_prop)}["']/i
|
|
772
|
+
m = html.match(pattern) || html.match(alt_pattern)
|
|
773
|
+
m ? strip_html(m[1]) : nil
|
|
774
|
+
end
|
|
775
|
+
|
|
776
|
+
def extract_first_h1(html)
|
|
777
|
+
extract_tag_content(html, 'h1')
|
|
778
|
+
end
|
|
779
|
+
|
|
780
|
+
def extract_first_image(html, base_url)
|
|
781
|
+
m = html.match(/<img[^>]+src=["']([^"']+)["']/i)
|
|
782
|
+
return nil unless m
|
|
783
|
+
src = m[1].strip
|
|
784
|
+
return src if src =~ %r{^https?://}
|
|
785
|
+
URI.join(base_url, src).to_s rescue src
|
|
786
|
+
end
|
|
787
|
+
end
|
|
788
|
+
end
|