gsc-cli 2.1.0 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +410 -414
  3. data/bin/gsc +28067 -5661
  4. data/dist/gsc +29121 -5046
  5. data/lib/gsc/aio_hunter.rb +343 -0
  6. data/lib/gsc/answer_synthesizer.rb +157 -0
  7. data/lib/gsc/api.rb +53 -1
  8. data/lib/gsc/auth.rb +26 -0
  9. data/lib/gsc/brand_segmenter.rb +140 -0
  10. data/lib/gsc/cache_manager.rb +806 -0
  11. data/lib/gsc/cannibalization_analyzer.rb +141 -0
  12. data/lib/gsc/canonical_chains.rb +367 -0
  13. data/lib/gsc/citation_simulator.rb +339 -0
  14. data/lib/gsc/cli/aio_hunter.rb +154 -0
  15. data/lib/gsc/cli/analytics.rb +788 -0
  16. data/lib/gsc/cli/audit.rb +1976 -0
  17. data/lib/gsc/cli/base.rb +384 -0
  18. data/lib/gsc/cli/cache.rb +266 -0
  19. data/lib/gsc/cli/canonical.rb +223 -0
  20. data/lib/gsc/cli/citation_simulator.rb +152 -0
  21. data/lib/gsc/cli/dashboard.rb +354 -0
  22. data/lib/gsc/cli/doctor.rb +129 -0
  23. data/lib/gsc/cli/eeat.rb +125 -0
  24. data/lib/gsc/cli/ga4.rb +852 -0
  25. data/lib/gsc/cli/growth.rb +650 -0
  26. data/lib/gsc/cli/hreflang.rb +164 -0
  27. data/lib/gsc/cli/image_seo.rb +162 -0
  28. data/lib/gsc/cli/indexing.rb +458 -0
  29. data/lib/gsc/cli/intent_shift.rb +125 -0
  30. data/lib/gsc/cli/keyword_value.rb +134 -0
  31. data/lib/gsc/cli/keywords.rb +795 -0
  32. data/lib/gsc/cli/landing_roi.rb +308 -0
  33. data/lib/gsc/cli/low_ctr.rb +213 -0
  34. data/lib/gsc/cli/mobile_parity.rb +150 -0
  35. data/lib/gsc/cli/report.rb +100 -0
  36. data/lib/gsc/cli/rich_results.rb +172 -0
  37. data/lib/gsc/cli/schema_generate.rb +149 -0
  38. data/lib/gsc/cli/seasonal.rb +232 -0
  39. data/lib/gsc/cli/security.rb +153 -0
  40. data/lib/gsc/cli/setup.rb +1291 -0
  41. data/lib/gsc/cli/sitemap_tree.rb +143 -0
  42. data/lib/gsc/cli/skill_pack.rb +62 -0
  43. data/lib/gsc/cli/soft_404.rb +199 -0
  44. data/lib/gsc/cli/sparkline.rb +227 -0
  45. data/lib/gsc/cli/watchdog.rb +150 -0
  46. data/lib/gsc/cli/zombie_purger.rb +208 -0
  47. data/lib/gsc/cli.rb +707 -5265
  48. data/lib/gsc/cli_advanced.rb +987 -44
  49. data/lib/gsc/client.rb +17 -2
  50. data/lib/gsc/color.rb +16 -1
  51. data/lib/gsc/command_registry.rb +47 -9
  52. data/lib/gsc/config.rb +2 -2
  53. data/lib/gsc/ctr_curve.rb +115 -0
  54. data/lib/gsc/decay_predictor.rb +322 -0
  55. data/lib/gsc/doctor.rb +434 -0
  56. data/lib/gsc/eeat_auditor.rb +428 -0
  57. data/lib/gsc/entity_auditor.rb +229 -0
  58. data/lib/gsc/firewall_scanner.rb +733 -0
  59. data/lib/gsc/geo_auditor.rb +368 -0
  60. data/lib/gsc/google_trends.rb +8 -1
  61. data/lib/gsc/heading_validator.rb +283 -0
  62. data/lib/gsc/hreflang_validator.rb +412 -0
  63. data/lib/gsc/image_seo.rb +286 -0
  64. data/lib/gsc/indexing_queue.rb +179 -0
  65. data/lib/gsc/indexnow.rb +93 -0
  66. data/lib/gsc/intent_shift.rb +188 -0
  67. data/lib/gsc/internal_links.rb +153 -36
  68. data/lib/gsc/keyword_value.rb +191 -0
  69. data/lib/gsc/landing_roi.rb +195 -0
  70. data/lib/gsc/llms_generator.rb +343 -22
  71. data/lib/gsc/low_ctr_rewriter.rb +408 -0
  72. data/lib/gsc/mobile_parity.rb +222 -0
  73. data/lib/gsc/network_tracer.rb +8 -1
  74. data/lib/gsc/page_analyzer.rb +47 -7
  75. data/lib/gsc/prompts.rb +38 -29
  76. data/lib/gsc/questions_harvester.rb +178 -0
  77. data/lib/gsc/report_generator.rb +461 -0
  78. data/lib/gsc/rich_results.rb +388 -0
  79. data/lib/gsc/robots_checker.rb +46 -15
  80. data/lib/gsc/schema_generator.rb +788 -0
  81. data/lib/gsc/schema_validator.rb +36 -38
  82. data/lib/gsc/seasonal_predictor.rb +381 -0
  83. data/lib/gsc/security_scanner.rb +496 -0
  84. data/lib/gsc/serp_feature_detector.rb +359 -0
  85. data/lib/gsc/serp_preview.rb +108 -22
  86. data/lib/gsc/site_crawler.rb +113 -21
  87. data/lib/gsc/sitemap_loader.rb +15 -4
  88. data/lib/gsc/sitemap_tree.rb +301 -0
  89. data/lib/gsc/skill_pack.rb +195 -0
  90. data/lib/gsc/soft_404_analyzer.rb +385 -0
  91. data/lib/gsc/sparkline.rb +171 -0
  92. data/lib/gsc/speed_correlator.rb +416 -0
  93. data/lib/gsc/striking_playbook.rb +190 -0
  94. data/lib/gsc/title_optimizer.rb +420 -0
  95. data/lib/gsc/vault.rb +260 -0
  96. data/lib/gsc/version.rb +1 -1
  97. data/lib/gsc/watchdog.rb +235 -0
  98. data/lib/gsc/zombie_purger.rb +366 -0
  99. data/lib/gsc.rb +118 -0
  100. metadata +75 -1
@@ -0,0 +1,788 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'json'
5
+ require 'uri'
6
+ require 'net/http'
7
+ require 'time'
8
+
9
+ module GSC
10
+ class SchemaGenerator
11
+ attr_reader :options
12
+
13
+ def initialize(options = {})
14
+ @options = options
15
+ end
16
+
17
+ # 1. Generate schema programmatically from input options
18
+ def generate(type, params = {})
19
+ normalized_type = type.to_s.downcase.gsub(/[^a-z]/, '')
20
+
21
+ schema = case normalized_type
22
+ when 'product'
23
+ build_product(params)
24
+ when 'faq', 'faqpage'
25
+ build_faq(params)
26
+ when 'howto', 'how-to'
27
+ build_howto(params)
28
+ when 'article', 'blog', 'blogposting', 'newsarticle'
29
+ build_article(params)
30
+ when 'software', 'softwareapplication', 'app', 'webapplication', 'mobileapplication'
31
+ build_software(params)
32
+ when 'localbusiness', 'store', 'restaurant', 'business'
33
+ build_local_business(params)
34
+ when 'organization', 'org', 'corp'
35
+ build_organization(params)
36
+ when 'breadcrumb', 'breadcrumbs', 'breadcrumblist'
37
+ build_breadcrumbs(params)
38
+ when 'course'
39
+ build_course(params)
40
+ when 'job', 'jobposting', 'careers'
41
+ build_job_posting(params)
42
+ when 'event'
43
+ build_event(params)
44
+ when 'video', 'videoobject'
45
+ build_video(params)
46
+ when 'recipe'
47
+ build_recipe(params)
48
+ else
49
+ build_article(params)
50
+ end
51
+
52
+ validation = validate_schema(schema)
53
+
54
+ {
55
+ schema: schema,
56
+ type: schema['@type'],
57
+ validation: validation,
58
+ snippets: format_snippets(schema)
59
+ }
60
+ end
61
+
62
+ # 2. Extract content from a live URL or raw HTML and auto-synthesize optimal Schema
63
+ def extract_from_url(url, explicit_type = nil)
64
+ uri = URI.parse(url)
65
+ req = Net::HTTP::Get.new(uri)
66
+ req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36'
67
+
68
+ http = Net::HTTP.new(uri.host, uri.port)
69
+ http.use_ssl = (uri.scheme == 'https')
70
+ http.open_timeout = 8
71
+ http.read_timeout = 10
72
+
73
+ res = http.request(req)
74
+ html = res.body.to_s.force_encoding('UTF-8').scrub
75
+
76
+ extract_from_html(html, url, explicit_type)
77
+ rescue => e
78
+ path_slug = URI.parse(url).path.split('/').reject(&:empty?).last || 'home' rescue 'home'
79
+ headline = path_slug.gsub(/[-_]/, ' ').split.map(&:capitalize).join(' ')
80
+ type_to_use = explicit_type || options[:type] || 'article'
81
+ generate(type_to_use, {
82
+ headline: headline,
83
+ name: headline,
84
+ url: url,
85
+ description: "Comprehensive guide and overview for #{headline}."
86
+ })
87
+ end
88
+
89
+ def extract_from_html(html, url = nil, explicit_type = nil)
90
+ title = extract_tag_content(html, 'title') || extract_meta_content(html, 'og:title') || 'Untitled Page'
91
+ desc = extract_meta_content(html, 'description') || extract_meta_content(html, 'og:description') || ''
92
+ image = extract_meta_content(html, 'og:image') || extract_first_image(html, url)
93
+ h1 = extract_first_h1(html)
94
+ primary_name = (h1 && !h1.empty?) ? h1 : title
95
+
96
+ # Price extraction heuristic
97
+ price_match = html.match(/(?:\$|USD|EUR|GBP|\u00A3|\u20AC)\s*([0-9]+(?:\.[0-9]{2})?)/i) ||
98
+ html.match(/"price"\s*:\s*"?([0-9]+(?:\.[0-9]{2})?)"?/i) ||
99
+ html.match(/itemprop=["']price["'][^>]*content=["']([0-9.]+)["']/i)
100
+
101
+ # FAQ extraction heuristic
102
+ q_matches = html.scan(/<(?:h[234]|summary|dt)[^>]*>([^<]*\?[\s\S]*?)<\/(?:h[234]|summary|dt)>/i).flatten.map { |q| strip_html(q) }.reject(&:empty?)
103
+
104
+ # Breadcrumb extraction heuristic
105
+ bc_matches = html.scan(/<(?:a|span)[^>]*class=["'][^"']*(?:breadcrumb|crumb)[^"']*["'][^>]*>(.*?)<\/(?:a|span)>/i).flatten.map { |b| strip_html(b) }.reject(&:empty?)
106
+
107
+ auto_detected = if price_match || html =~ /class=["'][^"']*(?:product|ecommerce|price|add-to-cart)/i
108
+ 'product'
109
+ elsif q_matches.size >= 2
110
+ 'faq'
111
+ elsif html =~ /class=["'][^"']*(?:software|app|download|pricing-table)/i
112
+ 'software'
113
+ elsif bc_matches.size >= 2
114
+ 'breadcrumb'
115
+ else
116
+ 'article'
117
+ end
118
+
119
+ target_type = explicit_type || options[:type] || auto_detected
120
+ brand_extracted = URI.parse(url).host.to_s.sub(/^www\./, '').split('.').first.capitalize rescue 'Brand'
121
+
122
+ params = {
123
+ url: options[:url] || url,
124
+ name: options[:name] || primary_name,
125
+ headline: options[:headline] || options[:name] || primary_name,
126
+ description: options[:description] || (desc.empty? ? "#{primary_name} overview and details." : desc),
127
+ image: options[:image] || image,
128
+ brand: options[:brand] || brand_extracted,
129
+ price: options[:price] || (price_match ? price_match[1].to_f : 0.0),
130
+ currency: options[:currency] || 'USD',
131
+ rating: options[:rating],
132
+ reviews: options[:reviews],
133
+ author: options[:author],
134
+ operating_system: options[:os] || options[:operating_system] || 'Web, macOS, Windows, iOS, Android',
135
+ category: options[:category] || 'UtilitiesApplication'
136
+ }
137
+
138
+ if q_matches.size >= 2
139
+ params[:questions] = q_matches.first(5).map do |clean_q|
140
+ { q: clean_q, a: "Detailed answer and explanation regarding #{clean_q.sub(/\?*$/, '')}." }
141
+ end
142
+ end
143
+
144
+ if bc_matches.size >= 2
145
+ params[:breadcrumbs] = bc_matches.map { |b| { name: b, url: url } }
146
+ end
147
+
148
+ existing_schemas = extract_existing_schemas(html)
149
+ result = generate(target_type, params)
150
+
151
+ target_class = result[:schema]['@type'].to_s.downcase
152
+ matching_existing = existing_schemas.find do |s|
153
+ s_type = s['@type'].to_s.downcase
154
+ s_type == target_class ||
155
+ (target_class == 'softwareapplication' && s_type =~ /software|app/) ||
156
+ (target_class == 'faqpage' && s_type =~ /faq/) ||
157
+ (target_class == 'article' && s_type =~ /article|blog/)
158
+ end
159
+
160
+ existing_types = existing_schemas.map { |s| s['@type'] }.compact.uniq
161
+
162
+ result[:existing_analysis] = {
163
+ existing_count: existing_schemas.size,
164
+ existing_types: existing_types,
165
+ duplicate_detected: !matching_existing.nil?,
166
+ matching_type: matching_existing ? matching_existing['@type'] : nil,
167
+ matching_schema: matching_existing
168
+ }
169
+
170
+ result
171
+ end
172
+
173
+ def extract_existing_schemas(html)
174
+ schemas = []
175
+ return schemas if html.nil? || html.empty?
176
+
177
+ html.scan(%r{<script[^>]*type=["']application/ld\+json["'][^>]*>(.*?)</script>}im) do |match|
178
+ json_str = match[0].to_s.strip
179
+ next if json_str.empty?
180
+
181
+ begin
182
+ parsed = JSON.parse(json_str)
183
+ schemas.concat(flatten_existing(parsed))
184
+ rescue JSON::ParserError
185
+ # skip corrupted
186
+ end
187
+ end
188
+ schemas
189
+ end
190
+
191
+ def flatten_existing(data)
192
+ case data
193
+ when Array
194
+ data.flat_map { |item| flatten_existing(item) }
195
+ when Hash
196
+ if data['@graph'].is_a?(Array)
197
+ data['@graph'].flat_map { |item| flatten_existing(item) }
198
+ else
199
+ [data]
200
+ end
201
+ else
202
+ []
203
+ end
204
+ end
205
+
206
+ # 3. Validation against Google Rich Result Guidelines
207
+ def validate_schema(schema)
208
+ if schema.is_a?(Hash) && schema['@graph'].is_a?(Array)
209
+ aggregated_errors = []
210
+ aggregated_warnings = []
211
+ schema['@graph'].each do |item|
212
+ sub_val = validate_schema(item)
213
+ aggregated_errors.concat(sub_val[:errors])
214
+ aggregated_warnings.concat(sub_val[:warnings])
215
+ end
216
+ return {
217
+ valid: aggregated_errors.empty?,
218
+ errors: aggregated_errors,
219
+ warnings: aggregated_warnings
220
+ }
221
+ end
222
+
223
+ type = schema['@type']
224
+ errors = []
225
+ warnings = []
226
+
227
+ case type
228
+ when 'Product'
229
+ errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
230
+ errors << 'Missing "offers" or "aggregateRating"' unless schema['offers'] || schema['aggregateRating']
231
+ warnings << 'Missing "image" (strongly recommended for Google Shopping rich results)' unless schema['image']
232
+ warnings << 'Missing "brand"' unless schema['brand']
233
+
234
+ if schema['offers']
235
+ offers = schema['offers']
236
+ errors << 'Offers missing "price"' unless offers['price']
237
+ errors << 'Offers missing "priceCurrency"' unless offers['priceCurrency']
238
+ end
239
+
240
+ when 'FAQPage'
241
+ main_entity = schema['mainEntity']
242
+ if !main_entity.is_a?(Array) || main_entity.empty?
243
+ errors << 'FAQPage must contain a non-empty "mainEntity" array'
244
+ else
245
+ main_entity.each_with_index do |q, idx|
246
+ errors << "Question ##{idx + 1} missing 'name'" if q['name'].to_s.strip.empty?
247
+ errors << "Question ##{idx + 1} missing 'acceptedAnswer'" unless q['acceptedAnswer'] && q['acceptedAnswer']['text']
248
+ end
249
+ end
250
+
251
+ when 'HowTo'
252
+ errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
253
+ errors << 'Missing "step" array' unless schema['step'].is_a?(Array) && !schema['step'].empty?
254
+ warnings << 'Missing "totalTime" or "estimatedCost"' unless schema['totalTime'] || schema['estimatedCost']
255
+
256
+ when 'Article', 'BlogPosting', 'NewsArticle'
257
+ errors << 'Missing "headline"' if schema['headline'].to_s.strip.empty?
258
+ errors << 'Missing "author"' unless schema['author']
259
+ warnings << 'Missing "datePublished"' unless schema['datePublished']
260
+ warnings << 'Missing "image" (Google Top Stories requires image >= 1200px)' unless schema['image']
261
+
262
+ when 'SoftwareApplication', 'WebApplication', 'MobileApplication'
263
+ errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
264
+ warnings << 'Missing "operatingSystem"' unless schema['operatingSystem']
265
+ warnings << 'Missing "applicationCategory"' unless schema['applicationCategory']
266
+
267
+ when 'LocalBusiness', 'Store', 'Restaurant'
268
+ errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
269
+ errors << 'Missing "address"' unless schema['address']
270
+ warnings << 'Missing "telephone"' unless schema['telephone']
271
+ warnings << 'Missing "openingHoursSpecification"' unless schema['openingHoursSpecification']
272
+
273
+ when 'BreadcrumbList'
274
+ items = schema['itemListElement']
275
+ if !items.is_a?(Array) || items.empty?
276
+ errors << 'BreadcrumbList must contain "itemListElement" array'
277
+ else
278
+ items.each_with_index do |it, idx|
279
+ errors << "Breadcrumb ##{idx + 1} missing 'position'" unless it['position']
280
+ errors << "Breadcrumb ##{idx + 1} missing 'name'" unless it['name']
281
+ errors << "Breadcrumb ##{idx + 1} missing 'item'" unless it['item']
282
+ end
283
+ end
284
+
285
+ when 'Course'
286
+ errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
287
+ errors << 'Missing "description"' if schema['description'].to_s.strip.empty?
288
+ errors << 'Missing "provider"' unless schema['provider']
289
+
290
+ when 'JobPosting'
291
+ errors << 'Missing "title"' if schema['title'].to_s.strip.empty?
292
+ errors << 'Missing "description"' if schema['description'].to_s.strip.empty?
293
+ errors << 'Missing "datePosted"' unless schema['datePosted']
294
+ errors << 'Missing "hiringOrganization"' unless schema['hiringOrganization']
295
+
296
+ when 'Event'
297
+ errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
298
+ errors << 'Missing "startDate"' unless schema['startDate']
299
+ errors << 'Missing "location"' unless schema['location']
300
+
301
+ when 'VideoObject'
302
+ errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
303
+ errors << 'Missing "description"' if schema['description'].to_s.strip.empty?
304
+ errors << 'Missing "thumbnailUrl"' unless schema['thumbnailUrl']
305
+ errors << 'Missing "uploadDate"' unless schema['uploadDate']
306
+
307
+ when 'Recipe'
308
+ errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
309
+ errors << 'Missing "image"' unless schema['image']
310
+ errors << 'Missing "recipeIngredient"' unless schema['recipeIngredient']
311
+ errors << 'Missing "recipeInstructions"' unless schema['recipeInstructions']
312
+
313
+ when 'Organization'
314
+ errors << 'Missing "name"' if schema['name'].to_s.strip.empty?
315
+ warnings << 'Missing "url"' unless schema['url']
316
+ warnings << 'Missing "logo"' unless schema['logo']
317
+ end
318
+
319
+ {
320
+ valid: errors.empty?,
321
+ errors: errors,
322
+ warnings: warnings,
323
+ score: calculate_schema_score(errors, warnings)
324
+ }
325
+ end
326
+
327
+ def format_snippets(schema)
328
+ json_str = JSON.pretty_generate(schema)
329
+
330
+ html_script = "<script type=\"application/ld+json\">\n#{json_str}\n</script>"
331
+
332
+ nextjs_script = "<script\n type=\"application/ld+json\"\n dangerouslySetInnerHTML={{ __html: JSON.stringify(#{json_str.lines.map { |l| ' ' + l }.join.strip}) }}\n/>"
333
+
334
+ shopify_snippet = "{% comment %} Schema.org #{schema['@type']} JSON-LD (Generated by gsc-cli) {% endcomment %}\n<script type=\"application/ld+json\">\n#{json_str}\n</script>"
335
+
336
+ {
337
+ html: html_script,
338
+ nextjs: nextjs_script,
339
+ shopify: shopify_snippet,
340
+ raw_json: json_str
341
+ }
342
+ end
343
+
344
+ def strip_html(str)
345
+ return '' unless str
346
+ s = str.to_s.dup
347
+ s = s.gsub(/<script\b[^<]*(?:(?!<\/script>)<[^<]*)*<\/script>/im, '')
348
+ s = s.gsub(/<style\b[^<]*(?:(?!<\/style>)<[^<]*)*<\/style>/im, '')
349
+ s = s.gsub(/<br\s*\/?>/i, ' ')
350
+ s = s.gsub(/<[^>]+>/, ' ')
351
+ s = s.gsub(/&nbsp;/i, ' ')
352
+ .gsub(/&amp;/i, '&')
353
+ .gsub(/&quot;/i, '"')
354
+ .gsub(/&#39;/i, "'")
355
+ .gsub(/&lt;/i, '<')
356
+ .gsub(/&gt;/i, '>')
357
+ s.gsub(/\s+/, ' ').strip
358
+ end
359
+
360
+ private
361
+
362
+ def build_product(p)
363
+ name = p[:name] || p['name'] || 'Featured Product'
364
+ desc = p[:description] || p['description'] || "#{name} description."
365
+ price = (p[:price] || p['price']) ? (p[:price] || p['price']).to_f.round(2) : nil
366
+ currency = p[:currency] || p['currency'] || 'USD'
367
+ image = p[:image] || p['image']
368
+ brand_name = p[:brand] || p['brand'] || 'Brand'
369
+ sku = p[:sku] || p['sku']
370
+ rating_val = (p[:rating] || p['rating']) ? (p[:rating] || p['rating']).to_f.round(1) : nil
371
+ review_count = (p[:reviews] || p['reviews']) ? (p[:reviews] || p['reviews']).to_i : nil
372
+
373
+ schema = {
374
+ '@context' => 'https://schema.org',
375
+ '@type' => 'Product',
376
+ 'name' => name,
377
+ 'description' => desc,
378
+ 'brand' => {
379
+ '@type' => 'Brand',
380
+ 'name' => brand_name
381
+ }
382
+ }
383
+
384
+ schema['image'] = [image] if image
385
+ schema['sku'] = sku if sku
386
+
387
+ if price
388
+ schema['offers'] = {
389
+ '@type' => 'Offer',
390
+ 'url' => p[:url] || p['url'] || '',
391
+ 'priceCurrency' => currency,
392
+ 'price' => price.to_s,
393
+ 'priceValidUntil' => (Time.now + (365 * 24 * 3600)).strftime('%Y-%m-%d'),
394
+ 'itemCondition' => 'https://schema.org/NewCondition',
395
+ 'availability' => 'https://schema.org/InStock'
396
+ }
397
+ end
398
+
399
+ if rating_val && review_count
400
+ schema['aggregateRating'] = {
401
+ '@type' => 'AggregateRating',
402
+ 'ratingValue' => rating_val.to_s,
403
+ 'reviewCount' => review_count.to_s
404
+ }
405
+ end
406
+
407
+ schema
408
+ end
409
+
410
+ def build_faq(p)
411
+ questions = p[:questions] || p['questions'] || []
412
+ if questions.empty?
413
+ # Fallback default questions
414
+ q_text = p[:q] || p['q'] || p[:question] || p['question'] || 'How does this work?'
415
+ a_text = p[:a] || p['a'] || p[:answer] || p['answer'] || 'It operates automatically using advanced heuristics.'
416
+ questions = [{ q: q_text, a: a_text }]
417
+ end
418
+
419
+ entities = questions.map do |item|
420
+ q_str = item[:q] || item['q'] || item[:question] || item['question'] || 'Frequently Asked Question'
421
+ a_str = item[:a] || item['a'] || item[:answer] || item['answer'] || 'Answer description.'
422
+ {
423
+ '@type' => 'Question',
424
+ 'name' => q_str,
425
+ 'acceptedAnswer' => {
426
+ '@type' => 'Answer',
427
+ 'text' => a_str
428
+ }
429
+ }
430
+ end
431
+
432
+ {
433
+ '@context' => 'https://schema.org',
434
+ '@type' => 'FAQPage',
435
+ 'mainEntity' => entities
436
+ }
437
+ end
438
+
439
+ def build_howto(p)
440
+ name = p[:name] || p['name'] || p[:headline] || 'How to Optimize Performance'
441
+ desc = p[:description] || p['description'] || "Step-by-step actionable guide for #{name}."
442
+ steps = p[:steps] || p['steps'] || [
443
+ { name: 'Initial Assessment', text: 'Audit current baseline metrics and identify bottlenecks.' },
444
+ { name: 'Implementation', text: 'Deploy high-impact optimizations across key components.' },
445
+ { name: 'Verification', text: 'Validate performance gains and ensure zero regressions.' }
446
+ ]
447
+
448
+ step_entities = steps.each_with_index.map do |s, idx|
449
+ {
450
+ '@type' => 'HowToStep',
451
+ 'position' => (idx + 1).to_s,
452
+ 'name' => s[:name] || s['name'] || "Step #{idx + 1}",
453
+ 'text' => s[:text] || s['text'] || s.to_s
454
+ }
455
+ end
456
+
457
+ {
458
+ '@context' => 'https://schema.org',
459
+ '@type' => 'HowTo',
460
+ 'name' => name,
461
+ 'description' => desc,
462
+ 'totalTime' => p[:time] || p['time'] || 'PT15M',
463
+ 'step' => step_entities
464
+ }
465
+ end
466
+
467
+ def build_article(p)
468
+ headline = p[:headline] || p['headline'] || p[:name] || 'Comprehensive Guide'
469
+ desc = p[:description] || p['description'] || "#{headline}: In-depth analysis and expert insights."
470
+ author_name = p[:author] || p['author'] || 'Editorial Team'
471
+ publisher_name = p[:publisher] || p['publisher'] || 'Publisher'
472
+ url = p[:url] || p['url'] || ''
473
+ image = p[:image] || p['image']
474
+ published = p[:date_published] || p['date_published'] || Time.now.strftime('%Y-%m-%dT%H:%M:%SZ')
475
+
476
+ schema = {
477
+ '@context' => 'https://schema.org',
478
+ '@type' => 'Article',
479
+ 'headline' => headline,
480
+ 'description' => desc,
481
+ 'datePublished' => published,
482
+ 'dateModified' => Time.now.strftime('%Y-%m-%dT%H:%M:%SZ'),
483
+ 'author' => {
484
+ '@type' => 'Person',
485
+ 'name' => author_name
486
+ },
487
+ 'publisher' => {
488
+ '@type' => 'Organization',
489
+ 'name' => publisher_name
490
+ }
491
+ }
492
+
493
+ schema['image'] = [image] if image
494
+ schema['publisher']['logo'] = { '@type' => 'ImageObject', 'url' => p[:logo] || p['logo'] } if (p[:logo] || p['logo'])
495
+ schema['mainEntityOfPage'] = { '@type' => 'WebPage', '@id' => url } unless url.empty?
496
+
497
+ schema
498
+ end
499
+
500
+ def build_software(p)
501
+ name = p[:name] || p['name'] || 'Software Application'
502
+ desc = p[:description] || p['description'] || "#{name} application."
503
+ os = p[:os] || p['os'] || p[:operating_system] || 'Web, macOS, Windows, Linux'
504
+ category = p[:category] || p['category'] || 'BusinessApplication'
505
+ price = (p[:price] || p['price'] || 0.0).to_f.round(2)
506
+ rating_val = (p[:rating] || p['rating']) ? (p[:rating] || p['rating']).to_f.round(1) : nil
507
+ review_count = (p[:reviews] || p['reviews']) ? (p[:reviews] || p['reviews']).to_i : nil
508
+
509
+ schema = {
510
+ '@context' => 'https://schema.org',
511
+ '@type' => 'SoftwareApplication',
512
+ 'name' => name,
513
+ 'operatingSystem' => os,
514
+ 'applicationCategory' => category,
515
+ 'description' => desc,
516
+ 'offers' => {
517
+ '@type' => 'Offer',
518
+ 'price' => price.to_s,
519
+ 'priceCurrency' => p[:currency] || p['currency'] || 'USD'
520
+ }
521
+ }
522
+
523
+ if rating_val && review_count
524
+ schema['aggregateRating'] = {
525
+ '@type' => 'AggregateRating',
526
+ 'ratingValue' => rating_val.to_s,
527
+ 'reviewCount' => review_count.to_s
528
+ }
529
+ end
530
+
531
+ schema
532
+ end
533
+
534
+ def build_local_business(p)
535
+ name = p[:name] || p['name'] || 'Local Business'
536
+ url = p[:url] || p['url'] || ''
537
+
538
+ schema = {
539
+ '@context' => 'https://schema.org',
540
+ '@type' => 'LocalBusiness',
541
+ 'name' => name,
542
+ 'telephone' => p[:telephone] || '+1-555-0199',
543
+ 'address' => {
544
+ '@type' => 'PostalAddress',
545
+ 'streetAddress' => p[:street] || '123 Market Street',
546
+ 'addressLocality' => p[:city] || 'San Francisco',
547
+ 'addressRegion' => p[:state] || 'CA',
548
+ 'postalCode' => p[:zip] || '94103',
549
+ 'addressCountry' => 'US'
550
+ },
551
+ 'openingHoursSpecification' => [
552
+ {
553
+ '@type' => 'OpeningHoursSpecification',
554
+ 'dayOfWeek' => %w[Monday Tuesday Wednesday Thursday Friday],
555
+ 'opens' => '09:00',
556
+ 'closes' => '18:00'
557
+ }
558
+ ]
559
+ }
560
+
561
+ schema['url'] = url unless url.empty?
562
+ schema['image'] = [p[:image]] if p[:image]
563
+ schema
564
+ end
565
+
566
+ def build_breadcrumbs(p)
567
+ items = p[:breadcrumbs] || p['breadcrumbs'] || []
568
+ if items.empty?
569
+ base_url = p[:url] || 'https://example.com'
570
+ items = [
571
+ { name: 'Home', url: base_url },
572
+ { name: p[:name] || 'Current Page', url: "#{base_url}/current" }
573
+ ]
574
+ end
575
+
576
+ item_list = items.each_with_index.map do |it, idx|
577
+ {
578
+ '@type' => 'ListItem',
579
+ 'position' => idx + 1,
580
+ 'name' => it[:name] || it['name'] || "Item #{idx + 1}",
581
+ 'item' => it[:url] || it['url'] || "#{p[:url]}/step-#{idx + 1}"
582
+ }
583
+ end
584
+
585
+ {
586
+ '@context' => 'https://schema.org',
587
+ '@type' => 'BreadcrumbList',
588
+ 'itemListElement' => item_list
589
+ }
590
+ end
591
+
592
+ def build_course(p)
593
+ name = p[:name] || p['name'] || p[:headline] || 'Comprehensive Online Course'
594
+ desc = p[:description] || p['description'] || "Complete course on #{name}."
595
+ provider = p[:brand] || p[:provider] || 'Academy'
596
+
597
+ schema = {
598
+ '@context' => 'https://schema.org',
599
+ '@type' => 'Course',
600
+ 'name' => name,
601
+ 'description' => desc,
602
+ 'provider' => {
603
+ '@type' => 'Organization',
604
+ 'name' => provider,
605
+ 'sameAs' => p[:url] || ''
606
+ }
607
+ }
608
+
609
+ if p[:price]
610
+ schema['offers'] = {
611
+ '@type' => 'Offer',
612
+ 'category' => 'Paid',
613
+ 'price' => (p[:price] || 0.0).to_f.round(2).to_s,
614
+ 'priceCurrency' => p[:currency] || 'USD'
615
+ }
616
+ end
617
+
618
+ schema
619
+ end
620
+
621
+ def build_job_posting(p)
622
+ title = p[:title] || p[:name] || 'Technical Specialist'
623
+ desc = p[:description] || "Exciting opportunity for #{title}."
624
+ hiring_org = p[:brand] || p[:hiring_organization] || 'Company'
625
+ date_posted = p[:date_posted] || Time.now.strftime('%Y-%m-%d')
626
+ valid_through = (Time.now + (90 * 24 * 3600)).strftime('%Y-%m-%d')
627
+
628
+ schema = {
629
+ '@context' => 'https://schema.org',
630
+ '@type' => 'JobPosting',
631
+ 'title' => title,
632
+ 'description' => desc,
633
+ 'datePosted' => date_posted,
634
+ 'validThrough' => valid_through,
635
+ 'employmentType' => p[:employment_type] || 'FULL_TIME',
636
+ 'hiringOrganization' => {
637
+ '@type' => 'Organization',
638
+ 'name' => hiring_org,
639
+ 'sameAs' => p[:url] || ''
640
+ },
641
+ 'jobLocation' => {
642
+ '@type' => 'Place',
643
+ 'address' => {
644
+ '@type' => 'PostalAddress',
645
+ 'addressCountry' => 'US'
646
+ }
647
+ }
648
+ }
649
+
650
+ if p[:price] || p[:salary]
651
+ schema['baseSalary'] = {
652
+ '@type' => 'MonetaryAmount',
653
+ 'currency' => p[:currency] || 'USD',
654
+ 'value' => {
655
+ '@type' => 'QuantitativeValue',
656
+ 'value' => (p[:price] || p[:salary]).to_f,
657
+ 'unitText' => 'YEAR'
658
+ }
659
+ }
660
+ end
661
+
662
+ schema
663
+ end
664
+
665
+ def build_event(p)
666
+ name = p[:name] || p['name'] || 'Featured Event & Workshop'
667
+ desc = p[:description] || p['description'] || "#{name} overview."
668
+ start_date = p[:start_date] || (Time.now + (14 * 24 * 3600)).strftime('%Y-%m-%dT09:00:00Z')
669
+ end_date = p[:end_date] || (Time.now + (14 * 24 * 3600) + 28800).strftime('%Y-%m-%dT17:00:00Z')
670
+
671
+ schema = {
672
+ '@context' => 'https://schema.org',
673
+ '@type' => 'Event',
674
+ 'name' => name,
675
+ 'description' => desc,
676
+ 'startDate' => start_date,
677
+ 'endDate' => end_date,
678
+ 'eventAttendanceMode' => 'https://schema.org/OnlineEventAttendanceMode',
679
+ 'eventStatus' => 'https://schema.org/EventScheduled',
680
+ 'location' => {
681
+ '@type' => 'VirtualLocation',
682
+ 'url' => p[:url] || 'https://example.com/webinar'
683
+ },
684
+ 'organizer' => {
685
+ '@type' => 'Organization',
686
+ 'name' => p[:brand] || 'Event Organizer',
687
+ 'url' => p[:url] || ''
688
+ }
689
+ }
690
+
691
+ if p[:price]
692
+ schema['offers'] = {
693
+ '@type' => 'Offer',
694
+ 'url' => p[:url] || '',
695
+ 'price' => (p[:price] || 0.0).to_f.round(2).to_s,
696
+ 'priceCurrency' => p[:currency] || 'USD',
697
+ 'availability' => 'https://schema.org/InStock'
698
+ }
699
+ end
700
+
701
+ schema
702
+ end
703
+
704
+ def build_video(p)
705
+ name = p[:name] || p['name'] || p[:headline] || 'Product Demo & Walkthrough'
706
+ desc = p[:description] || p['description'] || "#{name} video overview."
707
+ thumb = p[:image] || p[:thumbnail] || 'https://example.com/thumbnail.jpg'
708
+
709
+ {
710
+ '@context' => 'https://schema.org',
711
+ '@type' => 'VideoObject',
712
+ 'name' => name,
713
+ 'description' => desc,
714
+ 'thumbnailUrl' => [thumb],
715
+ 'uploadDate' => p[:upload_date] || Time.now.strftime('%Y-%m-%d'),
716
+ 'contentUrl' => p[:url] || 'https://example.com/video.mp4'
717
+ }
718
+ end
719
+
720
+ def build_recipe(p)
721
+ name = p[:name] || p['name'] || 'Classic Recipe'
722
+ desc = p[:description] || p['description'] || "#{name} step-by-step recipe."
723
+ image = p[:image] || 'https://example.com/recipe.jpg'
724
+
725
+ {
726
+ '@context' => 'https://schema.org',
727
+ '@type' => 'Recipe',
728
+ 'name' => name,
729
+ 'description' => desc,
730
+ 'image' => [image],
731
+ 'recipeIngredient' => p[:ingredients] || ['1 cup ingredient A', '2 tbsp ingredient B'],
732
+ 'recipeInstructions' => [
733
+ { '@type' => 'HowToStep', 'text' => 'Mix ingredients in a large bowl.' },
734
+ { '@type' => 'HowToStep', 'text' => 'Cook for 15 minutes until golden brown.' }
735
+ ]
736
+ }
737
+ end
738
+
739
+ def build_organization(p)
740
+ name = p[:name] || p['name'] || 'Organization Name'
741
+ url = p[:url] || p['url'] || ''
742
+ logo = p[:logo] || p['logo'] || (!url.empty? ? "#{url}/logo.png" : nil)
743
+
744
+ schema = {
745
+ '@context' => 'https://schema.org',
746
+ '@type' => 'Organization',
747
+ 'name' => name,
748
+ 'sameAs' => p[:same_as] || p['same_as'] || []
749
+ }
750
+
751
+ schema['url'] = url unless url.empty?
752
+ schema['logo'] = logo if logo
753
+
754
+ schema
755
+ end
756
+
757
+ def calculate_schema_score(errors, warnings)
758
+ score = 100
759
+ score -= (errors.size * 35)
760
+ score -= (warnings.size * 10)
761
+ [0, score].max
762
+ end
763
+
764
+ def extract_tag_content(html, tag)
765
+ match = html.match(/<#{tag}[^>]*>(.*?)<\/#{tag}>/im)
766
+ match ? strip_html(match[1]) : nil
767
+ end
768
+
769
+ def extract_meta_content(html, name_or_prop)
770
+ pattern = /<meta[^>]+(?:name|property)=["']#{Regexp.escape(name_or_prop)}["'][^>]+content=["']([^"']*)["']/i
771
+ alt_pattern = /<meta[^>]+content=["']([^"']*)["'][^>]+(?:name|property)=["']#{Regexp.escape(name_or_prop)}["']/i
772
+ m = html.match(pattern) || html.match(alt_pattern)
773
+ m ? strip_html(m[1]) : nil
774
+ end
775
+
776
+ def extract_first_h1(html)
777
+ extract_tag_content(html, 'h1')
778
+ end
779
+
780
+ def extract_first_image(html, base_url)
781
+ m = html.match(/<img[^>]+src=["']([^"']+)["']/i)
782
+ return nil unless m
783
+ src = m[1].strip
784
+ return src if src =~ %r{^https?://}
785
+ URI.join(base_url, src).to_s rescue src
786
+ end
787
+ end
788
+ end