gsc-cli 2.0.2 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. checksums.yaml +4 -4
  2. data/AUTH.md +205 -0
  3. data/FUNDING.md +120 -0
  4. data/README.md +463 -299
  5. data/bin/gsc +29158 -4921
  6. data/dist/gsc +29158 -4921
  7. data/lib/gsc/aio_hunter.rb +343 -0
  8. data/lib/gsc/answer_synthesizer.rb +157 -0
  9. data/lib/gsc/api.rb +53 -1
  10. data/lib/gsc/auth.rb +26 -0
  11. data/lib/gsc/backlinks_manager.rb +96 -0
  12. data/lib/gsc/brand_segmenter.rb +140 -0
  13. data/lib/gsc/cache_manager.rb +806 -0
  14. data/lib/gsc/cannibalization_analyzer.rb +141 -0
  15. data/lib/gsc/canonical_chains.rb +367 -0
  16. data/lib/gsc/citation_simulator.rb +339 -0
  17. data/lib/gsc/cli/aio_hunter.rb +154 -0
  18. data/lib/gsc/cli/analytics.rb +788 -0
  19. data/lib/gsc/cli/audit.rb +1976 -0
  20. data/lib/gsc/cli/base.rb +384 -0
  21. data/lib/gsc/cli/cache.rb +266 -0
  22. data/lib/gsc/cli/canonical.rb +223 -0
  23. data/lib/gsc/cli/citation_simulator.rb +152 -0
  24. data/lib/gsc/cli/dashboard.rb +354 -0
  25. data/lib/gsc/cli/doctor.rb +129 -0
  26. data/lib/gsc/cli/eeat.rb +125 -0
  27. data/lib/gsc/cli/ga4.rb +852 -0
  28. data/lib/gsc/cli/growth.rb +650 -0
  29. data/lib/gsc/cli/hreflang.rb +164 -0
  30. data/lib/gsc/cli/image_seo.rb +162 -0
  31. data/lib/gsc/cli/indexing.rb +458 -0
  32. data/lib/gsc/cli/intent_shift.rb +125 -0
  33. data/lib/gsc/cli/keyword_value.rb +134 -0
  34. data/lib/gsc/cli/keywords.rb +795 -0
  35. data/lib/gsc/cli/landing_roi.rb +308 -0
  36. data/lib/gsc/cli/low_ctr.rb +213 -0
  37. data/lib/gsc/cli/mobile_parity.rb +150 -0
  38. data/lib/gsc/cli/report.rb +100 -0
  39. data/lib/gsc/cli/rich_results.rb +172 -0
  40. data/lib/gsc/cli/schema_generate.rb +149 -0
  41. data/lib/gsc/cli/seasonal.rb +232 -0
  42. data/lib/gsc/cli/security.rb +153 -0
  43. data/lib/gsc/cli/setup.rb +1291 -0
  44. data/lib/gsc/cli/sitemap_tree.rb +143 -0
  45. data/lib/gsc/cli/skill_pack.rb +62 -0
  46. data/lib/gsc/cli/soft_404.rb +199 -0
  47. data/lib/gsc/cli/sparkline.rb +227 -0
  48. data/lib/gsc/cli/watchdog.rb +150 -0
  49. data/lib/gsc/cli/zombie_purger.rb +208 -0
  50. data/lib/gsc/cli.rb +706 -5111
  51. data/lib/gsc/cli_advanced.rb +1513 -0
  52. data/lib/gsc/client.rb +17 -2
  53. data/lib/gsc/color.rb +16 -1
  54. data/lib/gsc/command_registry.rb +47 -9
  55. data/lib/gsc/config.rb +11 -2
  56. data/lib/gsc/content_gap.rb +112 -0
  57. data/lib/gsc/ctr_curve.rb +115 -0
  58. data/lib/gsc/decay_predictor.rb +322 -0
  59. data/lib/gsc/doctor.rb +434 -0
  60. data/lib/gsc/eeat_auditor.rb +428 -0
  61. data/lib/gsc/entity_auditor.rb +229 -0
  62. data/lib/gsc/firewall_scanner.rb +733 -0
  63. data/lib/gsc/geo_auditor.rb +368 -0
  64. data/lib/gsc/google_suggest.rb +109 -0
  65. data/lib/gsc/google_trends.rb +8 -1
  66. data/lib/gsc/heading_validator.rb +283 -0
  67. data/lib/gsc/hreflang_validator.rb +412 -0
  68. data/lib/gsc/image_seo.rb +286 -0
  69. data/lib/gsc/indexing_queue.rb +179 -0
  70. data/lib/gsc/indexnow.rb +93 -0
  71. data/lib/gsc/intent_shift.rb +188 -0
  72. data/lib/gsc/internal_links.rb +249 -0
  73. data/lib/gsc/keyword_value.rb +191 -0
  74. data/lib/gsc/landing_roi.rb +195 -0
  75. data/lib/gsc/llms_generator.rb +425 -0
  76. data/lib/gsc/low_ctr_rewriter.rb +408 -0
  77. data/lib/gsc/mobile_parity.rb +222 -0
  78. data/lib/gsc/network_tracer.rb +93 -0
  79. data/lib/gsc/open_page_rank.rb +72 -0
  80. data/lib/gsc/page_analyzer.rb +47 -7
  81. data/lib/gsc/page_comparator.rb +108 -0
  82. data/lib/gsc/page_speed.rb +110 -0
  83. data/lib/gsc/prompts.rb +38 -29
  84. data/lib/gsc/questions_harvester.rb +178 -0
  85. data/lib/gsc/report_generator.rb +461 -0
  86. data/lib/gsc/rich_results.rb +388 -0
  87. data/lib/gsc/robots_checker.rb +114 -0
  88. data/lib/gsc/schema_generator.rb +788 -0
  89. data/lib/gsc/schema_validator.rb +120 -0
  90. data/lib/gsc/seasonal_predictor.rb +381 -0
  91. data/lib/gsc/security_scanner.rb +496 -0
  92. data/lib/gsc/serp_feature_detector.rb +359 -0
  93. data/lib/gsc/serp_preview.rb +152 -0
  94. data/lib/gsc/site_crawler.rb +113 -21
  95. data/lib/gsc/sitemap_loader.rb +15 -4
  96. data/lib/gsc/sitemap_tree.rb +301 -0
  97. data/lib/gsc/skill_pack.rb +195 -0
  98. data/lib/gsc/soft_404_analyzer.rb +385 -0
  99. data/lib/gsc/sparkline.rb +171 -0
  100. data/lib/gsc/speed_correlator.rb +416 -0
  101. data/lib/gsc/striking_playbook.rb +190 -0
  102. data/lib/gsc/title_optimizer.rb +420 -0
  103. data/lib/gsc/vault.rb +260 -0
  104. data/lib/gsc/version.rb +1 -1
  105. data/lib/gsc/watchdog.rb +235 -0
  106. data/lib/gsc/zombie_purger.rb +366 -0
  107. data/lib/gsc.rb +144 -0
  108. metadata +91 -2
@@ -0,0 +1,428 @@
1
+ # encoding: utf-8
2
+ # frozen_string_literal: true
3
+
4
+ require 'net/http'
5
+ require 'uri'
6
+ require 'json'
7
+ require 'time'
8
+
9
+ module GSC
10
+ class EeatAuditor
11
+ CREDENTIAL_PATTERNS = %r{\b(?:MD|PhD|DO|PharmD|RN|CPA|Esq|JD|MBA|MS|MA|BSc|PE|PMP|CTO|CEO|Founder|Lead Architect|Senior Editor|Staff Writer|Medical Director|Specialist|Fellow)\b}i
12
+ REVIEWER_PATTERNS = %r{\b(?:reviewed by|medically reviewed by|fact-checked by|fact checked by|edited by|scientifically verified by|legal review by)\b}i
13
+ DISCLOSURE_PATTERNS = %r{\b(?:editorial guidelines|editorial policy|conflict of interest|affiliate disclosure|corrections policy|code of ethics)\b}i
14
+
15
+ attr_reader :options, :url, :html
16
+
17
+ def initialize(options = {})
18
+ @options = options
19
+ end
20
+
21
+ def self.audit(target, options = {})
22
+ new(options).audit(target)
23
+ end
24
+
25
+ def audit(target)
26
+ @url, @html = load_content(target)
27
+ signals = extract_signals(@html, @url)
28
+ evaluate_health(signals)
29
+ end
30
+
31
+ private
32
+
33
+ def load_content(target)
34
+ target_str = target.to_s.strip
35
+ if target_str.match?(%r{^https?://})
36
+ uri = URI.parse(target_str)
37
+ req = Net::HTTP::Get.new(uri)
38
+ req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) gsc-cli/2.1'
39
+ req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
40
+
41
+ res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 8, read_timeout: 10) do |http|
42
+ http.request(req)
43
+ end
44
+ [target_str, res.body.to_s.dup.force_encoding('UTF-8').scrub]
45
+ elsif File.exist?(target_str)
46
+ [target_str, File.read(target_str, encoding: 'UTF-8')]
47
+ else
48
+ [target_str.start_with?('http') ? target_str : 'local-document', target_str]
49
+ end
50
+ rescue StandardError => e
51
+ [target_str.to_s, "<html><body><!-- Error: #{e.message} --></body></html>"]
52
+ end
53
+
54
+ def extract_signals(html_str, base_url)
55
+ json_ld_schemas = extract_json_ld(html_str)
56
+ article_schema = find_article_schema(json_ld_schemas)
57
+ person_schema = find_person_schema(json_ld_schemas, article_schema)
58
+
59
+ # 1. Author Identification
60
+ dom_author = extract_dom_author(html_str)
61
+ author_name = person_schema&.dig('name') || dom_author[:name]
62
+ has_author_byline = !author_name.nil? && !author_name.strip.empty?
63
+ has_author_url = !person_schema&.dig('url').nil? || dom_author[:url]
64
+
65
+ # 2. Author Bio & Credentials
66
+ bio_snippet = person_schema&.dig('description') || dom_author[:bio]
67
+ credentials = []
68
+ credentials += author_name.scan(CREDENTIAL_PATTERNS).flatten if author_name
69
+ credentials += bio_snippet.scan(CREDENTIAL_PATTERNS).flatten if bio_snippet
70
+ credentials += person_schema['jobTitle'].scan(CREDENTIAL_PATTERNS).flatten if person_schema && person_schema['jobTitle']
71
+ credentials.uniq!
72
+
73
+ # 3. Entity Disambiguation & Social Profiles (sameAs)
74
+ same_as = Array(person_schema&.dig('sameAs'))
75
+ dom_social_links = extract_author_social_links(html_str)
76
+ all_social_links = (same_as + dom_social_links).uniq
77
+
78
+ has_linkedin = all_social_links.any? { |l| l.include?('linkedin.com') }
79
+ has_twitter = all_social_links.any? { |l| l.include?('twitter.com') || l.include?('x.com') }
80
+ has_wiki = all_social_links.any? { |l| l.include?('wikipedia.org') || l.include?('wikidata.org') }
81
+ has_scholar = all_social_links.any? { |l| l.include?('scholar.google.com') || l.include?('orcid.org') }
82
+
83
+ # 4. Editorial Review Signals
84
+ reviewed_by = article_schema&.dig('reviewedBy')
85
+ reviewer_name = if reviewed_by.is_a?(Hash)
86
+ reviewed_by['name']
87
+ elsif reviewed_by.is_a?(String)
88
+ reviewed_by
89
+ else
90
+ extract_dom_reviewer(html_str)
91
+ end
92
+ has_reviewer = !reviewer_name.nil? && !reviewer_name.strip.empty?
93
+
94
+ # 5. Temporal Freshness Dates
95
+ date_published = article_schema&.dig('datePublished') || extract_dom_meta(html_str, 'article:published_time')
96
+ date_modified = article_schema&.dig('dateModified') || extract_dom_meta(html_str, 'article:modified_time')
97
+
98
+ # 6. Authoritative Outbound Sources (.gov, .edu, doi.org, PubMed)
99
+ authoritative_citations = extract_authoritative_citations(html_str)
100
+
101
+ # 7. Editorial Transparency & Disclosures
102
+ has_disclosure = html_str.match?(DISCLOSURE_PATTERNS)
103
+
104
+ {
105
+ author_name: author_name,
106
+ has_author_byline: has_author_byline,
107
+ has_author_url: has_author_url,
108
+ bio_snippet: bio_snippet,
109
+ credentials: credentials,
110
+ person_schema: person_schema,
111
+ article_schema: article_schema,
112
+ social_profiles: all_social_links,
113
+ has_linkedin: has_linkedin,
114
+ has_twitter: has_twitter,
115
+ has_wiki: has_wiki,
116
+ has_scholar: has_scholar,
117
+ reviewer_name: reviewer_name,
118
+ has_reviewer: has_reviewer,
119
+ date_published: date_published,
120
+ date_modified: date_modified,
121
+ citations: authoritative_citations,
122
+ has_disclosure: has_disclosure
123
+ }
124
+ end
125
+
126
+ def extract_json_ld(html_str)
127
+ schemas = []
128
+ html_str.scan(/<script\b[^>]*?type=["']application\/ld\+json["'][^>]*?>(.*?)<\/script>/im) do |match|
129
+ content = match.first.strip
130
+ begin
131
+ parsed = JSON.parse(content)
132
+ if parsed.is_a?(Array)
133
+ schemas.concat(parsed)
134
+ elsif parsed.is_a?(Hash)
135
+ if parsed['@graph'].is_a?(Array)
136
+ schemas.concat(parsed['@graph'])
137
+ else
138
+ schemas << parsed
139
+ end
140
+ end
141
+ rescue StandardError
142
+ # Non-JSON or malformed
143
+ end
144
+ end
145
+ schemas
146
+ end
147
+
148
+ def find_article_schema(schemas)
149
+ article_types = %w[Article NewsArticle BlogPosting TechArticle ScholarlyArticle MedicalWebPage WebPage]
150
+ schemas.find do |s|
151
+ type = s['@type']
152
+ if type.is_a?(Array)
153
+ (type & article_types).any?
154
+ else
155
+ article_types.include?(type)
156
+ end
157
+ end
158
+ end
159
+
160
+ def find_person_schema(schemas, article_schema)
161
+ if article_schema && article_schema['author']
162
+ auth = article_schema['author']
163
+ return auth if auth.is_a?(Hash) && auth['@type'] == 'Person'
164
+ return auth.first if auth.is_a?(Array) && auth.first.is_a?(Hash)
165
+ end
166
+
167
+ schemas.find { |s| s['@type'] == 'Person' || s['@type'] == 'ProfilePage' }
168
+ end
169
+
170
+ def extract_dom_author(html_str)
171
+ name = nil
172
+ url = nil
173
+ bio = nil
174
+
175
+ # Meta tag author
176
+ if html_str =~ /<meta\b[^>]*?name=["']author["'][^>]*?content=["'](.*?)["']/im
177
+ name = $1.strip
178
+ elsif html_str =~ /<meta\b[^>]*?property=["']article:author["'][^>]*?content=["'](.*?)["']/im
179
+ name = $1.strip
180
+ end
181
+
182
+ # DOM rel="author" link
183
+ if html_str =~ /<a\b[^>]*?rel=["'][^"']*?\bauthor\b[^"']*?["'][^>]*?href=["'](.*?)["'][^>]*?>(.*?)<\/a>/im
184
+ url = $1.strip
185
+ name ||= strip_tags($2).strip
186
+ end
187
+
188
+ # Author bio container check
189
+ if html_str =~ /<div\b[^>]*?(?:class|id)=["'][^"']*?(?:author-bio|author-description|biography|about-author)[^"']*?["'][^>]*?>(.*?)<\/div>/im
190
+ bio = strip_tags($1).strip
191
+ end
192
+
193
+ { name: name, url: url, bio: bio }
194
+ end
195
+
196
+ def extract_dom_reviewer(html_str)
197
+ if html_str =~ REVIEWER_PATTERNS
198
+ match_idx = $~.end(0)
199
+ snippet = html_str[match_idx, 120]
200
+ if snippet =~ /^\s*<(?:a|span|strong|b)[^>]*?>(.*?)<\/(?:a|span|strong|b)>/im
201
+ return strip_tags($1).strip
202
+ elsif snippet =~ /^\s*([^<\n\r]+)/im
203
+ clean = strip_tags($1).strip.sub(/[,.]+$/, '')
204
+ return clean unless clean.empty?
205
+ end
206
+ end
207
+ nil
208
+ end
209
+
210
+ def extract_dom_meta(html_str, prop_name)
211
+ if html_str =~ /<meta\b[^>]*?(?:property|name)=["']#{prop_name}["'][^>]*?content=["'](.*?)["']/im
212
+ $1.strip
213
+ else
214
+ nil
215
+ end
216
+ end
217
+
218
+ def extract_author_social_links(html_str)
219
+ links = []
220
+ patterns = [
221
+ %r{https?://(?:www\.)?linkedin\.com/in/[a-zA-Z0-9_-]+},
222
+ %r{https?://(?:www\.)?(?:twitter|x)\.com/[a-zA-Z0-9_]+},
223
+ %r{https?://(?:en\.)?wikipedia\.org/wiki/[a-zA-Z0-9_%-]+},
224
+ %r{https?://scholar\.google\.com/citations\?[a-zA-Z0-9_=&-]+},
225
+ %r{https?://orcid\.org/[0-9-]+}
226
+ ]
227
+
228
+ patterns.each do |pat|
229
+ html_str.scan(pat) { |m| links << m }
230
+ end
231
+
232
+ links.uniq
233
+ end
234
+
235
+ def extract_authoritative_citations(html_str)
236
+ cites = []
237
+ html_str.scan(/<a\b[^>]*?href=["'](https?:\/\/[^"']+)["'][^>]*?>/im) do |match|
238
+ href = match.first
239
+ if href =~ %r{https?://[^/]*?\.(?:gov|edu)(?:/|$)}i ||
240
+ href =~ %r{https?://(?:www\.)?(?:doi\.org|ncbi\.nlm\.nih\.gov|pubmed\.ncbi\.nlm\.nih\.gov|nature\.com|sciencedirect\.com|wikipedia\.org)}i
241
+ cites << href
242
+ end
243
+ end
244
+ cites.uniq
245
+ end
246
+
247
+ def evaluate_health(s)
248
+ score = 0.0
249
+ issues = []
250
+
251
+ # 1. Author Identification (20 pts)
252
+ if s[:has_author_byline]
253
+ score += 15.0
254
+ score += 5.0 if s[:has_author_url]
255
+ else
256
+ issues << :missing_author_byline
257
+ end
258
+
259
+ # 2. Schema.org Person & Article Markup (20 pts)
260
+ if s[:person_schema]
261
+ score += 10.0
262
+ score += 5.0 if s[:person_schema]['jobTitle']
263
+ score += 5.0 if Array(s[:person_schema]['sameAs']).any?
264
+ else
265
+ issues << :missing_person_schema
266
+ end
267
+
268
+ # 3. Credentials & Expertise Proof (15 pts)
269
+ if s[:credentials].any?
270
+ score += 15.0
271
+ elsif s[:bio_snippet] && s[:bio_snippet].length > 60
272
+ score += 10.0
273
+ else
274
+ issues << :missing_credentials_or_bio
275
+ end
276
+
277
+ # 4. Social & External Entity Disambiguation (15 pts)
278
+ if s[:has_linkedin] || s[:has_wiki] || s[:has_scholar]
279
+ score += 15.0
280
+ elsif s[:has_twitter]
281
+ score += 10.0
282
+ else
283
+ issues << :missing_social_proof
284
+ end
285
+
286
+ # 5. Editorial Review / Fact-Check Verification (10 pts)
287
+ if s[:has_reviewer]
288
+ score += 10.0
289
+ else
290
+ issues << :missing_editorial_review
291
+ end
292
+
293
+ # 6. Temporal Freshness / Dates (10 pts)
294
+ if s[:date_published] && s[:date_modified]
295
+ score += 10.0
296
+ elsif s[:date_published] || s[:date_modified]
297
+ score += 5.0
298
+ issues << :missing_date_modified
299
+ else
300
+ issues << :missing_dates
301
+ end
302
+
303
+ # 7. Authoritative External Citations (5 pts)
304
+ if s[:citations].any?
305
+ score += 5.0
306
+ else
307
+ issues << :no_authoritative_citations
308
+ end
309
+
310
+ # 8. Editorial Policy & Disclosures (5 pts)
311
+ if s[:has_disclosure]
312
+ score += 5.0
313
+ else
314
+ issues << :missing_editorial_disclosure
315
+ end
316
+
317
+ score = [[score.round(1), 100.0].min, 0.0].max
318
+ grade = compute_grade(score)
319
+
320
+ prescriptions = generate_prescriptions(s, issues)
321
+ snippet = generate_eeat_schema(s, @url)
322
+
323
+ {
324
+ url: @url,
325
+ health_score: score,
326
+ grade: grade,
327
+ author_name: s[:author_name],
328
+ credentials: s[:credentials],
329
+ reviewer_name: s[:reviewer_name],
330
+ date_published: s[:date_published],
331
+ date_modified: s[:date_modified],
332
+ social_profiles: s[:social_profiles],
333
+ citations_count: s[:citations].size,
334
+ issues: issues,
335
+ prescriptions: prescriptions,
336
+ schema_fix_snippet: snippet
337
+ }
338
+ end
339
+
340
+ def generate_prescriptions(s, issues)
341
+ recs = []
342
+
343
+ if issues.include?(:missing_author_byline)
344
+ recs << "Add an explicit author byline and biographical link to attribute content to a named human subject matter expert."
345
+ end
346
+
347
+ if issues.include?(:missing_person_schema)
348
+ recs << "Inject Schema.org Person JSON-LD into the page <head> specifying author name, jobTitle, worksFor, and sameAs profiles."
349
+ end
350
+
351
+ if issues.include?(:missing_credentials_or_bio)
352
+ recs << "State the author's verifiable professional credentials (e.g., MD, CPA, Lead Architect, PhD) and practical industry experience in an author bio box."
353
+ end
354
+
355
+ if issues.include?(:missing_social_proof)
356
+ recs << "Add outgoing links to the author's authoritative profiles (LinkedIn, Wikipedia, or Google Scholar) to establish entity authority in Google's Knowledge Graph."
357
+ end
358
+
359
+ if issues.include?(:missing_editorial_review)
360
+ recs << "Include a secondary reviewer or fact-checker byline (and schema 'reviewedBy') to fulfill Google's YMYL (Your Money or Your Life) standards."
361
+ end
362
+
363
+ if issues.include?(:missing_dates) || issues.include?(:missing_date_modified)
364
+ recs << "Expose clear datePublished and dateModified timestamps in both DOM and schema to signal content freshness."
365
+ end
366
+
367
+ if issues.include?(:no_authoritative_citations)
368
+ recs << "Add 1–3 outbound citations to peer-reviewed studies, government (.gov), or academic (.edu) sources to anchor factual assertions."
369
+ end
370
+
371
+ recs
372
+ end
373
+
374
+ def generate_eeat_schema(s, page_url)
375
+ auth_name = s[:author_name] || "Expert Author"
376
+ job_title = s[:credentials].first ? "#{s[:credentials].first} & Subject Matter Expert" : "Subject Matter Expert"
377
+ same_as_json = s[:social_profiles].any? ? JSON.generate(s[:social_profiles]) : "[\"https://www.linkedin.com/in/author-profile\"]"
378
+
379
+ pub_date = s[:date_published] || Time.now.strftime("%Y-%m-%d")
380
+ mod_date = s[:date_modified] || Time.now.strftime("%Y-%m-%d")
381
+
382
+ reviewer_block = if s[:reviewer_name]
383
+ <<~JSON.strip
384
+ ,
385
+ "reviewedBy": {
386
+ "@type": "Person",
387
+ "name": "#{s[:reviewer_name]}",
388
+ "jobTitle": "Editorial & Fact-Checking Lead"
389
+ }
390
+ JSON
391
+ else
392
+ ""
393
+ end
394
+
395
+ <<~JSON.strip
396
+ {
397
+ "@context": "https://schema.org",
398
+ "@type": "Article",
399
+ "mainEntityOfPage": "#{page_url}",
400
+ "headline": "Article Headline Here",
401
+ "datePublished": "#{pub_date}",
402
+ "dateModified": "#{mod_date}",
403
+ "author": {
404
+ "@type": "Person",
405
+ "name": "#{auth_name}",
406
+ "jobTitle": "#{job_title}",
407
+ "sameAs": #{same_as_json}
408
+ }#{reviewer_block}
409
+ }
410
+ JSON
411
+ end
412
+
413
+ def compute_grade(score)
414
+ case score
415
+ when 90.0..100.0 then 'A+'
416
+ when 80.0...90.0 then 'A'
417
+ when 70.0...80.0 then 'B'
418
+ when 55.0...70.0 then 'C'
419
+ when 40.0...55.0 then 'D'
420
+ else 'F'
421
+ end
422
+ end
423
+
424
+ def strip_tags(html)
425
+ html.gsub(/<[^>]*>/, '').strip
426
+ end
427
+ end
428
+ end
@@ -0,0 +1,229 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'net/http'
4
+ require 'uri'
5
+ require 'json'
6
+
7
+ module GSC
8
+ class EntityAuditor
9
+ ENTITY_TYPES = %w[
10
+ Organization Brand Corporation SoftwareApplication WebApplication
11
+ Person LocalBusiness MedicalOrganization EducationalOrganization WebSite
12
+ ].freeze
13
+
14
+ HIGH_AUTHORITY_SAMEAS = {
15
+ wikidata: /wikidata\.org\/wiki\/Q\d+/i,
16
+ wikipedia: /wikipedia\.org\/wiki\//i,
17
+ crunchbase: /crunchbase\.com\/(organization|person)\//i,
18
+ linkedin: /linkedin\.com\/(company|in|school)\//i,
19
+ github: /github\.com\//i,
20
+ twitter: /(twitter|x)\.com\//i,
21
+ youtube: /youtube\.com\/(channel|@)/i,
22
+ trustpilot: /trustpilot\.com\/review\//i,
23
+ google_knowledge_graph: /google\.com\/search\?.*kgmid/i
24
+ }.freeze
25
+
26
+ def self.audit(url, custom_html: nil)
27
+ html = custom_html || fetch_html(url)
28
+ return { error: 'Failed to retrieve HTML', score: 0 } unless html
29
+
30
+ schemas = extract_json_ld(html)
31
+ entities = find_entities(schemas)
32
+
33
+ score = 0
34
+ findings = []
35
+ recommendations = []
36
+ same_as_urls = []
37
+ authoritative_links = {}
38
+
39
+ if entities.empty?
40
+ recommendations << 'Add an @type: Organization or Brand JSON-LD schema block to anchor your entity in LLM knowledge graphs.'
41
+ else
42
+ score += 25
43
+ findings << "Detected #{entities.size} Knowledge Graph entity node(s): #{entities.map { |e| e['@type'] }.uniq.join(', ')}"
44
+
45
+ # Analyze primary entity
46
+ primary = entities.find { |e| %w[Organization Corporation Brand SoftwareApplication].include?(e['@type']) } || entities.first
47
+
48
+ # Name check
49
+ if primary['name'] || primary['legalName']
50
+ score += 15
51
+ findings << "Entity Name: \"#{primary['name'] || primary['legalName']}\""
52
+ else
53
+ recommendations << 'Specify explicit "name" and "legalName" in Organization schema.'
54
+ end
55
+
56
+ # URL check
57
+ if primary['url']
58
+ score += 10
59
+ findings << "Canonical URL: #{primary['url']}"
60
+ else
61
+ recommendations << 'Add canonical "url" to Organization schema.'
62
+ end
63
+
64
+ # Logo / Image check
65
+ if primary['logo'] || primary['image']
66
+ score += 10
67
+ findings << 'Branding asset (logo/image) is defined.'
68
+ else
69
+ recommendations << 'Add "logo" URL to enhance visual entity search and AI overview previews.'
70
+ end
71
+
72
+ # Description
73
+ if primary['description'] && primary['description'].to_s.length >= 20
74
+ score += 10
75
+ findings << 'Entity description is defined (>= 20 characters).'
76
+ else
77
+ recommendations << 'Add a detailed "description" (40-80 words) summarizing core business category and value prop.'
78
+ end
79
+
80
+ # sameAs Analysis (Crucial for AI / LLM disambiguation)
81
+ raw_same_as = Array(primary['sameAs']).map(&:to_s).reject(&:empty?)
82
+ same_as_urls = raw_same_as
83
+
84
+ if raw_same_as.empty?
85
+ recommendations << 'CRITICAL FOR AI: Add "sameAs" array linking your Wikidata, Wikipedia, LinkedIn, Crunchbase, and official social profiles.'
86
+ else
87
+ score += 15
88
+ findings << "Found #{raw_same_as.size} sameAs identity URL(s)."
89
+
90
+ HIGH_AUTHORITY_SAMEAS.each do |source, pattern|
91
+ match = raw_same_as.find { |u| u =~ pattern }
92
+ if match
93
+ authoritative_links[source] = match
94
+ end
95
+ end
96
+
97
+ if authoritative_links[:wikidata] || authoritative_links[:wikipedia]
98
+ score += 15
99
+ findings << "Strong Entity Anchor: Verified Wikipedia/Wikidata link (#{authoritative_links[:wikidata] || authoritative_links[:wikipedia]})."
100
+ else
101
+ recommendations << 'Add Wikidata or Wikipedia link to sameAs for definitive Google Knowledge Graph & LLM disambiguation.'
102
+ end
103
+
104
+ if authoritative_links[:linkedin] || authoritative_links[:crunchbase]
105
+ findings << "Corporate Validation: Linked #{authoritative_links.keys.map(&:to_s).join(', ')}."
106
+ end
107
+ end
108
+
109
+ # knowsAbout / topic authority
110
+ if primary['knowsAbout']
111
+ score += 5
112
+ findings << "Topic authority defined via knowsAbout: #{Array(primary['knowsAbout']).first(3).join(', ')}"
113
+ end
114
+ end
115
+
116
+ score = [score, 100].min
117
+
118
+ grade = case score
119
+ when 90..100 then 'A'
120
+ when 75..89 then 'B'
121
+ when 50..74 then 'C'
122
+ when 30..49 then 'D'
123
+ else 'F'
124
+ end
125
+
126
+ recommended_schema = generate_template(url, entities.first)
127
+
128
+ {
129
+ url: url,
130
+ score: score,
131
+ grade: grade,
132
+ entities_found: entities.size,
133
+ entities: entities,
134
+ same_as_urls: same_as_urls,
135
+ authoritative_links: authoritative_links,
136
+ findings: findings,
137
+ recommendations: recommendations,
138
+ recommended_json_ld: recommended_schema
139
+ }
140
+ end
141
+
142
+ def self.extract_json_ld(html)
143
+ schemas = []
144
+ html.scan(/<script[^>]*type=["']application\/ld\+json["'][^>]*>(.*?)<\/script>/m) do |match|
145
+ content = match[0].to_s.strip
146
+ parsed = JSON.parse(content) rescue nil
147
+ if parsed.is_a?(Array)
148
+ schemas.concat(parsed)
149
+ elsif parsed.is_a?(Hash)
150
+ if parsed['@graph'].is_a?(Array)
151
+ schemas.concat(parsed['@graph'])
152
+ else
153
+ schemas << parsed
154
+ end
155
+ end
156
+ end
157
+ schemas
158
+ end
159
+
160
+ def self.find_entities(schemas)
161
+ entities = []
162
+ schemas.each do |s|
163
+ next unless s.is_a?(Hash)
164
+
165
+ type = s['@type']
166
+ if type.is_a?(Array)
167
+ entities << s if (type & ENTITY_TYPES).any?
168
+ elsif ENTITY_TYPES.include?(type.to_s)
169
+ entities << s
170
+ end
171
+
172
+ # Check nested entity attributes
173
+ %w[publisher author brand organization].each do |k|
174
+ nested = s[k]
175
+ if nested.is_a?(Hash) && ENTITY_TYPES.include?(nested['@type'].to_s)
176
+ entities << nested
177
+ end
178
+ end
179
+ end
180
+ entities.uniq { |e| "#{e['@type']}_#{e['name']}" }
181
+ end
182
+
183
+ def self.generate_template(url, existing = nil)
184
+ uri = URI.parse(url) rescue nil
185
+ host = uri&.host ? uri.host.sub(/^www\./, '') : url.to_s.sub(%r{^https?://}, '').sub(%r{/.*$}, '')
186
+ host = 'organization' if host.empty?
187
+ name = existing && existing['name'] ? existing['name'] : host.split('.').first.capitalize
188
+
189
+ {
190
+ '@context' => 'https://schema.org',
191
+ '@type' => 'Organization',
192
+ 'name' => name,
193
+ 'url' => uri ? "#{uri.scheme}://#{uri.host}" : "https://#{host}",
194
+ 'logo' => uri ? "#{uri.scheme}://#{uri.host}/logo.png" : "https://#{host}/logo.png",
195
+ 'description' => "#{name} official organization entity.",
196
+ 'sameAs' => [
197
+ "https://www.linkedin.com/company/#{name.downcase}",
198
+ "https://twitter.com/#{name.downcase}",
199
+ "https://github.com/#{name.downcase}",
200
+ "https://www.crunchbase.com/organization/#{name.downcase}"
201
+ ]
202
+ }
203
+ end
204
+
205
+ def self.fetch_html(url)
206
+ uri = URI.parse(url)
207
+ req = Net::HTTP::Get.new(uri)
208
+ req['User-Agent'] = 'Mozilla/5.0 (compatible; GSC-CLI-EntityAuditor/2.1.1; +https://github.com)'
209
+ req['Accept'] = 'text/html,application/xhtml+xml'
210
+
211
+ http = Net::HTTP.new(uri.host, uri.port)
212
+ http.use_ssl = (uri.scheme == 'https')
213
+ http.open_timeout = 8
214
+ http.read_timeout = 10
215
+
216
+ res = http.request(req)
217
+ return nil unless res.is_a?(Net::HTTPSuccess) || res.is_a?(Net::HTTPRedirection)
218
+
219
+ if res.is_a?(Net::HTTPRedirection) && res['location']
220
+ new_url = URI.join(url, res['location']).to_s
221
+ return fetch_html(new_url)
222
+ end
223
+
224
+ res.body.to_s.dup.force_encoding('UTF-8').scrub
225
+ rescue StandardError
226
+ nil
227
+ end
228
+ end
229
+ end