gsc-cli 2.0.2 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/AUTH.md +205 -0
- data/FUNDING.md +120 -0
- data/README.md +463 -299
- data/bin/gsc +29158 -4921
- data/dist/gsc +29158 -4921
- data/lib/gsc/aio_hunter.rb +343 -0
- data/lib/gsc/answer_synthesizer.rb +157 -0
- data/lib/gsc/api.rb +53 -1
- data/lib/gsc/auth.rb +26 -0
- data/lib/gsc/backlinks_manager.rb +96 -0
- data/lib/gsc/brand_segmenter.rb +140 -0
- data/lib/gsc/cache_manager.rb +806 -0
- data/lib/gsc/cannibalization_analyzer.rb +141 -0
- data/lib/gsc/canonical_chains.rb +367 -0
- data/lib/gsc/citation_simulator.rb +339 -0
- data/lib/gsc/cli/aio_hunter.rb +154 -0
- data/lib/gsc/cli/analytics.rb +788 -0
- data/lib/gsc/cli/audit.rb +1976 -0
- data/lib/gsc/cli/base.rb +384 -0
- data/lib/gsc/cli/cache.rb +266 -0
- data/lib/gsc/cli/canonical.rb +223 -0
- data/lib/gsc/cli/citation_simulator.rb +152 -0
- data/lib/gsc/cli/dashboard.rb +354 -0
- data/lib/gsc/cli/doctor.rb +129 -0
- data/lib/gsc/cli/eeat.rb +125 -0
- data/lib/gsc/cli/ga4.rb +852 -0
- data/lib/gsc/cli/growth.rb +650 -0
- data/lib/gsc/cli/hreflang.rb +164 -0
- data/lib/gsc/cli/image_seo.rb +162 -0
- data/lib/gsc/cli/indexing.rb +458 -0
- data/lib/gsc/cli/intent_shift.rb +125 -0
- data/lib/gsc/cli/keyword_value.rb +134 -0
- data/lib/gsc/cli/keywords.rb +795 -0
- data/lib/gsc/cli/landing_roi.rb +308 -0
- data/lib/gsc/cli/low_ctr.rb +213 -0
- data/lib/gsc/cli/mobile_parity.rb +150 -0
- data/lib/gsc/cli/report.rb +100 -0
- data/lib/gsc/cli/rich_results.rb +172 -0
- data/lib/gsc/cli/schema_generate.rb +149 -0
- data/lib/gsc/cli/seasonal.rb +232 -0
- data/lib/gsc/cli/security.rb +153 -0
- data/lib/gsc/cli/setup.rb +1291 -0
- data/lib/gsc/cli/sitemap_tree.rb +143 -0
- data/lib/gsc/cli/skill_pack.rb +62 -0
- data/lib/gsc/cli/soft_404.rb +199 -0
- data/lib/gsc/cli/sparkline.rb +227 -0
- data/lib/gsc/cli/watchdog.rb +150 -0
- data/lib/gsc/cli/zombie_purger.rb +208 -0
- data/lib/gsc/cli.rb +706 -5111
- data/lib/gsc/cli_advanced.rb +1513 -0
- data/lib/gsc/client.rb +17 -2
- data/lib/gsc/color.rb +16 -1
- data/lib/gsc/command_registry.rb +47 -9
- data/lib/gsc/config.rb +11 -2
- data/lib/gsc/content_gap.rb +112 -0
- data/lib/gsc/ctr_curve.rb +115 -0
- data/lib/gsc/decay_predictor.rb +322 -0
- data/lib/gsc/doctor.rb +434 -0
- data/lib/gsc/eeat_auditor.rb +428 -0
- data/lib/gsc/entity_auditor.rb +229 -0
- data/lib/gsc/firewall_scanner.rb +733 -0
- data/lib/gsc/geo_auditor.rb +368 -0
- data/lib/gsc/google_suggest.rb +109 -0
- data/lib/gsc/google_trends.rb +8 -1
- data/lib/gsc/heading_validator.rb +283 -0
- data/lib/gsc/hreflang_validator.rb +412 -0
- data/lib/gsc/image_seo.rb +286 -0
- data/lib/gsc/indexing_queue.rb +179 -0
- data/lib/gsc/indexnow.rb +93 -0
- data/lib/gsc/intent_shift.rb +188 -0
- data/lib/gsc/internal_links.rb +249 -0
- data/lib/gsc/keyword_value.rb +191 -0
- data/lib/gsc/landing_roi.rb +195 -0
- data/lib/gsc/llms_generator.rb +425 -0
- data/lib/gsc/low_ctr_rewriter.rb +408 -0
- data/lib/gsc/mobile_parity.rb +222 -0
- data/lib/gsc/network_tracer.rb +93 -0
- data/lib/gsc/open_page_rank.rb +72 -0
- data/lib/gsc/page_analyzer.rb +47 -7
- data/lib/gsc/page_comparator.rb +108 -0
- data/lib/gsc/page_speed.rb +110 -0
- data/lib/gsc/prompts.rb +38 -29
- data/lib/gsc/questions_harvester.rb +178 -0
- data/lib/gsc/report_generator.rb +461 -0
- data/lib/gsc/rich_results.rb +388 -0
- data/lib/gsc/robots_checker.rb +114 -0
- data/lib/gsc/schema_generator.rb +788 -0
- data/lib/gsc/schema_validator.rb +120 -0
- data/lib/gsc/seasonal_predictor.rb +381 -0
- data/lib/gsc/security_scanner.rb +496 -0
- data/lib/gsc/serp_feature_detector.rb +359 -0
- data/lib/gsc/serp_preview.rb +152 -0
- data/lib/gsc/site_crawler.rb +113 -21
- data/lib/gsc/sitemap_loader.rb +15 -4
- data/lib/gsc/sitemap_tree.rb +301 -0
- data/lib/gsc/skill_pack.rb +195 -0
- data/lib/gsc/soft_404_analyzer.rb +385 -0
- data/lib/gsc/sparkline.rb +171 -0
- data/lib/gsc/speed_correlator.rb +416 -0
- data/lib/gsc/striking_playbook.rb +190 -0
- data/lib/gsc/title_optimizer.rb +420 -0
- data/lib/gsc/vault.rb +260 -0
- data/lib/gsc/version.rb +1 -1
- data/lib/gsc/watchdog.rb +235 -0
- data/lib/gsc/zombie_purger.rb +366 -0
- data/lib/gsc.rb +144 -0
- metadata +91 -2
|
@@ -0,0 +1,428 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'net/http'
|
|
5
|
+
require 'uri'
|
|
6
|
+
require 'json'
|
|
7
|
+
require 'time'
|
|
8
|
+
|
|
9
|
+
module GSC
|
|
10
|
+
class EeatAuditor
|
|
11
|
+
CREDENTIAL_PATTERNS = %r{\b(?:MD|PhD|DO|PharmD|RN|CPA|Esq|JD|MBA|MS|MA|BSc|PE|PMP|CTO|CEO|Founder|Lead Architect|Senior Editor|Staff Writer|Medical Director|Specialist|Fellow)\b}i
|
|
12
|
+
REVIEWER_PATTERNS = %r{\b(?:reviewed by|medically reviewed by|fact-checked by|fact checked by|edited by|scientifically verified by|legal review by)\b}i
|
|
13
|
+
DISCLOSURE_PATTERNS = %r{\b(?:editorial guidelines|editorial policy|conflict of interest|affiliate disclosure|corrections policy|code of ethics)\b}i
|
|
14
|
+
|
|
15
|
+
attr_reader :options, :url, :html
|
|
16
|
+
|
|
17
|
+
def initialize(options = {})
|
|
18
|
+
@options = options
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
def self.audit(target, options = {})
|
|
22
|
+
new(options).audit(target)
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def audit(target)
|
|
26
|
+
@url, @html = load_content(target)
|
|
27
|
+
signals = extract_signals(@html, @url)
|
|
28
|
+
evaluate_health(signals)
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
private
|
|
32
|
+
|
|
33
|
+
def load_content(target)
|
|
34
|
+
target_str = target.to_s.strip
|
|
35
|
+
if target_str.match?(%r{^https?://})
|
|
36
|
+
uri = URI.parse(target_str)
|
|
37
|
+
req = Net::HTTP::Get.new(uri)
|
|
38
|
+
req['User-Agent'] = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) gsc-cli/2.1'
|
|
39
|
+
req['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8'
|
|
40
|
+
|
|
41
|
+
res = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == 'https', open_timeout: 8, read_timeout: 10) do |http|
|
|
42
|
+
http.request(req)
|
|
43
|
+
end
|
|
44
|
+
[target_str, res.body.to_s.dup.force_encoding('UTF-8').scrub]
|
|
45
|
+
elsif File.exist?(target_str)
|
|
46
|
+
[target_str, File.read(target_str, encoding: 'UTF-8')]
|
|
47
|
+
else
|
|
48
|
+
[target_str.start_with?('http') ? target_str : 'local-document', target_str]
|
|
49
|
+
end
|
|
50
|
+
rescue StandardError => e
|
|
51
|
+
[target_str.to_s, "<html><body><!-- Error: #{e.message} --></body></html>"]
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def extract_signals(html_str, base_url)
|
|
55
|
+
json_ld_schemas = extract_json_ld(html_str)
|
|
56
|
+
article_schema = find_article_schema(json_ld_schemas)
|
|
57
|
+
person_schema = find_person_schema(json_ld_schemas, article_schema)
|
|
58
|
+
|
|
59
|
+
# 1. Author Identification
|
|
60
|
+
dom_author = extract_dom_author(html_str)
|
|
61
|
+
author_name = person_schema&.dig('name') || dom_author[:name]
|
|
62
|
+
has_author_byline = !author_name.nil? && !author_name.strip.empty?
|
|
63
|
+
has_author_url = !person_schema&.dig('url').nil? || dom_author[:url]
|
|
64
|
+
|
|
65
|
+
# 2. Author Bio & Credentials
|
|
66
|
+
bio_snippet = person_schema&.dig('description') || dom_author[:bio]
|
|
67
|
+
credentials = []
|
|
68
|
+
credentials += author_name.scan(CREDENTIAL_PATTERNS).flatten if author_name
|
|
69
|
+
credentials += bio_snippet.scan(CREDENTIAL_PATTERNS).flatten if bio_snippet
|
|
70
|
+
credentials += person_schema['jobTitle'].scan(CREDENTIAL_PATTERNS).flatten if person_schema && person_schema['jobTitle']
|
|
71
|
+
credentials.uniq!
|
|
72
|
+
|
|
73
|
+
# 3. Entity Disambiguation & Social Profiles (sameAs)
|
|
74
|
+
same_as = Array(person_schema&.dig('sameAs'))
|
|
75
|
+
dom_social_links = extract_author_social_links(html_str)
|
|
76
|
+
all_social_links = (same_as + dom_social_links).uniq
|
|
77
|
+
|
|
78
|
+
has_linkedin = all_social_links.any? { |l| l.include?('linkedin.com') }
|
|
79
|
+
has_twitter = all_social_links.any? { |l| l.include?('twitter.com') || l.include?('x.com') }
|
|
80
|
+
has_wiki = all_social_links.any? { |l| l.include?('wikipedia.org') || l.include?('wikidata.org') }
|
|
81
|
+
has_scholar = all_social_links.any? { |l| l.include?('scholar.google.com') || l.include?('orcid.org') }
|
|
82
|
+
|
|
83
|
+
# 4. Editorial Review Signals
|
|
84
|
+
reviewed_by = article_schema&.dig('reviewedBy')
|
|
85
|
+
reviewer_name = if reviewed_by.is_a?(Hash)
|
|
86
|
+
reviewed_by['name']
|
|
87
|
+
elsif reviewed_by.is_a?(String)
|
|
88
|
+
reviewed_by
|
|
89
|
+
else
|
|
90
|
+
extract_dom_reviewer(html_str)
|
|
91
|
+
end
|
|
92
|
+
has_reviewer = !reviewer_name.nil? && !reviewer_name.strip.empty?
|
|
93
|
+
|
|
94
|
+
# 5. Temporal Freshness Dates
|
|
95
|
+
date_published = article_schema&.dig('datePublished') || extract_dom_meta(html_str, 'article:published_time')
|
|
96
|
+
date_modified = article_schema&.dig('dateModified') || extract_dom_meta(html_str, 'article:modified_time')
|
|
97
|
+
|
|
98
|
+
# 6. Authoritative Outbound Sources (.gov, .edu, doi.org, PubMed)
|
|
99
|
+
authoritative_citations = extract_authoritative_citations(html_str)
|
|
100
|
+
|
|
101
|
+
# 7. Editorial Transparency & Disclosures
|
|
102
|
+
has_disclosure = html_str.match?(DISCLOSURE_PATTERNS)
|
|
103
|
+
|
|
104
|
+
{
|
|
105
|
+
author_name: author_name,
|
|
106
|
+
has_author_byline: has_author_byline,
|
|
107
|
+
has_author_url: has_author_url,
|
|
108
|
+
bio_snippet: bio_snippet,
|
|
109
|
+
credentials: credentials,
|
|
110
|
+
person_schema: person_schema,
|
|
111
|
+
article_schema: article_schema,
|
|
112
|
+
social_profiles: all_social_links,
|
|
113
|
+
has_linkedin: has_linkedin,
|
|
114
|
+
has_twitter: has_twitter,
|
|
115
|
+
has_wiki: has_wiki,
|
|
116
|
+
has_scholar: has_scholar,
|
|
117
|
+
reviewer_name: reviewer_name,
|
|
118
|
+
has_reviewer: has_reviewer,
|
|
119
|
+
date_published: date_published,
|
|
120
|
+
date_modified: date_modified,
|
|
121
|
+
citations: authoritative_citations,
|
|
122
|
+
has_disclosure: has_disclosure
|
|
123
|
+
}
|
|
124
|
+
end
|
|
125
|
+
|
|
126
|
+
def extract_json_ld(html_str)
|
|
127
|
+
schemas = []
|
|
128
|
+
html_str.scan(/<script\b[^>]*?type=["']application\/ld\+json["'][^>]*?>(.*?)<\/script>/im) do |match|
|
|
129
|
+
content = match.first.strip
|
|
130
|
+
begin
|
|
131
|
+
parsed = JSON.parse(content)
|
|
132
|
+
if parsed.is_a?(Array)
|
|
133
|
+
schemas.concat(parsed)
|
|
134
|
+
elsif parsed.is_a?(Hash)
|
|
135
|
+
if parsed['@graph'].is_a?(Array)
|
|
136
|
+
schemas.concat(parsed['@graph'])
|
|
137
|
+
else
|
|
138
|
+
schemas << parsed
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
rescue StandardError
|
|
142
|
+
# Non-JSON or malformed
|
|
143
|
+
end
|
|
144
|
+
end
|
|
145
|
+
schemas
|
|
146
|
+
end
|
|
147
|
+
|
|
148
|
+
def find_article_schema(schemas)
|
|
149
|
+
article_types = %w[Article NewsArticle BlogPosting TechArticle ScholarlyArticle MedicalWebPage WebPage]
|
|
150
|
+
schemas.find do |s|
|
|
151
|
+
type = s['@type']
|
|
152
|
+
if type.is_a?(Array)
|
|
153
|
+
(type & article_types).any?
|
|
154
|
+
else
|
|
155
|
+
article_types.include?(type)
|
|
156
|
+
end
|
|
157
|
+
end
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
def find_person_schema(schemas, article_schema)
|
|
161
|
+
if article_schema && article_schema['author']
|
|
162
|
+
auth = article_schema['author']
|
|
163
|
+
return auth if auth.is_a?(Hash) && auth['@type'] == 'Person'
|
|
164
|
+
return auth.first if auth.is_a?(Array) && auth.first.is_a?(Hash)
|
|
165
|
+
end
|
|
166
|
+
|
|
167
|
+
schemas.find { |s| s['@type'] == 'Person' || s['@type'] == 'ProfilePage' }
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
def extract_dom_author(html_str)
|
|
171
|
+
name = nil
|
|
172
|
+
url = nil
|
|
173
|
+
bio = nil
|
|
174
|
+
|
|
175
|
+
# Meta tag author
|
|
176
|
+
if html_str =~ /<meta\b[^>]*?name=["']author["'][^>]*?content=["'](.*?)["']/im
|
|
177
|
+
name = $1.strip
|
|
178
|
+
elsif html_str =~ /<meta\b[^>]*?property=["']article:author["'][^>]*?content=["'](.*?)["']/im
|
|
179
|
+
name = $1.strip
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
# DOM rel="author" link
|
|
183
|
+
if html_str =~ /<a\b[^>]*?rel=["'][^"']*?\bauthor\b[^"']*?["'][^>]*?href=["'](.*?)["'][^>]*?>(.*?)<\/a>/im
|
|
184
|
+
url = $1.strip
|
|
185
|
+
name ||= strip_tags($2).strip
|
|
186
|
+
end
|
|
187
|
+
|
|
188
|
+
# Author bio container check
|
|
189
|
+
if html_str =~ /<div\b[^>]*?(?:class|id)=["'][^"']*?(?:author-bio|author-description|biography|about-author)[^"']*?["'][^>]*?>(.*?)<\/div>/im
|
|
190
|
+
bio = strip_tags($1).strip
|
|
191
|
+
end
|
|
192
|
+
|
|
193
|
+
{ name: name, url: url, bio: bio }
|
|
194
|
+
end
|
|
195
|
+
|
|
196
|
+
def extract_dom_reviewer(html_str)
|
|
197
|
+
if html_str =~ REVIEWER_PATTERNS
|
|
198
|
+
match_idx = $~.end(0)
|
|
199
|
+
snippet = html_str[match_idx, 120]
|
|
200
|
+
if snippet =~ /^\s*<(?:a|span|strong|b)[^>]*?>(.*?)<\/(?:a|span|strong|b)>/im
|
|
201
|
+
return strip_tags($1).strip
|
|
202
|
+
elsif snippet =~ /^\s*([^<\n\r]+)/im
|
|
203
|
+
clean = strip_tags($1).strip.sub(/[,.]+$/, '')
|
|
204
|
+
return clean unless clean.empty?
|
|
205
|
+
end
|
|
206
|
+
end
|
|
207
|
+
nil
|
|
208
|
+
end
|
|
209
|
+
|
|
210
|
+
def extract_dom_meta(html_str, prop_name)
|
|
211
|
+
if html_str =~ /<meta\b[^>]*?(?:property|name)=["']#{prop_name}["'][^>]*?content=["'](.*?)["']/im
|
|
212
|
+
$1.strip
|
|
213
|
+
else
|
|
214
|
+
nil
|
|
215
|
+
end
|
|
216
|
+
end
|
|
217
|
+
|
|
218
|
+
def extract_author_social_links(html_str)
|
|
219
|
+
links = []
|
|
220
|
+
patterns = [
|
|
221
|
+
%r{https?://(?:www\.)?linkedin\.com/in/[a-zA-Z0-9_-]+},
|
|
222
|
+
%r{https?://(?:www\.)?(?:twitter|x)\.com/[a-zA-Z0-9_]+},
|
|
223
|
+
%r{https?://(?:en\.)?wikipedia\.org/wiki/[a-zA-Z0-9_%-]+},
|
|
224
|
+
%r{https?://scholar\.google\.com/citations\?[a-zA-Z0-9_=&-]+},
|
|
225
|
+
%r{https?://orcid\.org/[0-9-]+}
|
|
226
|
+
]
|
|
227
|
+
|
|
228
|
+
patterns.each do |pat|
|
|
229
|
+
html_str.scan(pat) { |m| links << m }
|
|
230
|
+
end
|
|
231
|
+
|
|
232
|
+
links.uniq
|
|
233
|
+
end
|
|
234
|
+
|
|
235
|
+
def extract_authoritative_citations(html_str)
|
|
236
|
+
cites = []
|
|
237
|
+
html_str.scan(/<a\b[^>]*?href=["'](https?:\/\/[^"']+)["'][^>]*?>/im) do |match|
|
|
238
|
+
href = match.first
|
|
239
|
+
if href =~ %r{https?://[^/]*?\.(?:gov|edu)(?:/|$)}i ||
|
|
240
|
+
href =~ %r{https?://(?:www\.)?(?:doi\.org|ncbi\.nlm\.nih\.gov|pubmed\.ncbi\.nlm\.nih\.gov|nature\.com|sciencedirect\.com|wikipedia\.org)}i
|
|
241
|
+
cites << href
|
|
242
|
+
end
|
|
243
|
+
end
|
|
244
|
+
cites.uniq
|
|
245
|
+
end
|
|
246
|
+
|
|
247
|
+
def evaluate_health(s)
|
|
248
|
+
score = 0.0
|
|
249
|
+
issues = []
|
|
250
|
+
|
|
251
|
+
# 1. Author Identification (20 pts)
|
|
252
|
+
if s[:has_author_byline]
|
|
253
|
+
score += 15.0
|
|
254
|
+
score += 5.0 if s[:has_author_url]
|
|
255
|
+
else
|
|
256
|
+
issues << :missing_author_byline
|
|
257
|
+
end
|
|
258
|
+
|
|
259
|
+
# 2. Schema.org Person & Article Markup (20 pts)
|
|
260
|
+
if s[:person_schema]
|
|
261
|
+
score += 10.0
|
|
262
|
+
score += 5.0 if s[:person_schema]['jobTitle']
|
|
263
|
+
score += 5.0 if Array(s[:person_schema]['sameAs']).any?
|
|
264
|
+
else
|
|
265
|
+
issues << :missing_person_schema
|
|
266
|
+
end
|
|
267
|
+
|
|
268
|
+
# 3. Credentials & Expertise Proof (15 pts)
|
|
269
|
+
if s[:credentials].any?
|
|
270
|
+
score += 15.0
|
|
271
|
+
elsif s[:bio_snippet] && s[:bio_snippet].length > 60
|
|
272
|
+
score += 10.0
|
|
273
|
+
else
|
|
274
|
+
issues << :missing_credentials_or_bio
|
|
275
|
+
end
|
|
276
|
+
|
|
277
|
+
# 4. Social & External Entity Disambiguation (15 pts)
|
|
278
|
+
if s[:has_linkedin] || s[:has_wiki] || s[:has_scholar]
|
|
279
|
+
score += 15.0
|
|
280
|
+
elsif s[:has_twitter]
|
|
281
|
+
score += 10.0
|
|
282
|
+
else
|
|
283
|
+
issues << :missing_social_proof
|
|
284
|
+
end
|
|
285
|
+
|
|
286
|
+
# 5. Editorial Review / Fact-Check Verification (10 pts)
|
|
287
|
+
if s[:has_reviewer]
|
|
288
|
+
score += 10.0
|
|
289
|
+
else
|
|
290
|
+
issues << :missing_editorial_review
|
|
291
|
+
end
|
|
292
|
+
|
|
293
|
+
# 6. Temporal Freshness / Dates (10 pts)
|
|
294
|
+
if s[:date_published] && s[:date_modified]
|
|
295
|
+
score += 10.0
|
|
296
|
+
elsif s[:date_published] || s[:date_modified]
|
|
297
|
+
score += 5.0
|
|
298
|
+
issues << :missing_date_modified
|
|
299
|
+
else
|
|
300
|
+
issues << :missing_dates
|
|
301
|
+
end
|
|
302
|
+
|
|
303
|
+
# 7. Authoritative External Citations (5 pts)
|
|
304
|
+
if s[:citations].any?
|
|
305
|
+
score += 5.0
|
|
306
|
+
else
|
|
307
|
+
issues << :no_authoritative_citations
|
|
308
|
+
end
|
|
309
|
+
|
|
310
|
+
# 8. Editorial Policy & Disclosures (5 pts)
|
|
311
|
+
if s[:has_disclosure]
|
|
312
|
+
score += 5.0
|
|
313
|
+
else
|
|
314
|
+
issues << :missing_editorial_disclosure
|
|
315
|
+
end
|
|
316
|
+
|
|
317
|
+
score = [[score.round(1), 100.0].min, 0.0].max
|
|
318
|
+
grade = compute_grade(score)
|
|
319
|
+
|
|
320
|
+
prescriptions = generate_prescriptions(s, issues)
|
|
321
|
+
snippet = generate_eeat_schema(s, @url)
|
|
322
|
+
|
|
323
|
+
{
|
|
324
|
+
url: @url,
|
|
325
|
+
health_score: score,
|
|
326
|
+
grade: grade,
|
|
327
|
+
author_name: s[:author_name],
|
|
328
|
+
credentials: s[:credentials],
|
|
329
|
+
reviewer_name: s[:reviewer_name],
|
|
330
|
+
date_published: s[:date_published],
|
|
331
|
+
date_modified: s[:date_modified],
|
|
332
|
+
social_profiles: s[:social_profiles],
|
|
333
|
+
citations_count: s[:citations].size,
|
|
334
|
+
issues: issues,
|
|
335
|
+
prescriptions: prescriptions,
|
|
336
|
+
schema_fix_snippet: snippet
|
|
337
|
+
}
|
|
338
|
+
end
|
|
339
|
+
|
|
340
|
+
def generate_prescriptions(s, issues)
|
|
341
|
+
recs = []
|
|
342
|
+
|
|
343
|
+
if issues.include?(:missing_author_byline)
|
|
344
|
+
recs << "Add an explicit author byline and biographical link to attribute content to a named human subject matter expert."
|
|
345
|
+
end
|
|
346
|
+
|
|
347
|
+
if issues.include?(:missing_person_schema)
|
|
348
|
+
recs << "Inject Schema.org Person JSON-LD into the page <head> specifying author name, jobTitle, worksFor, and sameAs profiles."
|
|
349
|
+
end
|
|
350
|
+
|
|
351
|
+
if issues.include?(:missing_credentials_or_bio)
|
|
352
|
+
recs << "State the author's verifiable professional credentials (e.g., MD, CPA, Lead Architect, PhD) and practical industry experience in an author bio box."
|
|
353
|
+
end
|
|
354
|
+
|
|
355
|
+
if issues.include?(:missing_social_proof)
|
|
356
|
+
recs << "Add outgoing links to the author's authoritative profiles (LinkedIn, Wikipedia, or Google Scholar) to establish entity authority in Google's Knowledge Graph."
|
|
357
|
+
end
|
|
358
|
+
|
|
359
|
+
if issues.include?(:missing_editorial_review)
|
|
360
|
+
recs << "Include a secondary reviewer or fact-checker byline (and schema 'reviewedBy') to fulfill Google's YMYL (Your Money or Your Life) standards."
|
|
361
|
+
end
|
|
362
|
+
|
|
363
|
+
if issues.include?(:missing_dates) || issues.include?(:missing_date_modified)
|
|
364
|
+
recs << "Expose clear datePublished and dateModified timestamps in both DOM and schema to signal content freshness."
|
|
365
|
+
end
|
|
366
|
+
|
|
367
|
+
if issues.include?(:no_authoritative_citations)
|
|
368
|
+
recs << "Add 1–3 outbound citations to peer-reviewed studies, government (.gov), or academic (.edu) sources to anchor factual assertions."
|
|
369
|
+
end
|
|
370
|
+
|
|
371
|
+
recs
|
|
372
|
+
end
|
|
373
|
+
|
|
374
|
+
def generate_eeat_schema(s, page_url)
|
|
375
|
+
auth_name = s[:author_name] || "Expert Author"
|
|
376
|
+
job_title = s[:credentials].first ? "#{s[:credentials].first} & Subject Matter Expert" : "Subject Matter Expert"
|
|
377
|
+
same_as_json = s[:social_profiles].any? ? JSON.generate(s[:social_profiles]) : "[\"https://www.linkedin.com/in/author-profile\"]"
|
|
378
|
+
|
|
379
|
+
pub_date = s[:date_published] || Time.now.strftime("%Y-%m-%d")
|
|
380
|
+
mod_date = s[:date_modified] || Time.now.strftime("%Y-%m-%d")
|
|
381
|
+
|
|
382
|
+
reviewer_block = if s[:reviewer_name]
|
|
383
|
+
<<~JSON.strip
|
|
384
|
+
,
|
|
385
|
+
"reviewedBy": {
|
|
386
|
+
"@type": "Person",
|
|
387
|
+
"name": "#{s[:reviewer_name]}",
|
|
388
|
+
"jobTitle": "Editorial & Fact-Checking Lead"
|
|
389
|
+
}
|
|
390
|
+
JSON
|
|
391
|
+
else
|
|
392
|
+
""
|
|
393
|
+
end
|
|
394
|
+
|
|
395
|
+
<<~JSON.strip
|
|
396
|
+
{
|
|
397
|
+
"@context": "https://schema.org",
|
|
398
|
+
"@type": "Article",
|
|
399
|
+
"mainEntityOfPage": "#{page_url}",
|
|
400
|
+
"headline": "Article Headline Here",
|
|
401
|
+
"datePublished": "#{pub_date}",
|
|
402
|
+
"dateModified": "#{mod_date}",
|
|
403
|
+
"author": {
|
|
404
|
+
"@type": "Person",
|
|
405
|
+
"name": "#{auth_name}",
|
|
406
|
+
"jobTitle": "#{job_title}",
|
|
407
|
+
"sameAs": #{same_as_json}
|
|
408
|
+
}#{reviewer_block}
|
|
409
|
+
}
|
|
410
|
+
JSON
|
|
411
|
+
end
|
|
412
|
+
|
|
413
|
+
def compute_grade(score)
|
|
414
|
+
case score
|
|
415
|
+
when 90.0..100.0 then 'A+'
|
|
416
|
+
when 80.0...90.0 then 'A'
|
|
417
|
+
when 70.0...80.0 then 'B'
|
|
418
|
+
when 55.0...70.0 then 'C'
|
|
419
|
+
when 40.0...55.0 then 'D'
|
|
420
|
+
else 'F'
|
|
421
|
+
end
|
|
422
|
+
end
|
|
423
|
+
|
|
424
|
+
def strip_tags(html)
|
|
425
|
+
html.gsub(/<[^>]*>/, '').strip
|
|
426
|
+
end
|
|
427
|
+
end
|
|
428
|
+
end
|
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'net/http'
|
|
4
|
+
require 'uri'
|
|
5
|
+
require 'json'
|
|
6
|
+
|
|
7
|
+
module GSC
|
|
8
|
+
class EntityAuditor
|
|
9
|
+
ENTITY_TYPES = %w[
|
|
10
|
+
Organization Brand Corporation SoftwareApplication WebApplication
|
|
11
|
+
Person LocalBusiness MedicalOrganization EducationalOrganization WebSite
|
|
12
|
+
].freeze
|
|
13
|
+
|
|
14
|
+
HIGH_AUTHORITY_SAMEAS = {
|
|
15
|
+
wikidata: /wikidata\.org\/wiki\/Q\d+/i,
|
|
16
|
+
wikipedia: /wikipedia\.org\/wiki\//i,
|
|
17
|
+
crunchbase: /crunchbase\.com\/(organization|person)\//i,
|
|
18
|
+
linkedin: /linkedin\.com\/(company|in|school)\//i,
|
|
19
|
+
github: /github\.com\//i,
|
|
20
|
+
twitter: /(twitter|x)\.com\//i,
|
|
21
|
+
youtube: /youtube\.com\/(channel|@)/i,
|
|
22
|
+
trustpilot: /trustpilot\.com\/review\//i,
|
|
23
|
+
google_knowledge_graph: /google\.com\/search\?.*kgmid/i
|
|
24
|
+
}.freeze
|
|
25
|
+
|
|
26
|
+
def self.audit(url, custom_html: nil)
|
|
27
|
+
html = custom_html || fetch_html(url)
|
|
28
|
+
return { error: 'Failed to retrieve HTML', score: 0 } unless html
|
|
29
|
+
|
|
30
|
+
schemas = extract_json_ld(html)
|
|
31
|
+
entities = find_entities(schemas)
|
|
32
|
+
|
|
33
|
+
score = 0
|
|
34
|
+
findings = []
|
|
35
|
+
recommendations = []
|
|
36
|
+
same_as_urls = []
|
|
37
|
+
authoritative_links = {}
|
|
38
|
+
|
|
39
|
+
if entities.empty?
|
|
40
|
+
recommendations << 'Add an @type: Organization or Brand JSON-LD schema block to anchor your entity in LLM knowledge graphs.'
|
|
41
|
+
else
|
|
42
|
+
score += 25
|
|
43
|
+
findings << "Detected #{entities.size} Knowledge Graph entity node(s): #{entities.map { |e| e['@type'] }.uniq.join(', ')}"
|
|
44
|
+
|
|
45
|
+
# Analyze primary entity
|
|
46
|
+
primary = entities.find { |e| %w[Organization Corporation Brand SoftwareApplication].include?(e['@type']) } || entities.first
|
|
47
|
+
|
|
48
|
+
# Name check
|
|
49
|
+
if primary['name'] || primary['legalName']
|
|
50
|
+
score += 15
|
|
51
|
+
findings << "Entity Name: \"#{primary['name'] || primary['legalName']}\""
|
|
52
|
+
else
|
|
53
|
+
recommendations << 'Specify explicit "name" and "legalName" in Organization schema.'
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# URL check
|
|
57
|
+
if primary['url']
|
|
58
|
+
score += 10
|
|
59
|
+
findings << "Canonical URL: #{primary['url']}"
|
|
60
|
+
else
|
|
61
|
+
recommendations << 'Add canonical "url" to Organization schema.'
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
# Logo / Image check
|
|
65
|
+
if primary['logo'] || primary['image']
|
|
66
|
+
score += 10
|
|
67
|
+
findings << 'Branding asset (logo/image) is defined.'
|
|
68
|
+
else
|
|
69
|
+
recommendations << 'Add "logo" URL to enhance visual entity search and AI overview previews.'
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
# Description
|
|
73
|
+
if primary['description'] && primary['description'].to_s.length >= 20
|
|
74
|
+
score += 10
|
|
75
|
+
findings << 'Entity description is defined (>= 20 characters).'
|
|
76
|
+
else
|
|
77
|
+
recommendations << 'Add a detailed "description" (40-80 words) summarizing core business category and value prop.'
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
# sameAs Analysis (Crucial for AI / LLM disambiguation)
|
|
81
|
+
raw_same_as = Array(primary['sameAs']).map(&:to_s).reject(&:empty?)
|
|
82
|
+
same_as_urls = raw_same_as
|
|
83
|
+
|
|
84
|
+
if raw_same_as.empty?
|
|
85
|
+
recommendations << 'CRITICAL FOR AI: Add "sameAs" array linking your Wikidata, Wikipedia, LinkedIn, Crunchbase, and official social profiles.'
|
|
86
|
+
else
|
|
87
|
+
score += 15
|
|
88
|
+
findings << "Found #{raw_same_as.size} sameAs identity URL(s)."
|
|
89
|
+
|
|
90
|
+
HIGH_AUTHORITY_SAMEAS.each do |source, pattern|
|
|
91
|
+
match = raw_same_as.find { |u| u =~ pattern }
|
|
92
|
+
if match
|
|
93
|
+
authoritative_links[source] = match
|
|
94
|
+
end
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
if authoritative_links[:wikidata] || authoritative_links[:wikipedia]
|
|
98
|
+
score += 15
|
|
99
|
+
findings << "Strong Entity Anchor: Verified Wikipedia/Wikidata link (#{authoritative_links[:wikidata] || authoritative_links[:wikipedia]})."
|
|
100
|
+
else
|
|
101
|
+
recommendations << 'Add Wikidata or Wikipedia link to sameAs for definitive Google Knowledge Graph & LLM disambiguation.'
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
if authoritative_links[:linkedin] || authoritative_links[:crunchbase]
|
|
105
|
+
findings << "Corporate Validation: Linked #{authoritative_links.keys.map(&:to_s).join(', ')}."
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
# knowsAbout / topic authority
|
|
110
|
+
if primary['knowsAbout']
|
|
111
|
+
score += 5
|
|
112
|
+
findings << "Topic authority defined via knowsAbout: #{Array(primary['knowsAbout']).first(3).join(', ')}"
|
|
113
|
+
end
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
score = [score, 100].min
|
|
117
|
+
|
|
118
|
+
grade = case score
|
|
119
|
+
when 90..100 then 'A'
|
|
120
|
+
when 75..89 then 'B'
|
|
121
|
+
when 50..74 then 'C'
|
|
122
|
+
when 30..49 then 'D'
|
|
123
|
+
else 'F'
|
|
124
|
+
end
|
|
125
|
+
|
|
126
|
+
recommended_schema = generate_template(url, entities.first)
|
|
127
|
+
|
|
128
|
+
{
|
|
129
|
+
url: url,
|
|
130
|
+
score: score,
|
|
131
|
+
grade: grade,
|
|
132
|
+
entities_found: entities.size,
|
|
133
|
+
entities: entities,
|
|
134
|
+
same_as_urls: same_as_urls,
|
|
135
|
+
authoritative_links: authoritative_links,
|
|
136
|
+
findings: findings,
|
|
137
|
+
recommendations: recommendations,
|
|
138
|
+
recommended_json_ld: recommended_schema
|
|
139
|
+
}
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
def self.extract_json_ld(html)
|
|
143
|
+
schemas = []
|
|
144
|
+
html.scan(/<script[^>]*type=["']application\/ld\+json["'][^>]*>(.*?)<\/script>/m) do |match|
|
|
145
|
+
content = match[0].to_s.strip
|
|
146
|
+
parsed = JSON.parse(content) rescue nil
|
|
147
|
+
if parsed.is_a?(Array)
|
|
148
|
+
schemas.concat(parsed)
|
|
149
|
+
elsif parsed.is_a?(Hash)
|
|
150
|
+
if parsed['@graph'].is_a?(Array)
|
|
151
|
+
schemas.concat(parsed['@graph'])
|
|
152
|
+
else
|
|
153
|
+
schemas << parsed
|
|
154
|
+
end
|
|
155
|
+
end
|
|
156
|
+
end
|
|
157
|
+
schemas
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
def self.find_entities(schemas)
|
|
161
|
+
entities = []
|
|
162
|
+
schemas.each do |s|
|
|
163
|
+
next unless s.is_a?(Hash)
|
|
164
|
+
|
|
165
|
+
type = s['@type']
|
|
166
|
+
if type.is_a?(Array)
|
|
167
|
+
entities << s if (type & ENTITY_TYPES).any?
|
|
168
|
+
elsif ENTITY_TYPES.include?(type.to_s)
|
|
169
|
+
entities << s
|
|
170
|
+
end
|
|
171
|
+
|
|
172
|
+
# Check nested entity attributes
|
|
173
|
+
%w[publisher author brand organization].each do |k|
|
|
174
|
+
nested = s[k]
|
|
175
|
+
if nested.is_a?(Hash) && ENTITY_TYPES.include?(nested['@type'].to_s)
|
|
176
|
+
entities << nested
|
|
177
|
+
end
|
|
178
|
+
end
|
|
179
|
+
end
|
|
180
|
+
entities.uniq { |e| "#{e['@type']}_#{e['name']}" }
|
|
181
|
+
end
|
|
182
|
+
|
|
183
|
+
def self.generate_template(url, existing = nil)
|
|
184
|
+
uri = URI.parse(url) rescue nil
|
|
185
|
+
host = uri&.host ? uri.host.sub(/^www\./, '') : url.to_s.sub(%r{^https?://}, '').sub(%r{/.*$}, '')
|
|
186
|
+
host = 'organization' if host.empty?
|
|
187
|
+
name = existing && existing['name'] ? existing['name'] : host.split('.').first.capitalize
|
|
188
|
+
|
|
189
|
+
{
|
|
190
|
+
'@context' => 'https://schema.org',
|
|
191
|
+
'@type' => 'Organization',
|
|
192
|
+
'name' => name,
|
|
193
|
+
'url' => uri ? "#{uri.scheme}://#{uri.host}" : "https://#{host}",
|
|
194
|
+
'logo' => uri ? "#{uri.scheme}://#{uri.host}/logo.png" : "https://#{host}/logo.png",
|
|
195
|
+
'description' => "#{name} official organization entity.",
|
|
196
|
+
'sameAs' => [
|
|
197
|
+
"https://www.linkedin.com/company/#{name.downcase}",
|
|
198
|
+
"https://twitter.com/#{name.downcase}",
|
|
199
|
+
"https://github.com/#{name.downcase}",
|
|
200
|
+
"https://www.crunchbase.com/organization/#{name.downcase}"
|
|
201
|
+
]
|
|
202
|
+
}
|
|
203
|
+
end
|
|
204
|
+
|
|
205
|
+
def self.fetch_html(url)
|
|
206
|
+
uri = URI.parse(url)
|
|
207
|
+
req = Net::HTTP::Get.new(uri)
|
|
208
|
+
req['User-Agent'] = 'Mozilla/5.0 (compatible; GSC-CLI-EntityAuditor/2.1.1; +https://github.com)'
|
|
209
|
+
req['Accept'] = 'text/html,application/xhtml+xml'
|
|
210
|
+
|
|
211
|
+
http = Net::HTTP.new(uri.host, uri.port)
|
|
212
|
+
http.use_ssl = (uri.scheme == 'https')
|
|
213
|
+
http.open_timeout = 8
|
|
214
|
+
http.read_timeout = 10
|
|
215
|
+
|
|
216
|
+
res = http.request(req)
|
|
217
|
+
return nil unless res.is_a?(Net::HTTPSuccess) || res.is_a?(Net::HTTPRedirection)
|
|
218
|
+
|
|
219
|
+
if res.is_a?(Net::HTTPRedirection) && res['location']
|
|
220
|
+
new_url = URI.join(url, res['location']).to_s
|
|
221
|
+
return fetch_html(new_url)
|
|
222
|
+
end
|
|
223
|
+
|
|
224
|
+
res.body.to_s.dup.force_encoding('UTF-8').scrub
|
|
225
|
+
rescue StandardError
|
|
226
|
+
nil
|
|
227
|
+
end
|
|
228
|
+
end
|
|
229
|
+
end
|