@nurkamol/seo-audit 1.40.1 → 1.41.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -5
- package/package.json +1 -1
- package/src/audit.mjs +34 -0
- package/src/checks.mjs +24 -5
- package/src/parse.mjs +30 -5
- package/src/site.mjs +6 -2
package/README.md
CHANGED
|
@@ -279,13 +279,17 @@ change to anything above it.
|
|
|
279
279
|
```
|
|
280
280
|
Preview a Site how big is this, and is it the right one — ~1s, 3 requests
|
|
281
281
|
Audit a Site crawl it and list what to change, worst first
|
|
282
|
-
Recent Reports runs the
|
|
282
|
+
Recent Reports runs the desktop app has already kept
|
|
283
283
|
```
|
|
284
284
|
|
|
285
|
-
`raycast/` is a Raycast extension
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
285
|
+
`raycast/` is a Raycast extension for macOS and Windows. It imports the engine
|
|
286
|
+
as the published `@nurkamol/seo-audit` package, so it re-implements nothing and
|
|
287
|
+
its reports match the terminal's. Raycast runs Node, so unlike the hosted
|
|
288
|
+
version the certificate checks work there.
|
|
289
|
+
|
|
290
|
+
It is waiting for review in the Raycast Store. Until it is listed,
|
|
291
|
+
[docs/raycast.md](docs/raycast.md) explains how to run it from this repository
|
|
292
|
+
in about two minutes, on either platform.
|
|
289
293
|
|
|
290
294
|
**Preview is the command it exists for.** A crawl takes minutes and a launcher
|
|
291
295
|
is built for the second you spend in it, so the headline command is the engine's
|
|
@@ -867,6 +871,7 @@ Findings come at three levels: **error** (wrong, and costing traffic), **warning
|
|
|
867
871
|
| `og:image` is not WebP — LinkedIn won't render it, WhatsApp is unreliable | warning |
|
|
868
872
|
| `og:image` declares width and height | note |
|
|
869
873
|
| `hreflang` codes are well formed — `en_US` with an underscore is the usual slip | error |
|
|
874
|
+
| `hreflang` declared in the XML sitemap counts as declared — Google reads both places, and a finding says which file to fix | — |
|
|
870
875
|
| `hreflang` lists the page itself, not only its translations | warning |
|
|
871
876
|
| `<html lang>` agrees with what the page's own `hreflang` calls it | warning |
|
|
872
877
|
| `<html lang>` agrees with the `Content-Language` header, compared by primary subtag — a header listing several languages agrees if the page's is one of them | warning |
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@nurkamol/seo-audit",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.41.0",
|
|
4
4
|
"description": "Crawl a site's sitemap and check every page for SEO, metadata and structured-data problems that single-page graders miss. Zero dependencies.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
package/src/audit.mjs
CHANGED
|
@@ -165,6 +165,38 @@ async function crawlByLinks(origin, fetcher, { limit, concurrency, robotsGroups,
|
|
|
165
165
|
* @param {string} target site origin, or a sitemap URL
|
|
166
166
|
* @param {{limit?: number, concurrency?: number, sitemap?: string}} opts
|
|
167
167
|
*/
|
|
168
|
+
/** Hreflang the sitemap declared, given to the pages it names.
|
|
169
|
+
*
|
|
170
|
+
* Google reads hreflang from the markup **or** from the sitemap, and treats
|
|
171
|
+
* them the same. This engine only ever read the markup, so a site that chose
|
|
172
|
+
* the sitemap was told "No page declares hreflang" — stated as a fact about
|
|
173
|
+
* the site, under checks reported as not applying. A check that could not run
|
|
174
|
+
* must say so honestly, and that one was saying something false.
|
|
175
|
+
*
|
|
176
|
+
* Only for a page whose own markup declares none. A page that declares both is
|
|
177
|
+
* answering for itself, and quietly merging a second set into it would invent
|
|
178
|
+
* a set neither source contains — which is how a reciprocity check starts
|
|
179
|
+
* reporting pairs nobody wrote.
|
|
180
|
+
*
|
|
181
|
+
* `from` travels with them so a finding can say where to go and fix it: the
|
|
182
|
+
* line to edit is in an XML file, not in the page somebody has open.
|
|
183
|
+
*/
|
|
184
|
+
export function adoptSitemapHreflang(pages, entries = []) {
|
|
185
|
+
const bare = (url) => url.replace(/\/$/, '');
|
|
186
|
+
const declared = new Map(
|
|
187
|
+
entries.filter((e) => e.alternates?.length).map((e) => [bare(e.loc), e.alternates]),
|
|
188
|
+
);
|
|
189
|
+
if (!declared.size) return;
|
|
190
|
+
|
|
191
|
+
for (const page of pages) {
|
|
192
|
+
if (!page.doc || page.doc.hreflang.length) continue;
|
|
193
|
+
const alternates = declared.get(bare(page.url));
|
|
194
|
+
if (!alternates) continue;
|
|
195
|
+
page.doc.hreflang = alternates;
|
|
196
|
+
page.doc.hreflangFrom = 'sitemap';
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
|
|
168
200
|
export async function audit(target, opts = {}) {
|
|
169
201
|
const started = Date.now();
|
|
170
202
|
const fetcher = new Fetcher({ concurrency: opts.concurrency ?? 6, userAgent: opts.userAgent });
|
|
@@ -446,6 +478,8 @@ export async function audit(target, opts = {}) {
|
|
|
446
478
|
|
|
447
479
|
onProgress?.({ phase: 'crawl', detail: `${pages.length} pages in ${((Date.now() - started) / 1000).toFixed(1)}s` });
|
|
448
480
|
|
|
481
|
+
adoptSitemapHreflang(pages, entries);
|
|
482
|
+
|
|
449
483
|
for (const page of pages) findings.push(...pageChecks(page, opts.limits));
|
|
450
484
|
// Click depth is measured from the homepage, and a sitemap need not list it.
|
|
451
485
|
// Fetched here only when the crawl did not already have it, and the fetcher
|
package/src/checks.mjs
CHANGED
|
@@ -164,6 +164,18 @@ export const anchorPhrase = (name) =>
|
|
|
164
164
|
.trim();
|
|
165
165
|
|
|
166
166
|
// --- hreflang ---------------------------------------------------------------
|
|
167
|
+
|
|
168
|
+
/** A sentence naming where a page's hreflang was declared, or nothing.
|
|
169
|
+
*
|
|
170
|
+
* Only when it was not the page itself. `adoptSitemapHreflang()` in audit.mjs
|
|
171
|
+
* gives a page the alternates its sitemap declared for it, because Google
|
|
172
|
+
* reads both; a finding that then sends somebody to a template with no
|
|
173
|
+
* annotation in it would be a correct finding with the wrong address on it. */
|
|
174
|
+
export const hreflangSource = (doc) =>
|
|
175
|
+
doc?.hreflangFrom === 'sitemap'
|
|
176
|
+
? ' This set is declared in the sitemap rather than in the page, so that is the file to fix.'
|
|
177
|
+
: '';
|
|
178
|
+
|
|
167
179
|
// A language, optionally a script, optionally a region, joined by hyphens:
|
|
168
180
|
// en, en-GB, zh-Hant, zh-Hant-TW, en-419. Case is not significant to Google.
|
|
169
181
|
// Only the shape is checked, not whether the codes exist — that would mean
|
|
@@ -485,12 +497,16 @@ export function pageChecks(page, limits = DEFAULT_LIMITS) {
|
|
|
485
497
|
// from the page alone is whether the annotation is well formed and whether
|
|
486
498
|
// the page agrees with it about what the page is.
|
|
487
499
|
if (doc.hreflang.length) {
|
|
500
|
+
// Where the set was declared, when it was not the page. Somebody sent to
|
|
501
|
+
// fix a malformed code should not be reading a template that never had one
|
|
502
|
+
// in it — the line to edit is in the sitemap.
|
|
503
|
+
const where = hreflangSource(doc);
|
|
488
504
|
const malformed = doc.hreflang.filter((alt) => !isLanguageTag(alt.lang));
|
|
489
505
|
for (const alt of malformed) {
|
|
490
506
|
out.push(f('error', 'hreflang-invalid', `Malformed hreflang code: "${alt.lang}"`,
|
|
491
507
|
'Google ignores an annotation it cannot parse, so this version is invisible to it. The form is ' +
|
|
492
508
|
'a language, optionally a script and a region, joined by hyphens — en, en-GB, zh-Hant-TW. ' +
|
|
493
|
-
'An underscore instead of a hyphen is the usual cause.', url));
|
|
509
|
+
'An underscore instead of a hyphen is the usual cause.' + where, url));
|
|
494
510
|
}
|
|
495
511
|
|
|
496
512
|
// Every version has to list itself alongside the others, or the set is
|
|
@@ -499,14 +515,15 @@ export function pageChecks(page, limits = DEFAULT_LIMITS) {
|
|
|
499
515
|
if (!self) {
|
|
500
516
|
out.push(f('warn', 'hreflang-no-self', 'hreflang does not list this page',
|
|
501
517
|
`It points at ${doc.hreflang.map((a) => a.lang).join(', ')} but never at itself. A version that ` +
|
|
502
|
-
'omits its own self-reference leaves the set incomplete.', url));
|
|
518
|
+
'omits its own self-reference leaves the set incomplete.' + where, url));
|
|
503
519
|
} else if (doc.lang && primaryLanguage(self.lang) !== primaryLanguage(doc.lang)) {
|
|
504
520
|
// The page's two statements about its own language, disagreeing. This is
|
|
505
521
|
// only ever visible on a translated page, which is the kind of page a
|
|
506
522
|
// homepage grader never opens.
|
|
507
523
|
out.push(f('warn', 'hreflang-lang-mismatch', 'The page disagrees with its own hreflang about its language',
|
|
508
524
|
`<html lang="${doc.lang}"> but hreflang calls this page "${self.lang}". Google reads both, and one ` +
|
|
509
|
-
'of them is wrong — usually a template that hardcodes lang while the annotation is generated.',
|
|
525
|
+
'of them is wrong — usually a template that hardcodes lang while the annotation is generated.' + where,
|
|
526
|
+
url));
|
|
510
527
|
}
|
|
511
528
|
}
|
|
512
529
|
|
|
@@ -1259,7 +1276,8 @@ export function crossPageChecks(pages, opts = {}) {
|
|
|
1259
1276
|
if (!hasDefault) {
|
|
1260
1277
|
out.push(f('info', 'hreflang-no-x-default', 'No x-default in the hreflang set',
|
|
1261
1278
|
`${translated.length} pages declare alternates and none names an x-default — the version to serve ` +
|
|
1262
|
-
'a visitor whose language matches none of the others. Usually the English or the country selector.'
|
|
1279
|
+
'a visitor whose language matches none of the others. Usually the English or the country selector.'
|
|
1280
|
+
+ hreflangSource(translated[0].doc),
|
|
1263
1281
|
translated[0].url));
|
|
1264
1282
|
}
|
|
1265
1283
|
}
|
|
@@ -1276,7 +1294,8 @@ export function crossPageChecks(pages, opts = {}) {
|
|
|
1276
1294
|
);
|
|
1277
1295
|
if (!returns) {
|
|
1278
1296
|
out.push(f('error', 'hreflang-one-way', 'hreflang is not reciprocal',
|
|
1279
|
-
`${p.url} → ${alt.href} (${alt.lang}), but the target does not link back. Google drops one-way pairs
|
|
1297
|
+
`${p.url} → ${alt.href} (${alt.lang}), but the target does not link back. Google drops one-way pairs.`
|
|
1298
|
+
+ hreflangSource(p.doc), p.url));
|
|
1280
1299
|
}
|
|
1281
1300
|
}
|
|
1282
1301
|
}
|
package/src/parse.mjs
CHANGED
|
@@ -364,11 +364,18 @@ export function parseHtml(rawHtml, pageUrl) {
|
|
|
364
364
|
|
|
365
365
|
/** URLs from a sitemap or sitemap index. Returns {urls, sitemaps, entries}.
|
|
366
366
|
*
|
|
367
|
-
* `entries` pairs each <loc> with its own <lastmod
|
|
368
|
-
* <url> block so
|
|
369
|
-
* rather than a replacement: `urls` stays a
|
|
370
|
-
* every caller wants exactly that and changing
|
|
371
|
-
* discovery for no gain.
|
|
367
|
+
* `entries` pairs each <loc> with its own <lastmod> and its own hreflang
|
|
368
|
+
* alternates, read from inside the <url> block so neither can drift onto a
|
|
369
|
+
* neighbouring URL. It is additional rather than a replacement: `urls` stays a
|
|
370
|
+
* plain list of strings, because every caller wants exactly that and changing
|
|
371
|
+
* it would ripple through discovery for no gain.
|
|
372
|
+
*
|
|
373
|
+
* `alternates` is the same `{ lang, href }` shape `parseHtml` returns for
|
|
374
|
+
* `<link rel="alternate" hreflang>`, because the sitemap is the other place
|
|
375
|
+
* Google reads hreflang from and the checks should not care which one a site
|
|
376
|
+
* chose. A site that declares them here and not in its markup was previously
|
|
377
|
+
* told "no page declares hreflang" — a sentence about the site that was not
|
|
378
|
+
* true of it. */
|
|
372
379
|
export function parseSitemap(xml) {
|
|
373
380
|
const locs = [...xml.matchAll(/<loc>\s*([^<\s]+)\s*<\/loc>/gi)].map((m) => decode(m[1]));
|
|
374
381
|
const isIndex = /<sitemapindex/i.test(xml);
|
|
@@ -378,12 +385,30 @@ export function parseSitemap(xml) {
|
|
|
378
385
|
.map((m) => ({
|
|
379
386
|
loc: decode(m[1].match(/<loc>\s*([^<\s]+)\s*<\/loc>/i)?.[1] ?? ''),
|
|
380
387
|
lastmod: m[1].match(/<lastmod>\s*([^<\s]+)\s*<\/lastmod>/i)?.[1] ?? null,
|
|
388
|
+
alternates: sitemapAlternates(m[1]),
|
|
381
389
|
}))
|
|
382
390
|
.filter((entry) => entry.loc);
|
|
383
391
|
|
|
384
392
|
return { urls: locs, sitemaps: [], entries };
|
|
385
393
|
}
|
|
386
394
|
|
|
395
|
+
/** The hreflang alternates declared inside one <url> block.
|
|
396
|
+
*
|
|
397
|
+
* The prefix is whatever the file bound the XHTML namespace to — `xhtml:link`
|
|
398
|
+
* is the convention Google documents, and a bare `<link>` is what a generator
|
|
399
|
+
* that declared the namespace as the default emits. Matching any prefix costs
|
|
400
|
+
* nothing here: a `<url>` block has no other kind of link in it.
|
|
401
|
+
*
|
|
402
|
+
* `rel` is required to be `alternate` rather than assumed, and an entry with
|
|
403
|
+
* no `hreflang` or no `href` is dropped — half a declaration is not one. */
|
|
404
|
+
function sitemapAlternates(block) {
|
|
405
|
+
return [...block.matchAll(/<(?:[a-z0-9]+:)?link\b[^>]*>/gi)]
|
|
406
|
+
.map((m) => m[0])
|
|
407
|
+
.filter((tag) => (attr(tag, 'rel') ?? '').toLowerCase() === 'alternate')
|
|
408
|
+
.map((tag) => ({ lang: attr(tag, 'hreflang'), href: decode(attr(tag, 'href') ?? '') }))
|
|
409
|
+
.filter((alt) => alt.lang && alt.href);
|
|
410
|
+
}
|
|
411
|
+
|
|
387
412
|
/**
|
|
388
413
|
* What a response body actually is, when the server has already claimed it is
|
|
389
414
|
* HTML. Returns `null` for anything that might be HTML, and a noun for the
|
package/src/site.mjs
CHANGED
|
@@ -5,7 +5,7 @@ import { mapLimit } from './http.mjs';
|
|
|
5
5
|
import { parseRobots, robotsVerdict } from './robots.mjs';
|
|
6
6
|
import { aiAccess, describeAccess } from './agents-ai.mjs';
|
|
7
7
|
import { parseHtml } from './parse.mjs';
|
|
8
|
-
import { schemaNodes, seriesOf, paginatedCanonical } from './checks.mjs';
|
|
8
|
+
import { schemaNodes, seriesOf, paginatedCanonical, hreflangSource } from './checks.mjs';
|
|
9
9
|
import { similarity } from './dupes.mjs';
|
|
10
10
|
import {
|
|
11
11
|
resolve as resolveDns, certificateNames, collapseFleets, rankHosts, looksLikeStaging, NXDOMAIN,
|
|
@@ -638,9 +638,13 @@ export async function siteChecks(origin, fetcher, pages, opts = {}) {
|
|
|
638
638
|
}
|
|
639
639
|
for (const [source, dead] of deadByPage) {
|
|
640
640
|
const shown = dead.slice(0, 3).join(', ');
|
|
641
|
+
// Same reason as the page-level hreflang findings: a set adopted from the
|
|
642
|
+
// sitemap is fixed in the sitemap, not on the page that carries it.
|
|
643
|
+
const doc = pages.find((p) => p.url === source)?.doc;
|
|
641
644
|
out.push(f('error', 'hreflang-dead', `${plural(dead.length, 'hreflang target')} do not load`,
|
|
642
645
|
`${shown}${dead.length > 3 ? `, and ${dead.length - 3} more` : ''} — declared on ${source}. Each ` +
|
|
643
|
-
'version that does not load drops out of the set, and the pages pointing at it lose the annotation.'
|
|
646
|
+
'version that does not load drops out of the set, and the pages pointing at it lose the annotation.'
|
|
647
|
+
+ hreflangSource(doc),
|
|
644
648
|
source));
|
|
645
649
|
}
|
|
646
650
|
|