pagetrace 0.10.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +31 -0
- package/README.md +21 -1
- package/dist/cli.cjs +154 -21
- package/dist/cli.js +154 -21
- package/dist/index.cjs +95 -11
- package/dist/index.d.cts +17 -5
- package/dist/index.d.ts +17 -5
- package/dist/index.js +95 -11
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,35 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
|
6
6
|
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
7
|
While the version is below 1.0.0, breaking changes ship in a minor release.
|
|
8
8
|
|
|
9
|
+
## [0.12.0] - 2026-09-08
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- `--external` checks links that leave the site, on `links`, `audit`, `check`
|
|
14
|
+
and `snapshot`. Only 404 and 410 count as dead: a 403 from a bot wall, a 429,
|
|
15
|
+
a timeout and a TLS failure all describe the request rather than the page.
|
|
16
|
+
`HEAD` first with a `GET` fallback, one request per unique URL across the
|
|
17
|
+
site, capped at 200. Off by default — it fans out to hosts you do not control,
|
|
18
|
+
and recording their availability in a lockfile turns a third party's bad
|
|
19
|
+
afternoon into a diff in your repository.
|
|
20
|
+
- `link.external.dead` (warn) and `link.external.dead.added`. A warning rather
|
|
21
|
+
than an error: the page belongs to someone who can delete it without asking.
|
|
22
|
+
|
|
23
|
+
### Fixed
|
|
24
|
+
|
|
25
|
+
- Absolute internal links were invisible. `extractLinks` dropped every href with
|
|
26
|
+
a scheme, so a site writing `https://example.com/about` rather than `/about`
|
|
27
|
+
had its links checked not at all. They are now resolved against the site's own
|
|
28
|
+
origin like any other link.
|
|
29
|
+
|
|
30
|
+
## [0.11.0] - 2026-09-08
|
|
31
|
+
|
|
32
|
+
### Added
|
|
33
|
+
|
|
34
|
+
- `pagetrace links`, a command for the one question. It crawls once and reports
|
|
35
|
+
only broken internal links, exiting 1 if there are any. The rules are the ones
|
|
36
|
+
`audit` runs, filtered rather than reimplemented, so the two cannot disagree.
|
|
37
|
+
|
|
9
38
|
## [0.10.0] - 2026-09-08
|
|
10
39
|
|
|
11
40
|
### Added
|
|
@@ -330,6 +359,8 @@ Initial release. `snapshot`, `check` and `audit` commands; filesystem and HTTP
|
|
|
330
359
|
crawling; diff classified by transition; absolute, cross-page and hreflang audit
|
|
331
360
|
rules; pretty, JSON, markdown, GitHub and HTML reporters.
|
|
332
361
|
|
|
362
|
+
[0.12.0]: https://github.com/shyamexe/pagetrace/compare/v0.11.0...v0.12.0
|
|
363
|
+
[0.11.0]: https://github.com/shyamexe/pagetrace/compare/v0.10.0...v0.11.0
|
|
333
364
|
[0.10.0]: https://github.com/shyamexe/pagetrace/compare/v0.9.1...v0.10.0
|
|
334
365
|
[0.9.1]: https://github.com/shyamexe/pagetrace/compare/v0.9.0...v0.9.1
|
|
335
366
|
[0.9.0]: https://github.com/shyamexe/pagetrace/compare/v0.8.1...v0.9.0
|
package/README.md
CHANGED
|
@@ -160,12 +160,32 @@ pagetrace update # install it globally
|
|
|
160
160
|
| `redirect.changed` | warn | A route redirects somewhere new |
|
|
161
161
|
| `canonical.redirects` | warn | A canonical points at a URL that redirects |
|
|
162
162
|
| `link.broken.added` | error | A page started linking to a URL that does not exist |
|
|
163
|
+
| `link.external.dead` | warn | An outbound link answers 404 (needs `--external`) |
|
|
163
164
|
| `sitemap.dead` | error | The sitemap lists a URL that answers 404 |
|
|
164
165
|
| `sitemap.redirect` | warn | The sitemap lists a URL that redirects |
|
|
165
166
|
| `og.removed` / `hreflang.removed` | warn | Social or i18n tags dropped |
|
|
166
167
|
| `title.changed` | info | Ordinary copy edit |
|
|
167
168
|
|
|
168
|
-
|
|
169
|
+
Broken links have a command of their own, when that is the only question you have:
|
|
170
|
+
|
|
171
|
+
```bash
|
|
172
|
+
npx pagetrace links --url https://example.com
|
|
173
|
+
npx pagetrace links --dir ./out --format json
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
It crawls once, runs the same rules as `audit`, and prints only the link findings — exit 1 if any, or `No broken links found — 42 pages checked.`
|
|
177
|
+
|
|
178
|
+
Internal links are checked always. Outbound links are checked on request:
|
|
179
|
+
|
|
180
|
+
```bash
|
|
181
|
+
npx pagetrace links --url https://example.com --external
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
Only `404` and `410` count as dead. A `403` from a bot wall, a `429`, a timeout or a TLS failure are all reported as nothing at all, because they describe the request rather than the page — which is how link checkers turn into noise nobody reads. `HEAD` first, falling back to `GET` for servers that refuse it, one request per unique URL across the whole site, capped at 200.
|
|
185
|
+
|
|
186
|
+
Think twice before putting `--external` in `check`. A third party's bad afternoon becomes a diff in your repository and a red build you cannot fix.
|
|
187
|
+
|
|
188
|
+
Internal links are checked too. Only the broken ones are stored, so a site's navigation never lands in the lockfile: `link.broken` for a link that is already dead, `link.broken.added` for one this build broke. A `--dir` crawl is authoritative — the build directory is the whole site — while a crawl confirms each candidate with a real request first, because a sitemap routinely omits pages that are live. External links are checked only with `--external`, and only a 404 or 410 counts.
|
|
169
189
|
|
|
170
190
|
Redirects are recorded from the response itself, so they cost no extra requests. A redirect that only adds or drops a trailing slash is server configuration rather than drift and is not reported. `canonical.redirects` is only raised when the canonical's target was actually crawled, so a `--limit` run cannot invent it.
|
|
171
191
|
|
package/dist/cli.cjs
CHANGED
|
@@ -254,13 +254,24 @@ function auditPage(page, config = {}) {
|
|
|
254
254
|
return findings;
|
|
255
255
|
}
|
|
256
256
|
function auditLinks(page) {
|
|
257
|
-
return
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
257
|
+
return [
|
|
258
|
+
...(page.brokenLinks ?? []).map((target) => ({
|
|
259
|
+
code: "link.broken",
|
|
260
|
+
severity: "error",
|
|
261
|
+
route: page.route,
|
|
262
|
+
message: `Links to ${target}, which does not exist.`,
|
|
263
|
+
after: target
|
|
264
|
+
})),
|
|
265
|
+
// A warning, not an error: the page it points at belongs to somebody else,
|
|
266
|
+
// who can delete it on a Tuesday without asking you.
|
|
267
|
+
...(page.deadExternal ?? []).map((target) => ({
|
|
268
|
+
code: "link.external.dead",
|
|
269
|
+
severity: "warn",
|
|
270
|
+
route: page.route,
|
|
271
|
+
message: `Links out to ${target}, which answers 404.`,
|
|
272
|
+
after: target
|
|
273
|
+
}))
|
|
274
|
+
];
|
|
264
275
|
}
|
|
265
276
|
function auditSite(snapshot) {
|
|
266
277
|
const findings = [];
|
|
@@ -631,6 +642,12 @@ function diffPage(before, after) {
|
|
|
631
642
|
push("link.broken.removed", "info", `Link to ${target} was fixed or removed.`, {
|
|
632
643
|
before: target
|
|
633
644
|
});
|
|
645
|
+
const deadBefore = before.deadExternal ?? [];
|
|
646
|
+
const deadAfter = after.deadExternal ?? [];
|
|
647
|
+
for (const target of deadAfter.filter((t) => !deadBefore.includes(t)))
|
|
648
|
+
push("link.external.dead.added", "warn", `Links out to ${target}, which answers 404.`, {
|
|
649
|
+
after: target
|
|
650
|
+
});
|
|
634
651
|
const droppedHreflang = Object.keys(before.hreflang).filter((k) => !(k in after.hreflang));
|
|
635
652
|
if (droppedHreflang.length > 0)
|
|
636
653
|
push("hreflang.removed", "warn", "hreflang alternates were removed.", {
|
|
@@ -739,6 +756,10 @@ var GUIDANCE = {
|
|
|
739
756
|
nextjs: "A `<Link href>` pointing at a route that no longer exists. TypeScript will not catch it \u2014 typed routes are opt-in via `experimental.typedRoutes`."
|
|
740
757
|
}
|
|
741
758
|
},
|
|
759
|
+
"link.external.dead": {
|
|
760
|
+
why: "An outbound link to a page that is gone sends readers to a 404 and spends the trust the link was passing on nothing. Unlike an internal link, the page is not yours to restore.",
|
|
761
|
+
fix: "Point the link at the current URL, at an archived copy, or remove it. A link that has been dead a while is usually a citation worth replacing rather than deleting."
|
|
762
|
+
},
|
|
742
763
|
"sitemap.dead": {
|
|
743
764
|
why: "A sitemap is a list of URLs you are asking to have crawled. Entries that 404 spend crawl budget on nothing and lower the trust placed in the rest of the file.",
|
|
744
765
|
fix: "Remove the URL from the sitemap, or restore the page. If it moved, redirect it and list the destination instead.",
|
|
@@ -1520,8 +1541,8 @@ function extractLinks(html) {
|
|
|
1520
1541
|
const hrefs = /* @__PURE__ */ new Set();
|
|
1521
1542
|
for (const anchor of root.querySelectorAll("a[href]")) {
|
|
1522
1543
|
const href = anchor.getAttribute("href")?.trim();
|
|
1523
|
-
if (!href) continue;
|
|
1524
|
-
if (
|
|
1544
|
+
if (!href || href.startsWith("#")) continue;
|
|
1545
|
+
if (/^[a-z][a-z0-9+.-]*:/i.test(href) && !/^https?:/i.test(href)) continue;
|
|
1525
1546
|
hrefs.add(href);
|
|
1526
1547
|
}
|
|
1527
1548
|
return [...hrefs];
|
|
@@ -1615,6 +1636,7 @@ async function walkHtml(dir, acc = []) {
|
|
|
1615
1636
|
}
|
|
1616
1637
|
return acc;
|
|
1617
1638
|
}
|
|
1639
|
+
var USER_AGENT = "pagetrace (+https://npmjs.com/package/pagetrace)";
|
|
1618
1640
|
var DEFAULT_TIMEOUT_MS = 15e3;
|
|
1619
1641
|
var MAX_ATTEMPTS = 3;
|
|
1620
1642
|
var RETRY_BASE_MS = 300;
|
|
@@ -1630,7 +1652,7 @@ async function fetchDoc(url, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
|
1630
1652
|
try {
|
|
1631
1653
|
response = await fetch(url, {
|
|
1632
1654
|
signal: AbortSignal.timeout(timeoutMs),
|
|
1633
|
-
headers: { "user-agent":
|
|
1655
|
+
headers: { "user-agent": USER_AGENT }
|
|
1634
1656
|
});
|
|
1635
1657
|
} catch (cause) {
|
|
1636
1658
|
lastError = new Error(`Could not reach ${url}: ${cause.message}`, { cause });
|
|
@@ -1675,7 +1697,9 @@ async function snapshotFromDir(dir, config = {}) {
|
|
|
1675
1697
|
if (robots !== null) site.robotsTxt = extractRobotsTxt(robots, agents);
|
|
1676
1698
|
const llms = await (0, import_promises.readFile)((0, import_node_path.join)(dir, "llms.txt"), "utf8").catch(() => null);
|
|
1677
1699
|
if (llms !== null) site.llmsTxt = extractLlmsTxt(llms);
|
|
1678
|
-
|
|
1700
|
+
const origin = site.origin ?? "https://pagetrace.invalid";
|
|
1701
|
+
await resolveBrokenLinks(pages, links, origin);
|
|
1702
|
+
if (config.checkExternal) await resolveDeadExternal(pages, links, origin, 5);
|
|
1679
1703
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1680
1704
|
}
|
|
1681
1705
|
var ASSET_PATH = /\.(?!html?$)[a-z0-9]+$/i;
|
|
@@ -1714,6 +1738,63 @@ async function resolveBrokenLinks(pages, links, origin, verify) {
|
|
|
1714
1738
|
if (confirmed.length > 0) pages[route].brokenLinks = confirmed.sort();
|
|
1715
1739
|
}
|
|
1716
1740
|
}
|
|
1741
|
+
function externalUrl(href, from, origin) {
|
|
1742
|
+
try {
|
|
1743
|
+
const url = new URL(href, `${origin}${from === "/" ? "" : from}/`);
|
|
1744
|
+
if (url.protocol !== "http:" && url.protocol !== "https:") return null;
|
|
1745
|
+
if (url.origin === origin) return null;
|
|
1746
|
+
url.hash = "";
|
|
1747
|
+
return url.href;
|
|
1748
|
+
} catch {
|
|
1749
|
+
return null;
|
|
1750
|
+
}
|
|
1751
|
+
}
|
|
1752
|
+
var MAX_EXTERNAL_CHECKS = 200;
|
|
1753
|
+
async function findDeadExternal(urls, concurrency, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
1754
|
+
const dead = /* @__PURE__ */ new Set();
|
|
1755
|
+
const queue = urls.slice(0, MAX_EXTERNAL_CHECKS);
|
|
1756
|
+
const request = (url, method) => fetch(url, {
|
|
1757
|
+
method,
|
|
1758
|
+
redirect: "follow",
|
|
1759
|
+
signal: AbortSignal.timeout(timeoutMs),
|
|
1760
|
+
headers: { "user-agent": USER_AGENT }
|
|
1761
|
+
});
|
|
1762
|
+
await Promise.all(
|
|
1763
|
+
Array.from({ length: Math.min(concurrency, queue.length) }, async () => {
|
|
1764
|
+
while (queue.length > 0) {
|
|
1765
|
+
const url = queue.shift();
|
|
1766
|
+
try {
|
|
1767
|
+
let response = await request(url, "HEAD");
|
|
1768
|
+
if (response.status === 405 || response.status === 501) {
|
|
1769
|
+
response = await request(url, "GET");
|
|
1770
|
+
}
|
|
1771
|
+
if (response.status === 404 || response.status === 410) dead.add(url);
|
|
1772
|
+
} catch {
|
|
1773
|
+
}
|
|
1774
|
+
}
|
|
1775
|
+
})
|
|
1776
|
+
);
|
|
1777
|
+
return dead;
|
|
1778
|
+
}
|
|
1779
|
+
async function resolveDeadExternal(pages, links, origin, concurrency, timeoutMs) {
|
|
1780
|
+
const perPage = /* @__PURE__ */ new Map();
|
|
1781
|
+
const unique = /* @__PURE__ */ new Set();
|
|
1782
|
+
for (const [route, hrefs] of Object.entries(links)) {
|
|
1783
|
+
const external = [
|
|
1784
|
+
...new Set(
|
|
1785
|
+
hrefs.map((href) => externalUrl(href, route, origin)).filter((url) => url !== null)
|
|
1786
|
+
)
|
|
1787
|
+
];
|
|
1788
|
+
if (external.length === 0) continue;
|
|
1789
|
+
perPage.set(route, external);
|
|
1790
|
+
for (const url of external) unique.add(url);
|
|
1791
|
+
}
|
|
1792
|
+
const dead = await findDeadExternal([...unique], concurrency, timeoutMs);
|
|
1793
|
+
for (const [route, external] of perPage) {
|
|
1794
|
+
const found = external.filter((url) => dead.has(url)).sort();
|
|
1795
|
+
if (found.length > 0) pages[route].deadExternal = found;
|
|
1796
|
+
}
|
|
1797
|
+
}
|
|
1717
1798
|
var MAX_NESTED_SITEMAPS = 50;
|
|
1718
1799
|
async function snapshotFromOrigin(origin, options = {}) {
|
|
1719
1800
|
const base = new URL(origin);
|
|
@@ -1802,6 +1883,9 @@ async function snapshotFromOrigin(origin, options = {}) {
|
|
|
1802
1883
|
return false;
|
|
1803
1884
|
}
|
|
1804
1885
|
});
|
|
1886
|
+
if (options.checkExternal) {
|
|
1887
|
+
await resolveDeadExternal(pages, links, base.origin, concurrency, timeout);
|
|
1888
|
+
}
|
|
1805
1889
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1806
1890
|
}
|
|
1807
1891
|
|
|
@@ -1854,13 +1938,15 @@ async function loadConfig(path = DEFAULT_CONFIG) {
|
|
|
1854
1938
|
}
|
|
1855
1939
|
}
|
|
1856
1940
|
async function build(flags, config) {
|
|
1857
|
-
|
|
1941
|
+
const checkExternal = flags.external ?? config.checkExternal;
|
|
1942
|
+
if (flags.dir) return snapshotFromDir(flags.dir, { ...config, checkExternal });
|
|
1858
1943
|
if (flags.url)
|
|
1859
1944
|
return snapshotFromOrigin(flags.url, {
|
|
1860
1945
|
...config,
|
|
1861
1946
|
limit: flags.limit,
|
|
1862
1947
|
concurrency: flags.concurrency,
|
|
1863
|
-
ignoreRobots: flags.ignoreRobots ?? config.ignoreRobots
|
|
1948
|
+
ignoreRobots: flags.ignoreRobots ?? config.ignoreRobots,
|
|
1949
|
+
checkExternal
|
|
1864
1950
|
});
|
|
1865
1951
|
throw new Error("Provide a source: --dir <build directory> or --url <origin>.");
|
|
1866
1952
|
}
|
|
@@ -1879,7 +1965,7 @@ function render(findings, format) {
|
|
|
1879
1965
|
}
|
|
1880
1966
|
}
|
|
1881
1967
|
var cli = (0, import_cac.cac)("pagetrace");
|
|
1882
|
-
cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
1968
|
+
cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
1883
1969
|
const config = await loadConfig(flags.config);
|
|
1884
1970
|
const snapshot = await build(flags, config);
|
|
1885
1971
|
const written = await writeLockfile(flags.out, snapshot);
|
|
@@ -1888,7 +1974,7 @@ cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").opti
|
|
|
1888
1974
|
written ? import_picocolors2.default.green(`Wrote ${flags.out} \u2014 ${count} page${count === 1 ? "" : "s"}.`) : import_picocolors2.default.dim(`${flags.out} is already up to date \u2014 ${count} page${count === 1 ? "" : "s"}.`)
|
|
1889
1975
|
);
|
|
1890
1976
|
});
|
|
1891
|
-
cli.command("check", "Compare the current surface against the lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--lockfile <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | github | sarif", { default: "pretty" }).option("--fail-on <severity>", "error | warn | info", { default: "error" }).option("--audit", "Also run absolute rules, not just the diff", { default: true }).option("--update", "Write the new state to the lockfile after reporting").option("--baseline-branch <ref>", "Read the baseline lockfile from a git ref instead of disk").action(async (flags) => {
|
|
1977
|
+
cli.command("check", "Compare the current surface against the lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--lockfile <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | github | sarif", { default: "pretty" }).option("--fail-on <severity>", "error | warn | info", { default: "error" }).option("--audit", "Also run absolute rules, not just the diff", { default: true }).option("--update", "Write the new state to the lockfile after reporting").option("--baseline-branch <ref>", "Read the baseline lockfile from a git ref instead of disk").action(async (flags) => {
|
|
1892
1978
|
const failOn = parseFailOn(flags.failOn, false);
|
|
1893
1979
|
const config = await loadConfig(flags.config);
|
|
1894
1980
|
const next = await build(flags, config);
|
|
@@ -1918,7 +2004,7 @@ Failing: ${summary.error} error, ${summary.warn} warning (--fail-on ${failOn}).`
|
|
|
1918
2004
|
process.exitCode = EXIT_FINDINGS;
|
|
1919
2005
|
}
|
|
1920
2006
|
});
|
|
1921
|
-
cli.command("audit", "Audit a site as it stands, with explanations and fixes").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | html", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "never" }).action(async (flags) => {
|
|
2007
|
+
cli.command("audit", "Audit a site as it stands, with explanations and fixes").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | html", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "never" }).action(async (flags) => {
|
|
1922
2008
|
const failOn = parseFailOn(flags.failOn, true);
|
|
1923
2009
|
const config = await loadConfig(flags.config);
|
|
1924
2010
|
const snapshot = await build(flags, config);
|
|
@@ -1958,7 +2044,54 @@ cli.command("audit", "Audit a site as it stands, with explanations and fixes").o
|
|
|
1958
2044
|
process.exitCode = EXIT_FINDINGS;
|
|
1959
2045
|
}
|
|
1960
2046
|
});
|
|
1961
|
-
cli.command("
|
|
2047
|
+
cli.command("links", "Find internal links that point at no page").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "error" }).action(async (flags) => {
|
|
2048
|
+
const failOn = parseFailOn(flags.failOn, true);
|
|
2049
|
+
const config = await loadConfig(flags.config);
|
|
2050
|
+
const snapshot = await build(flags, config);
|
|
2051
|
+
const pages = Object.values(snapshot.pages);
|
|
2052
|
+
if (pages.length === 0) {
|
|
2053
|
+
console.error(
|
|
2054
|
+
import_picocolors2.default.yellow(
|
|
2055
|
+
"No pages found. Check that the sitemap is reachable, or pass --dir with pre-rendered HTML."
|
|
2056
|
+
)
|
|
2057
|
+
);
|
|
2058
|
+
process.exitCode = EXIT_FAILURE;
|
|
2059
|
+
return;
|
|
2060
|
+
}
|
|
2061
|
+
const platform = detectPlatform(
|
|
2062
|
+
pages.map((p) => p.generator),
|
|
2063
|
+
pages.flatMap((p) => Object.values(p.og))
|
|
2064
|
+
);
|
|
2065
|
+
const findings = applyConfig(auditSnapshot(snapshot, config), config).filter((f) => f.code.startsWith("link.")).map((f) => withGuidance(f, platform));
|
|
2066
|
+
const target = flags.url ?? flags.dir ?? "site";
|
|
2067
|
+
if (findings.length === 0 && !flags.out) {
|
|
2068
|
+
console.log(
|
|
2069
|
+
import_picocolors2.default.green(
|
|
2070
|
+
`No broken links found \u2014 ${pages.length} page${pages.length === 1 ? "" : "s"} checked.`
|
|
2071
|
+
)
|
|
2072
|
+
);
|
|
2073
|
+
return;
|
|
2074
|
+
}
|
|
2075
|
+
const groups = aggregate(findings);
|
|
2076
|
+
const meta = {
|
|
2077
|
+
target,
|
|
2078
|
+
platform,
|
|
2079
|
+
pageCount: pages.length,
|
|
2080
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString().slice(0, 10)
|
|
2081
|
+
};
|
|
2082
|
+
const output = flags.format === "json" ? JSON.stringify({ schemaVersion: 1, meta, summary: summarize(findings), groups }, null, 2) : flags.format === "markdown" ? formatAuditMarkdown(groups, meta) : formatAuditPretty(groups, meta, process.stdout.columns);
|
|
2083
|
+
if (flags.out) {
|
|
2084
|
+
await (0, import_promises2.writeFile)(flags.out, `${output}
|
|
2085
|
+
`, "utf8");
|
|
2086
|
+
console.log(import_picocolors2.default.green(`Wrote ${flags.out} \u2014 ${groups.length} broken link${groups.length === 1 ? "" : "s"}.`));
|
|
2087
|
+
} else {
|
|
2088
|
+
console.log(output);
|
|
2089
|
+
}
|
|
2090
|
+
if (failOn !== "never" && shouldFail(findings, failOn)) {
|
|
2091
|
+
process.exitCode = EXIT_FINDINGS;
|
|
2092
|
+
}
|
|
2093
|
+
});
|
|
2094
|
+
cli.command("init", "Write a config file and take the first snapshot").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
1962
2095
|
const existing = await (0, import_promises2.readFile)(flags.config, "utf8").catch(() => null);
|
|
1963
2096
|
if (existing === null) {
|
|
1964
2097
|
const config = flags.url ? { siteUrl: new URL(flags.url).origin } : {};
|
|
@@ -1981,11 +2114,11 @@ Commit both files, then run \`pagetrace check ${flags.dir ? `--dir ${flags.dir}`
|
|
|
1981
2114
|
});
|
|
1982
2115
|
cli.command("update", "Check npm for a newer pagetrace and install it").option("--check", "Only report whether an update exists").action(async (flags) => {
|
|
1983
2116
|
const latest = await latestVersion("pagetrace");
|
|
1984
|
-
if (!isNewer(latest, "0.
|
|
1985
|
-
console.log(import_picocolors2.default.green(`pagetrace ${"0.
|
|
2117
|
+
if (!isNewer(latest, "0.12.0")) {
|
|
2118
|
+
console.log(import_picocolors2.default.green(`pagetrace ${"0.12.0"} is the latest version.`));
|
|
1986
2119
|
return;
|
|
1987
2120
|
}
|
|
1988
|
-
console.log(import_picocolors2.default.yellow(`Update available: ${"0.
|
|
2121
|
+
console.log(import_picocolors2.default.yellow(`Update available: ${"0.12.0"} \u2192 ${latest}`));
|
|
1989
2122
|
if (flags.check) return;
|
|
1990
2123
|
if (process.argv[1]?.startsWith(process.cwd())) {
|
|
1991
2124
|
console.log(
|
|
@@ -2003,7 +2136,7 @@ cli.command("update", "Check npm for a newer pagetrace and install it").option("
|
|
|
2003
2136
|
console.log(import_picocolors2.default.green(`Updated to pagetrace ${latest}.`));
|
|
2004
2137
|
});
|
|
2005
2138
|
cli.help();
|
|
2006
|
-
cli.version("0.
|
|
2139
|
+
cli.version("0.12.0");
|
|
2007
2140
|
async function main() {
|
|
2008
2141
|
try {
|
|
2009
2142
|
cli.parse(process.argv, { run: false });
|
package/dist/cli.js
CHANGED
|
@@ -231,13 +231,24 @@ function auditPage(page, config = {}) {
|
|
|
231
231
|
return findings;
|
|
232
232
|
}
|
|
233
233
|
function auditLinks(page) {
|
|
234
|
-
return
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
234
|
+
return [
|
|
235
|
+
...(page.brokenLinks ?? []).map((target) => ({
|
|
236
|
+
code: "link.broken",
|
|
237
|
+
severity: "error",
|
|
238
|
+
route: page.route,
|
|
239
|
+
message: `Links to ${target}, which does not exist.`,
|
|
240
|
+
after: target
|
|
241
|
+
})),
|
|
242
|
+
// A warning, not an error: the page it points at belongs to somebody else,
|
|
243
|
+
// who can delete it on a Tuesday without asking you.
|
|
244
|
+
...(page.deadExternal ?? []).map((target) => ({
|
|
245
|
+
code: "link.external.dead",
|
|
246
|
+
severity: "warn",
|
|
247
|
+
route: page.route,
|
|
248
|
+
message: `Links out to ${target}, which answers 404.`,
|
|
249
|
+
after: target
|
|
250
|
+
}))
|
|
251
|
+
];
|
|
241
252
|
}
|
|
242
253
|
function auditSite(snapshot) {
|
|
243
254
|
const findings = [];
|
|
@@ -608,6 +619,12 @@ function diffPage(before, after) {
|
|
|
608
619
|
push("link.broken.removed", "info", `Link to ${target} was fixed or removed.`, {
|
|
609
620
|
before: target
|
|
610
621
|
});
|
|
622
|
+
const deadBefore = before.deadExternal ?? [];
|
|
623
|
+
const deadAfter = after.deadExternal ?? [];
|
|
624
|
+
for (const target of deadAfter.filter((t) => !deadBefore.includes(t)))
|
|
625
|
+
push("link.external.dead.added", "warn", `Links out to ${target}, which answers 404.`, {
|
|
626
|
+
after: target
|
|
627
|
+
});
|
|
611
628
|
const droppedHreflang = Object.keys(before.hreflang).filter((k) => !(k in after.hreflang));
|
|
612
629
|
if (droppedHreflang.length > 0)
|
|
613
630
|
push("hreflang.removed", "warn", "hreflang alternates were removed.", {
|
|
@@ -716,6 +733,10 @@ var GUIDANCE = {
|
|
|
716
733
|
nextjs: "A `<Link href>` pointing at a route that no longer exists. TypeScript will not catch it \u2014 typed routes are opt-in via `experimental.typedRoutes`."
|
|
717
734
|
}
|
|
718
735
|
},
|
|
736
|
+
"link.external.dead": {
|
|
737
|
+
why: "An outbound link to a page that is gone sends readers to a 404 and spends the trust the link was passing on nothing. Unlike an internal link, the page is not yours to restore.",
|
|
738
|
+
fix: "Point the link at the current URL, at an archived copy, or remove it. A link that has been dead a while is usually a citation worth replacing rather than deleting."
|
|
739
|
+
},
|
|
719
740
|
"sitemap.dead": {
|
|
720
741
|
why: "A sitemap is a list of URLs you are asking to have crawled. Entries that 404 spend crawl budget on nothing and lower the trust placed in the rest of the file.",
|
|
721
742
|
fix: "Remove the URL from the sitemap, or restore the page. If it moved, redirect it and list the destination instead.",
|
|
@@ -1497,8 +1518,8 @@ function extractLinks(html) {
|
|
|
1497
1518
|
const hrefs = /* @__PURE__ */ new Set();
|
|
1498
1519
|
for (const anchor of root.querySelectorAll("a[href]")) {
|
|
1499
1520
|
const href = anchor.getAttribute("href")?.trim();
|
|
1500
|
-
if (!href) continue;
|
|
1501
|
-
if (
|
|
1521
|
+
if (!href || href.startsWith("#")) continue;
|
|
1522
|
+
if (/^[a-z][a-z0-9+.-]*:/i.test(href) && !/^https?:/i.test(href)) continue;
|
|
1502
1523
|
hrefs.add(href);
|
|
1503
1524
|
}
|
|
1504
1525
|
return [...hrefs];
|
|
@@ -1592,6 +1613,7 @@ async function walkHtml(dir, acc = []) {
|
|
|
1592
1613
|
}
|
|
1593
1614
|
return acc;
|
|
1594
1615
|
}
|
|
1616
|
+
var USER_AGENT = "pagetrace (+https://npmjs.com/package/pagetrace)";
|
|
1595
1617
|
var DEFAULT_TIMEOUT_MS = 15e3;
|
|
1596
1618
|
var MAX_ATTEMPTS = 3;
|
|
1597
1619
|
var RETRY_BASE_MS = 300;
|
|
@@ -1607,7 +1629,7 @@ async function fetchDoc(url, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
|
1607
1629
|
try {
|
|
1608
1630
|
response = await fetch(url, {
|
|
1609
1631
|
signal: AbortSignal.timeout(timeoutMs),
|
|
1610
|
-
headers: { "user-agent":
|
|
1632
|
+
headers: { "user-agent": USER_AGENT }
|
|
1611
1633
|
});
|
|
1612
1634
|
} catch (cause) {
|
|
1613
1635
|
lastError = new Error(`Could not reach ${url}: ${cause.message}`, { cause });
|
|
@@ -1652,7 +1674,9 @@ async function snapshotFromDir(dir, config = {}) {
|
|
|
1652
1674
|
if (robots !== null) site.robotsTxt = extractRobotsTxt(robots, agents);
|
|
1653
1675
|
const llms = await readFile(join(dir, "llms.txt"), "utf8").catch(() => null);
|
|
1654
1676
|
if (llms !== null) site.llmsTxt = extractLlmsTxt(llms);
|
|
1655
|
-
|
|
1677
|
+
const origin = site.origin ?? "https://pagetrace.invalid";
|
|
1678
|
+
await resolveBrokenLinks(pages, links, origin);
|
|
1679
|
+
if (config.checkExternal) await resolveDeadExternal(pages, links, origin, 5);
|
|
1656
1680
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1657
1681
|
}
|
|
1658
1682
|
var ASSET_PATH = /\.(?!html?$)[a-z0-9]+$/i;
|
|
@@ -1691,6 +1715,63 @@ async function resolveBrokenLinks(pages, links, origin, verify) {
|
|
|
1691
1715
|
if (confirmed.length > 0) pages[route].brokenLinks = confirmed.sort();
|
|
1692
1716
|
}
|
|
1693
1717
|
}
|
|
1718
|
+
function externalUrl(href, from, origin) {
|
|
1719
|
+
try {
|
|
1720
|
+
const url = new URL(href, `${origin}${from === "/" ? "" : from}/`);
|
|
1721
|
+
if (url.protocol !== "http:" && url.protocol !== "https:") return null;
|
|
1722
|
+
if (url.origin === origin) return null;
|
|
1723
|
+
url.hash = "";
|
|
1724
|
+
return url.href;
|
|
1725
|
+
} catch {
|
|
1726
|
+
return null;
|
|
1727
|
+
}
|
|
1728
|
+
}
|
|
1729
|
+
var MAX_EXTERNAL_CHECKS = 200;
|
|
1730
|
+
async function findDeadExternal(urls, concurrency, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
1731
|
+
const dead = /* @__PURE__ */ new Set();
|
|
1732
|
+
const queue = urls.slice(0, MAX_EXTERNAL_CHECKS);
|
|
1733
|
+
const request = (url, method) => fetch(url, {
|
|
1734
|
+
method,
|
|
1735
|
+
redirect: "follow",
|
|
1736
|
+
signal: AbortSignal.timeout(timeoutMs),
|
|
1737
|
+
headers: { "user-agent": USER_AGENT }
|
|
1738
|
+
});
|
|
1739
|
+
await Promise.all(
|
|
1740
|
+
Array.from({ length: Math.min(concurrency, queue.length) }, async () => {
|
|
1741
|
+
while (queue.length > 0) {
|
|
1742
|
+
const url = queue.shift();
|
|
1743
|
+
try {
|
|
1744
|
+
let response = await request(url, "HEAD");
|
|
1745
|
+
if (response.status === 405 || response.status === 501) {
|
|
1746
|
+
response = await request(url, "GET");
|
|
1747
|
+
}
|
|
1748
|
+
if (response.status === 404 || response.status === 410) dead.add(url);
|
|
1749
|
+
} catch {
|
|
1750
|
+
}
|
|
1751
|
+
}
|
|
1752
|
+
})
|
|
1753
|
+
);
|
|
1754
|
+
return dead;
|
|
1755
|
+
}
|
|
1756
|
+
async function resolveDeadExternal(pages, links, origin, concurrency, timeoutMs) {
|
|
1757
|
+
const perPage = /* @__PURE__ */ new Map();
|
|
1758
|
+
const unique = /* @__PURE__ */ new Set();
|
|
1759
|
+
for (const [route, hrefs] of Object.entries(links)) {
|
|
1760
|
+
const external = [
|
|
1761
|
+
...new Set(
|
|
1762
|
+
hrefs.map((href) => externalUrl(href, route, origin)).filter((url) => url !== null)
|
|
1763
|
+
)
|
|
1764
|
+
];
|
|
1765
|
+
if (external.length === 0) continue;
|
|
1766
|
+
perPage.set(route, external);
|
|
1767
|
+
for (const url of external) unique.add(url);
|
|
1768
|
+
}
|
|
1769
|
+
const dead = await findDeadExternal([...unique], concurrency, timeoutMs);
|
|
1770
|
+
for (const [route, external] of perPage) {
|
|
1771
|
+
const found = external.filter((url) => dead.has(url)).sort();
|
|
1772
|
+
if (found.length > 0) pages[route].deadExternal = found;
|
|
1773
|
+
}
|
|
1774
|
+
}
|
|
1694
1775
|
var MAX_NESTED_SITEMAPS = 50;
|
|
1695
1776
|
async function snapshotFromOrigin(origin, options = {}) {
|
|
1696
1777
|
const base = new URL(origin);
|
|
@@ -1779,6 +1860,9 @@ async function snapshotFromOrigin(origin, options = {}) {
|
|
|
1779
1860
|
return false;
|
|
1780
1861
|
}
|
|
1781
1862
|
});
|
|
1863
|
+
if (options.checkExternal) {
|
|
1864
|
+
await resolveDeadExternal(pages, links, base.origin, concurrency, timeout);
|
|
1865
|
+
}
|
|
1782
1866
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1783
1867
|
}
|
|
1784
1868
|
|
|
@@ -1831,13 +1915,15 @@ async function loadConfig(path = DEFAULT_CONFIG) {
|
|
|
1831
1915
|
}
|
|
1832
1916
|
}
|
|
1833
1917
|
async function build(flags, config) {
|
|
1834
|
-
|
|
1918
|
+
const checkExternal = flags.external ?? config.checkExternal;
|
|
1919
|
+
if (flags.dir) return snapshotFromDir(flags.dir, { ...config, checkExternal });
|
|
1835
1920
|
if (flags.url)
|
|
1836
1921
|
return snapshotFromOrigin(flags.url, {
|
|
1837
1922
|
...config,
|
|
1838
1923
|
limit: flags.limit,
|
|
1839
1924
|
concurrency: flags.concurrency,
|
|
1840
|
-
ignoreRobots: flags.ignoreRobots ?? config.ignoreRobots
|
|
1925
|
+
ignoreRobots: flags.ignoreRobots ?? config.ignoreRobots,
|
|
1926
|
+
checkExternal
|
|
1841
1927
|
});
|
|
1842
1928
|
throw new Error("Provide a source: --dir <build directory> or --url <origin>.");
|
|
1843
1929
|
}
|
|
@@ -1856,7 +1942,7 @@ function render(findings, format) {
|
|
|
1856
1942
|
}
|
|
1857
1943
|
}
|
|
1858
1944
|
var cli = cac("pagetrace");
|
|
1859
|
-
cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
1945
|
+
cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
1860
1946
|
const config = await loadConfig(flags.config);
|
|
1861
1947
|
const snapshot = await build(flags, config);
|
|
1862
1948
|
const written = await writeLockfile(flags.out, snapshot);
|
|
@@ -1865,7 +1951,7 @@ cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").opti
|
|
|
1865
1951
|
written ? pc2.green(`Wrote ${flags.out} \u2014 ${count} page${count === 1 ? "" : "s"}.`) : pc2.dim(`${flags.out} is already up to date \u2014 ${count} page${count === 1 ? "" : "s"}.`)
|
|
1866
1952
|
);
|
|
1867
1953
|
});
|
|
1868
|
-
cli.command("check", "Compare the current surface against the lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--lockfile <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | github | sarif", { default: "pretty" }).option("--fail-on <severity>", "error | warn | info", { default: "error" }).option("--audit", "Also run absolute rules, not just the diff", { default: true }).option("--update", "Write the new state to the lockfile after reporting").option("--baseline-branch <ref>", "Read the baseline lockfile from a git ref instead of disk").action(async (flags) => {
|
|
1954
|
+
cli.command("check", "Compare the current surface against the lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--lockfile <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | github | sarif", { default: "pretty" }).option("--fail-on <severity>", "error | warn | info", { default: "error" }).option("--audit", "Also run absolute rules, not just the diff", { default: true }).option("--update", "Write the new state to the lockfile after reporting").option("--baseline-branch <ref>", "Read the baseline lockfile from a git ref instead of disk").action(async (flags) => {
|
|
1869
1955
|
const failOn = parseFailOn(flags.failOn, false);
|
|
1870
1956
|
const config = await loadConfig(flags.config);
|
|
1871
1957
|
const next = await build(flags, config);
|
|
@@ -1895,7 +1981,7 @@ Failing: ${summary.error} error, ${summary.warn} warning (--fail-on ${failOn}).`
|
|
|
1895
1981
|
process.exitCode = EXIT_FINDINGS;
|
|
1896
1982
|
}
|
|
1897
1983
|
});
|
|
1898
|
-
cli.command("audit", "Audit a site as it stands, with explanations and fixes").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | html", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "never" }).action(async (flags) => {
|
|
1984
|
+
cli.command("audit", "Audit a site as it stands, with explanations and fixes").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | html", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "never" }).action(async (flags) => {
|
|
1899
1985
|
const failOn = parseFailOn(flags.failOn, true);
|
|
1900
1986
|
const config = await loadConfig(flags.config);
|
|
1901
1987
|
const snapshot = await build(flags, config);
|
|
@@ -1935,7 +2021,54 @@ cli.command("audit", "Audit a site as it stands, with explanations and fixes").o
|
|
|
1935
2021
|
process.exitCode = EXIT_FINDINGS;
|
|
1936
2022
|
}
|
|
1937
2023
|
});
|
|
1938
|
-
cli.command("
|
|
2024
|
+
cli.command("links", "Find internal links that point at no page").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "error" }).action(async (flags) => {
|
|
2025
|
+
const failOn = parseFailOn(flags.failOn, true);
|
|
2026
|
+
const config = await loadConfig(flags.config);
|
|
2027
|
+
const snapshot = await build(flags, config);
|
|
2028
|
+
const pages = Object.values(snapshot.pages);
|
|
2029
|
+
if (pages.length === 0) {
|
|
2030
|
+
console.error(
|
|
2031
|
+
pc2.yellow(
|
|
2032
|
+
"No pages found. Check that the sitemap is reachable, or pass --dir with pre-rendered HTML."
|
|
2033
|
+
)
|
|
2034
|
+
);
|
|
2035
|
+
process.exitCode = EXIT_FAILURE;
|
|
2036
|
+
return;
|
|
2037
|
+
}
|
|
2038
|
+
const platform = detectPlatform(
|
|
2039
|
+
pages.map((p) => p.generator),
|
|
2040
|
+
pages.flatMap((p) => Object.values(p.og))
|
|
2041
|
+
);
|
|
2042
|
+
const findings = applyConfig(auditSnapshot(snapshot, config), config).filter((f) => f.code.startsWith("link.")).map((f) => withGuidance(f, platform));
|
|
2043
|
+
const target = flags.url ?? flags.dir ?? "site";
|
|
2044
|
+
if (findings.length === 0 && !flags.out) {
|
|
2045
|
+
console.log(
|
|
2046
|
+
pc2.green(
|
|
2047
|
+
`No broken links found \u2014 ${pages.length} page${pages.length === 1 ? "" : "s"} checked.`
|
|
2048
|
+
)
|
|
2049
|
+
);
|
|
2050
|
+
return;
|
|
2051
|
+
}
|
|
2052
|
+
const groups = aggregate(findings);
|
|
2053
|
+
const meta = {
|
|
2054
|
+
target,
|
|
2055
|
+
platform,
|
|
2056
|
+
pageCount: pages.length,
|
|
2057
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString().slice(0, 10)
|
|
2058
|
+
};
|
|
2059
|
+
const output = flags.format === "json" ? JSON.stringify({ schemaVersion: 1, meta, summary: summarize(findings), groups }, null, 2) : flags.format === "markdown" ? formatAuditMarkdown(groups, meta) : formatAuditPretty(groups, meta, process.stdout.columns);
|
|
2060
|
+
if (flags.out) {
|
|
2061
|
+
await writeFile(flags.out, `${output}
|
|
2062
|
+
`, "utf8");
|
|
2063
|
+
console.log(pc2.green(`Wrote ${flags.out} \u2014 ${groups.length} broken link${groups.length === 1 ? "" : "s"}.`));
|
|
2064
|
+
} else {
|
|
2065
|
+
console.log(output);
|
|
2066
|
+
}
|
|
2067
|
+
if (failOn !== "never" && shouldFail(findings, failOn)) {
|
|
2068
|
+
process.exitCode = EXIT_FINDINGS;
|
|
2069
|
+
}
|
|
2070
|
+
});
|
|
2071
|
+
cli.command("init", "Write a config file and take the first snapshot").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
1939
2072
|
const existing = await readFile2(flags.config, "utf8").catch(() => null);
|
|
1940
2073
|
if (existing === null) {
|
|
1941
2074
|
const config = flags.url ? { siteUrl: new URL(flags.url).origin } : {};
|
|
@@ -1958,11 +2091,11 @@ Commit both files, then run \`pagetrace check ${flags.dir ? `--dir ${flags.dir}`
|
|
|
1958
2091
|
});
|
|
1959
2092
|
cli.command("update", "Check npm for a newer pagetrace and install it").option("--check", "Only report whether an update exists").action(async (flags) => {
|
|
1960
2093
|
const latest = await latestVersion("pagetrace");
|
|
1961
|
-
if (!isNewer(latest, "0.
|
|
1962
|
-
console.log(pc2.green(`pagetrace ${"0.
|
|
2094
|
+
if (!isNewer(latest, "0.12.0")) {
|
|
2095
|
+
console.log(pc2.green(`pagetrace ${"0.12.0"} is the latest version.`));
|
|
1963
2096
|
return;
|
|
1964
2097
|
}
|
|
1965
|
-
console.log(pc2.yellow(`Update available: ${"0.
|
|
2098
|
+
console.log(pc2.yellow(`Update available: ${"0.12.0"} \u2192 ${latest}`));
|
|
1966
2099
|
if (flags.check) return;
|
|
1967
2100
|
if (process.argv[1]?.startsWith(process.cwd())) {
|
|
1968
2101
|
console.log(
|
|
@@ -1980,7 +2113,7 @@ cli.command("update", "Check npm for a newer pagetrace and install it").option("
|
|
|
1980
2113
|
console.log(pc2.green(`Updated to pagetrace ${latest}.`));
|
|
1981
2114
|
});
|
|
1982
2115
|
cli.help();
|
|
1983
|
-
cli.version("0.
|
|
2116
|
+
cli.version("0.12.0");
|
|
1984
2117
|
async function main() {
|
|
1985
2118
|
try {
|
|
1986
2119
|
cli.parse(process.argv, { run: false });
|
package/dist/index.cjs
CHANGED
|
@@ -297,13 +297,24 @@ function auditPage(page, config = {}) {
|
|
|
297
297
|
return findings;
|
|
298
298
|
}
|
|
299
299
|
function auditLinks(page) {
|
|
300
|
-
return
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
300
|
+
return [
|
|
301
|
+
...(page.brokenLinks ?? []).map((target) => ({
|
|
302
|
+
code: "link.broken",
|
|
303
|
+
severity: "error",
|
|
304
|
+
route: page.route,
|
|
305
|
+
message: `Links to ${target}, which does not exist.`,
|
|
306
|
+
after: target
|
|
307
|
+
})),
|
|
308
|
+
// A warning, not an error: the page it points at belongs to somebody else,
|
|
309
|
+
// who can delete it on a Tuesday without asking you.
|
|
310
|
+
...(page.deadExternal ?? []).map((target) => ({
|
|
311
|
+
code: "link.external.dead",
|
|
312
|
+
severity: "warn",
|
|
313
|
+
route: page.route,
|
|
314
|
+
message: `Links out to ${target}, which answers 404.`,
|
|
315
|
+
after: target
|
|
316
|
+
}))
|
|
317
|
+
];
|
|
307
318
|
}
|
|
308
319
|
function auditSite(snapshot) {
|
|
309
320
|
const findings = [];
|
|
@@ -674,6 +685,12 @@ function diffPage(before, after) {
|
|
|
674
685
|
push("link.broken.removed", "info", `Link to ${target} was fixed or removed.`, {
|
|
675
686
|
before: target
|
|
676
687
|
});
|
|
688
|
+
const deadBefore = before.deadExternal ?? [];
|
|
689
|
+
const deadAfter = after.deadExternal ?? [];
|
|
690
|
+
for (const target of deadAfter.filter((t) => !deadBefore.includes(t)))
|
|
691
|
+
push("link.external.dead.added", "warn", `Links out to ${target}, which answers 404.`, {
|
|
692
|
+
after: target
|
|
693
|
+
});
|
|
677
694
|
const droppedHreflang = Object.keys(before.hreflang).filter((k) => !(k in after.hreflang));
|
|
678
695
|
if (droppedHreflang.length > 0)
|
|
679
696
|
push("hreflang.removed", "warn", "hreflang alternates were removed.", {
|
|
@@ -962,8 +979,8 @@ function extractLinks(html) {
|
|
|
962
979
|
const hrefs = /* @__PURE__ */ new Set();
|
|
963
980
|
for (const anchor of root.querySelectorAll("a[href]")) {
|
|
964
981
|
const href = anchor.getAttribute("href")?.trim();
|
|
965
|
-
if (!href) continue;
|
|
966
|
-
if (
|
|
982
|
+
if (!href || href.startsWith("#")) continue;
|
|
983
|
+
if (/^[a-z][a-z0-9+.-]*:/i.test(href) && !/^https?:/i.test(href)) continue;
|
|
967
984
|
hrefs.add(href);
|
|
968
985
|
}
|
|
969
986
|
return [...hrefs];
|
|
@@ -1021,6 +1038,10 @@ var GUIDANCE = {
|
|
|
1021
1038
|
nextjs: "A `<Link href>` pointing at a route that no longer exists. TypeScript will not catch it \u2014 typed routes are opt-in via `experimental.typedRoutes`."
|
|
1022
1039
|
}
|
|
1023
1040
|
},
|
|
1041
|
+
"link.external.dead": {
|
|
1042
|
+
why: "An outbound link to a page that is gone sends readers to a 404 and spends the trust the link was passing on nothing. Unlike an internal link, the page is not yours to restore.",
|
|
1043
|
+
fix: "Point the link at the current URL, at an archived copy, or remove it. A link that has been dead a while is usually a citation worth replacing rather than deleting."
|
|
1044
|
+
},
|
|
1024
1045
|
"sitemap.dead": {
|
|
1025
1046
|
why: "A sitemap is a list of URLs you are asking to have crawled. Entries that 404 spend crawl budget on nothing and lower the trust placed in the rest of the file.",
|
|
1026
1047
|
fix: "Remove the URL from the sitemap, or restore the page. If it moved, redirect it and list the destination instead.",
|
|
@@ -1656,6 +1677,7 @@ async function walkHtml(dir, acc = []) {
|
|
|
1656
1677
|
}
|
|
1657
1678
|
return acc;
|
|
1658
1679
|
}
|
|
1680
|
+
var USER_AGENT = "pagetrace (+https://npmjs.com/package/pagetrace)";
|
|
1659
1681
|
var DEFAULT_TIMEOUT_MS = 15e3;
|
|
1660
1682
|
var MAX_ATTEMPTS = 3;
|
|
1661
1683
|
var RETRY_BASE_MS = 300;
|
|
@@ -1671,7 +1693,7 @@ async function fetchDoc(url, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
|
1671
1693
|
try {
|
|
1672
1694
|
response = await fetch(url, {
|
|
1673
1695
|
signal: AbortSignal.timeout(timeoutMs),
|
|
1674
|
-
headers: { "user-agent":
|
|
1696
|
+
headers: { "user-agent": USER_AGENT }
|
|
1675
1697
|
});
|
|
1676
1698
|
} catch (cause) {
|
|
1677
1699
|
lastError = new Error(`Could not reach ${url}: ${cause.message}`, { cause });
|
|
@@ -1716,7 +1738,9 @@ async function snapshotFromDir(dir, config = {}) {
|
|
|
1716
1738
|
if (robots !== null) site.robotsTxt = extractRobotsTxt(robots, agents);
|
|
1717
1739
|
const llms = await (0, import_promises.readFile)((0, import_node_path.join)(dir, "llms.txt"), "utf8").catch(() => null);
|
|
1718
1740
|
if (llms !== null) site.llmsTxt = extractLlmsTxt(llms);
|
|
1719
|
-
|
|
1741
|
+
const origin = site.origin ?? "https://pagetrace.invalid";
|
|
1742
|
+
await resolveBrokenLinks(pages, links, origin);
|
|
1743
|
+
if (config.checkExternal) await resolveDeadExternal(pages, links, origin, 5);
|
|
1720
1744
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1721
1745
|
}
|
|
1722
1746
|
var ASSET_PATH = /\.(?!html?$)[a-z0-9]+$/i;
|
|
@@ -1755,6 +1779,63 @@ async function resolveBrokenLinks(pages, links, origin, verify) {
|
|
|
1755
1779
|
if (confirmed.length > 0) pages[route].brokenLinks = confirmed.sort();
|
|
1756
1780
|
}
|
|
1757
1781
|
}
|
|
1782
|
+
function externalUrl(href, from, origin) {
|
|
1783
|
+
try {
|
|
1784
|
+
const url = new URL(href, `${origin}${from === "/" ? "" : from}/`);
|
|
1785
|
+
if (url.protocol !== "http:" && url.protocol !== "https:") return null;
|
|
1786
|
+
if (url.origin === origin) return null;
|
|
1787
|
+
url.hash = "";
|
|
1788
|
+
return url.href;
|
|
1789
|
+
} catch {
|
|
1790
|
+
return null;
|
|
1791
|
+
}
|
|
1792
|
+
}
|
|
1793
|
+
var MAX_EXTERNAL_CHECKS = 200;
|
|
1794
|
+
async function findDeadExternal(urls, concurrency, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
1795
|
+
const dead = /* @__PURE__ */ new Set();
|
|
1796
|
+
const queue = urls.slice(0, MAX_EXTERNAL_CHECKS);
|
|
1797
|
+
const request = (url, method) => fetch(url, {
|
|
1798
|
+
method,
|
|
1799
|
+
redirect: "follow",
|
|
1800
|
+
signal: AbortSignal.timeout(timeoutMs),
|
|
1801
|
+
headers: { "user-agent": USER_AGENT }
|
|
1802
|
+
});
|
|
1803
|
+
await Promise.all(
|
|
1804
|
+
Array.from({ length: Math.min(concurrency, queue.length) }, async () => {
|
|
1805
|
+
while (queue.length > 0) {
|
|
1806
|
+
const url = queue.shift();
|
|
1807
|
+
try {
|
|
1808
|
+
let response = await request(url, "HEAD");
|
|
1809
|
+
if (response.status === 405 || response.status === 501) {
|
|
1810
|
+
response = await request(url, "GET");
|
|
1811
|
+
}
|
|
1812
|
+
if (response.status === 404 || response.status === 410) dead.add(url);
|
|
1813
|
+
} catch {
|
|
1814
|
+
}
|
|
1815
|
+
}
|
|
1816
|
+
})
|
|
1817
|
+
);
|
|
1818
|
+
return dead;
|
|
1819
|
+
}
|
|
1820
|
+
async function resolveDeadExternal(pages, links, origin, concurrency, timeoutMs) {
|
|
1821
|
+
const perPage = /* @__PURE__ */ new Map();
|
|
1822
|
+
const unique = /* @__PURE__ */ new Set();
|
|
1823
|
+
for (const [route, hrefs] of Object.entries(links)) {
|
|
1824
|
+
const external = [
|
|
1825
|
+
...new Set(
|
|
1826
|
+
hrefs.map((href) => externalUrl(href, route, origin)).filter((url) => url !== null)
|
|
1827
|
+
)
|
|
1828
|
+
];
|
|
1829
|
+
if (external.length === 0) continue;
|
|
1830
|
+
perPage.set(route, external);
|
|
1831
|
+
for (const url of external) unique.add(url);
|
|
1832
|
+
}
|
|
1833
|
+
const dead = await findDeadExternal([...unique], concurrency, timeoutMs);
|
|
1834
|
+
for (const [route, external] of perPage) {
|
|
1835
|
+
const found = external.filter((url) => dead.has(url)).sort();
|
|
1836
|
+
if (found.length > 0) pages[route].deadExternal = found;
|
|
1837
|
+
}
|
|
1838
|
+
}
|
|
1758
1839
|
var MAX_NESTED_SITEMAPS = 50;
|
|
1759
1840
|
async function snapshotFromOrigin(origin, options = {}) {
|
|
1760
1841
|
const base = new URL(origin);
|
|
@@ -1843,6 +1924,9 @@ async function snapshotFromOrigin(origin, options = {}) {
|
|
|
1843
1924
|
return false;
|
|
1844
1925
|
}
|
|
1845
1926
|
});
|
|
1927
|
+
if (options.checkExternal) {
|
|
1928
|
+
await resolveDeadExternal(pages, links, base.origin, concurrency, timeout);
|
|
1929
|
+
}
|
|
1846
1930
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1847
1931
|
}
|
|
1848
1932
|
// Annotate the CommonJS export names for ESM import in node:
|
package/dist/index.d.cts
CHANGED
|
@@ -50,6 +50,12 @@ interface PageFingerprint {
|
|
|
50
50
|
* property the lockfile has to have.
|
|
51
51
|
*/
|
|
52
52
|
brokenLinks?: string[];
|
|
53
|
+
/**
|
|
54
|
+
* External links that answered 404 or 410. Only present when the crawl was
|
|
55
|
+
* asked to check them: it means a request per unique external URL, to hosts
|
|
56
|
+
* you do not control.
|
|
57
|
+
*/
|
|
58
|
+
deadExternal?: string[];
|
|
53
59
|
}
|
|
54
60
|
/** Site-wide signals that live outside any single page. */
|
|
55
61
|
interface SiteFingerprint {
|
|
@@ -139,6 +145,12 @@ interface Config {
|
|
|
139
145
|
* worse than crawling a site you already own.
|
|
140
146
|
*/
|
|
141
147
|
ignoreRobots?: boolean;
|
|
148
|
+
/**
|
|
149
|
+
* Also check links that leave the site. Off by default: it fans out to hosts
|
|
150
|
+
* you do not control, and a lockfile that records their availability turns a
|
|
151
|
+
* third party's bad afternoon into a diff in your repository.
|
|
152
|
+
*/
|
|
153
|
+
checkExternal?: boolean;
|
|
142
154
|
}
|
|
143
155
|
|
|
144
156
|
/**
|
|
@@ -186,12 +198,12 @@ declare function isCrawlable(path: string, rules: {
|
|
|
186
198
|
allow?: string[];
|
|
187
199
|
} | null | undefined): boolean;
|
|
188
200
|
/**
|
|
189
|
-
*
|
|
201
|
+
* Every href on the page worth resolving, deduplicated, in document order.
|
|
202
|
+
* Both internal and external: snapshot.ts is the only layer that knows the
|
|
203
|
+
* page's own URL, so classification happens there.
|
|
190
204
|
*
|
|
191
|
-
*
|
|
192
|
-
*
|
|
193
|
-
* else's problem: an external link checker fans out to hosts you do not
|
|
194
|
-
* control, where a Cloudflare 403 and a rate limit both look like a dead page.
|
|
205
|
+
* Dropped here because they can never be dead: fragments, mailto:, tel:,
|
|
206
|
+
* javascript: and other non-http schemes.
|
|
195
207
|
*/
|
|
196
208
|
declare function extractLinks(html: string): string[];
|
|
197
209
|
/** Parse llms.txt, capturing section headings so truncation is detectable. */
|
package/dist/index.d.ts
CHANGED
|
@@ -50,6 +50,12 @@ interface PageFingerprint {
|
|
|
50
50
|
* property the lockfile has to have.
|
|
51
51
|
*/
|
|
52
52
|
brokenLinks?: string[];
|
|
53
|
+
/**
|
|
54
|
+
* External links that answered 404 or 410. Only present when the crawl was
|
|
55
|
+
* asked to check them: it means a request per unique external URL, to hosts
|
|
56
|
+
* you do not control.
|
|
57
|
+
*/
|
|
58
|
+
deadExternal?: string[];
|
|
53
59
|
}
|
|
54
60
|
/** Site-wide signals that live outside any single page. */
|
|
55
61
|
interface SiteFingerprint {
|
|
@@ -139,6 +145,12 @@ interface Config {
|
|
|
139
145
|
* worse than crawling a site you already own.
|
|
140
146
|
*/
|
|
141
147
|
ignoreRobots?: boolean;
|
|
148
|
+
/**
|
|
149
|
+
* Also check links that leave the site. Off by default: it fans out to hosts
|
|
150
|
+
* you do not control, and a lockfile that records their availability turns a
|
|
151
|
+
* third party's bad afternoon into a diff in your repository.
|
|
152
|
+
*/
|
|
153
|
+
checkExternal?: boolean;
|
|
142
154
|
}
|
|
143
155
|
|
|
144
156
|
/**
|
|
@@ -186,12 +198,12 @@ declare function isCrawlable(path: string, rules: {
|
|
|
186
198
|
allow?: string[];
|
|
187
199
|
} | null | undefined): boolean;
|
|
188
200
|
/**
|
|
189
|
-
*
|
|
201
|
+
* Every href on the page worth resolving, deduplicated, in document order.
|
|
202
|
+
* Both internal and external: snapshot.ts is the only layer that knows the
|
|
203
|
+
* page's own URL, so classification happens there.
|
|
190
204
|
*
|
|
191
|
-
*
|
|
192
|
-
*
|
|
193
|
-
* else's problem: an external link checker fans out to hosts you do not
|
|
194
|
-
* control, where a Cloudflare 403 and a rate limit both look like a dead page.
|
|
205
|
+
* Dropped here because they can never be dead: fragments, mailto:, tel:,
|
|
206
|
+
* javascript: and other non-http schemes.
|
|
195
207
|
*/
|
|
196
208
|
declare function extractLinks(html: string): string[];
|
|
197
209
|
/** Parse llms.txt, capturing section headings so truncation is detectable. */
|
package/dist/index.js
CHANGED
|
@@ -223,13 +223,24 @@ function auditPage(page, config = {}) {
|
|
|
223
223
|
return findings;
|
|
224
224
|
}
|
|
225
225
|
function auditLinks(page) {
|
|
226
|
-
return
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
226
|
+
return [
|
|
227
|
+
...(page.brokenLinks ?? []).map((target) => ({
|
|
228
|
+
code: "link.broken",
|
|
229
|
+
severity: "error",
|
|
230
|
+
route: page.route,
|
|
231
|
+
message: `Links to ${target}, which does not exist.`,
|
|
232
|
+
after: target
|
|
233
|
+
})),
|
|
234
|
+
// A warning, not an error: the page it points at belongs to somebody else,
|
|
235
|
+
// who can delete it on a Tuesday without asking you.
|
|
236
|
+
...(page.deadExternal ?? []).map((target) => ({
|
|
237
|
+
code: "link.external.dead",
|
|
238
|
+
severity: "warn",
|
|
239
|
+
route: page.route,
|
|
240
|
+
message: `Links out to ${target}, which answers 404.`,
|
|
241
|
+
after: target
|
|
242
|
+
}))
|
|
243
|
+
];
|
|
233
244
|
}
|
|
234
245
|
function auditSite(snapshot) {
|
|
235
246
|
const findings = [];
|
|
@@ -600,6 +611,12 @@ function diffPage(before, after) {
|
|
|
600
611
|
push("link.broken.removed", "info", `Link to ${target} was fixed or removed.`, {
|
|
601
612
|
before: target
|
|
602
613
|
});
|
|
614
|
+
const deadBefore = before.deadExternal ?? [];
|
|
615
|
+
const deadAfter = after.deadExternal ?? [];
|
|
616
|
+
for (const target of deadAfter.filter((t) => !deadBefore.includes(t)))
|
|
617
|
+
push("link.external.dead.added", "warn", `Links out to ${target}, which answers 404.`, {
|
|
618
|
+
after: target
|
|
619
|
+
});
|
|
603
620
|
const droppedHreflang = Object.keys(before.hreflang).filter((k) => !(k in after.hreflang));
|
|
604
621
|
if (droppedHreflang.length > 0)
|
|
605
622
|
push("hreflang.removed", "warn", "hreflang alternates were removed.", {
|
|
@@ -888,8 +905,8 @@ function extractLinks(html) {
|
|
|
888
905
|
const hrefs = /* @__PURE__ */ new Set();
|
|
889
906
|
for (const anchor of root.querySelectorAll("a[href]")) {
|
|
890
907
|
const href = anchor.getAttribute("href")?.trim();
|
|
891
|
-
if (!href) continue;
|
|
892
|
-
if (
|
|
908
|
+
if (!href || href.startsWith("#")) continue;
|
|
909
|
+
if (/^[a-z][a-z0-9+.-]*:/i.test(href) && !/^https?:/i.test(href)) continue;
|
|
893
910
|
hrefs.add(href);
|
|
894
911
|
}
|
|
895
912
|
return [...hrefs];
|
|
@@ -947,6 +964,10 @@ var GUIDANCE = {
|
|
|
947
964
|
nextjs: "A `<Link href>` pointing at a route that no longer exists. TypeScript will not catch it \u2014 typed routes are opt-in via `experimental.typedRoutes`."
|
|
948
965
|
}
|
|
949
966
|
},
|
|
967
|
+
"link.external.dead": {
|
|
968
|
+
why: "An outbound link to a page that is gone sends readers to a 404 and spends the trust the link was passing on nothing. Unlike an internal link, the page is not yours to restore.",
|
|
969
|
+
fix: "Point the link at the current URL, at an archived copy, or remove it. A link that has been dead a while is usually a citation worth replacing rather than deleting."
|
|
970
|
+
},
|
|
950
971
|
"sitemap.dead": {
|
|
951
972
|
why: "A sitemap is a list of URLs you are asking to have crawled. Entries that 404 spend crawl budget on nothing and lower the trust placed in the rest of the file.",
|
|
952
973
|
fix: "Remove the URL from the sitemap, or restore the page. If it moved, redirect it and list the destination instead.",
|
|
@@ -1582,6 +1603,7 @@ async function walkHtml(dir, acc = []) {
|
|
|
1582
1603
|
}
|
|
1583
1604
|
return acc;
|
|
1584
1605
|
}
|
|
1606
|
+
var USER_AGENT = "pagetrace (+https://npmjs.com/package/pagetrace)";
|
|
1585
1607
|
var DEFAULT_TIMEOUT_MS = 15e3;
|
|
1586
1608
|
var MAX_ATTEMPTS = 3;
|
|
1587
1609
|
var RETRY_BASE_MS = 300;
|
|
@@ -1597,7 +1619,7 @@ async function fetchDoc(url, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
|
1597
1619
|
try {
|
|
1598
1620
|
response = await fetch(url, {
|
|
1599
1621
|
signal: AbortSignal.timeout(timeoutMs),
|
|
1600
|
-
headers: { "user-agent":
|
|
1622
|
+
headers: { "user-agent": USER_AGENT }
|
|
1601
1623
|
});
|
|
1602
1624
|
} catch (cause) {
|
|
1603
1625
|
lastError = new Error(`Could not reach ${url}: ${cause.message}`, { cause });
|
|
@@ -1642,7 +1664,9 @@ async function snapshotFromDir(dir, config = {}) {
|
|
|
1642
1664
|
if (robots !== null) site.robotsTxt = extractRobotsTxt(robots, agents);
|
|
1643
1665
|
const llms = await readFile(join(dir, "llms.txt"), "utf8").catch(() => null);
|
|
1644
1666
|
if (llms !== null) site.llmsTxt = extractLlmsTxt(llms);
|
|
1645
|
-
|
|
1667
|
+
const origin = site.origin ?? "https://pagetrace.invalid";
|
|
1668
|
+
await resolveBrokenLinks(pages, links, origin);
|
|
1669
|
+
if (config.checkExternal) await resolveDeadExternal(pages, links, origin, 5);
|
|
1646
1670
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1647
1671
|
}
|
|
1648
1672
|
var ASSET_PATH = /\.(?!html?$)[a-z0-9]+$/i;
|
|
@@ -1681,6 +1705,63 @@ async function resolveBrokenLinks(pages, links, origin, verify) {
|
|
|
1681
1705
|
if (confirmed.length > 0) pages[route].brokenLinks = confirmed.sort();
|
|
1682
1706
|
}
|
|
1683
1707
|
}
|
|
1708
|
+
function externalUrl(href, from, origin) {
|
|
1709
|
+
try {
|
|
1710
|
+
const url = new URL(href, `${origin}${from === "/" ? "" : from}/`);
|
|
1711
|
+
if (url.protocol !== "http:" && url.protocol !== "https:") return null;
|
|
1712
|
+
if (url.origin === origin) return null;
|
|
1713
|
+
url.hash = "";
|
|
1714
|
+
return url.href;
|
|
1715
|
+
} catch {
|
|
1716
|
+
return null;
|
|
1717
|
+
}
|
|
1718
|
+
}
|
|
1719
|
+
var MAX_EXTERNAL_CHECKS = 200;
|
|
1720
|
+
async function findDeadExternal(urls, concurrency, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
1721
|
+
const dead = /* @__PURE__ */ new Set();
|
|
1722
|
+
const queue = urls.slice(0, MAX_EXTERNAL_CHECKS);
|
|
1723
|
+
const request = (url, method) => fetch(url, {
|
|
1724
|
+
method,
|
|
1725
|
+
redirect: "follow",
|
|
1726
|
+
signal: AbortSignal.timeout(timeoutMs),
|
|
1727
|
+
headers: { "user-agent": USER_AGENT }
|
|
1728
|
+
});
|
|
1729
|
+
await Promise.all(
|
|
1730
|
+
Array.from({ length: Math.min(concurrency, queue.length) }, async () => {
|
|
1731
|
+
while (queue.length > 0) {
|
|
1732
|
+
const url = queue.shift();
|
|
1733
|
+
try {
|
|
1734
|
+
let response = await request(url, "HEAD");
|
|
1735
|
+
if (response.status === 405 || response.status === 501) {
|
|
1736
|
+
response = await request(url, "GET");
|
|
1737
|
+
}
|
|
1738
|
+
if (response.status === 404 || response.status === 410) dead.add(url);
|
|
1739
|
+
} catch {
|
|
1740
|
+
}
|
|
1741
|
+
}
|
|
1742
|
+
})
|
|
1743
|
+
);
|
|
1744
|
+
return dead;
|
|
1745
|
+
}
|
|
1746
|
+
async function resolveDeadExternal(pages, links, origin, concurrency, timeoutMs) {
|
|
1747
|
+
const perPage = /* @__PURE__ */ new Map();
|
|
1748
|
+
const unique = /* @__PURE__ */ new Set();
|
|
1749
|
+
for (const [route, hrefs] of Object.entries(links)) {
|
|
1750
|
+
const external = [
|
|
1751
|
+
...new Set(
|
|
1752
|
+
hrefs.map((href) => externalUrl(href, route, origin)).filter((url) => url !== null)
|
|
1753
|
+
)
|
|
1754
|
+
];
|
|
1755
|
+
if (external.length === 0) continue;
|
|
1756
|
+
perPage.set(route, external);
|
|
1757
|
+
for (const url of external) unique.add(url);
|
|
1758
|
+
}
|
|
1759
|
+
const dead = await findDeadExternal([...unique], concurrency, timeoutMs);
|
|
1760
|
+
for (const [route, external] of perPage) {
|
|
1761
|
+
const found = external.filter((url) => dead.has(url)).sort();
|
|
1762
|
+
if (found.length > 0) pages[route].deadExternal = found;
|
|
1763
|
+
}
|
|
1764
|
+
}
|
|
1684
1765
|
var MAX_NESTED_SITEMAPS = 50;
|
|
1685
1766
|
async function snapshotFromOrigin(origin, options = {}) {
|
|
1686
1767
|
const base = new URL(origin);
|
|
@@ -1769,6 +1850,9 @@ async function snapshotFromOrigin(origin, options = {}) {
|
|
|
1769
1850
|
return false;
|
|
1770
1851
|
}
|
|
1771
1852
|
});
|
|
1853
|
+
if (options.checkExternal) {
|
|
1854
|
+
await resolveDeadExternal(pages, links, base.origin, concurrency, timeout);
|
|
1855
|
+
}
|
|
1772
1856
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1773
1857
|
}
|
|
1774
1858
|
export {
|