pagetrace 0.11.0 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +34 -0
- package/README.md +20 -1
- package/dist/cli.cjs +173 -22
- package/dist/cli.js +173 -22
- package/dist/index.cjs +126 -11
- package/dist/index.d.cts +26 -6
- package/dist/index.d.ts +26 -6
- package/dist/index.js +125 -11
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,38 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
|
6
6
|
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
7
|
While the version is below 1.0.0, breaking changes ship in a minor release.
|
|
8
8
|
|
|
9
|
+
## [0.13.0] - 2026-09-08
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- `pagetrace page <url>` checks a single page: every link on it confirmed with a
|
|
14
|
+
real request, external ones included by default, plus that page's own rules.
|
|
15
|
+
A crawl trusts its route set and only verifies what is missing from it; with
|
|
16
|
+
one page there is no route set to trust, so everything is checked. Passing a
|
|
17
|
+
deep URL to `--url` had crawled the whole site from its origin and ignored the
|
|
18
|
+
path, which is not what it looked like it did.
|
|
19
|
+
|
|
20
|
+
## [0.12.0] - 2026-09-08
|
|
21
|
+
|
|
22
|
+
### Added
|
|
23
|
+
|
|
24
|
+
- `--external` checks links that leave the site, on `links`, `audit`, `check`
|
|
25
|
+
and `snapshot`. Only 404 and 410 count as dead: a 403 from a bot wall, a 429,
|
|
26
|
+
a timeout and a TLS failure all describe the request rather than the page.
|
|
27
|
+
`HEAD` first with a `GET` fallback, one request per unique URL across the
|
|
28
|
+
site, capped at 200. Off by default — it fans out to hosts you do not control,
|
|
29
|
+
and recording their availability in a lockfile turns a third party's bad
|
|
30
|
+
afternoon into a diff in your repository.
|
|
31
|
+
- `link.external.dead` (warn) and `link.external.dead.added`. A warning rather
|
|
32
|
+
than an error: the page belongs to someone who can delete it without asking.
|
|
33
|
+
|
|
34
|
+
### Fixed
|
|
35
|
+
|
|
36
|
+
- Absolute internal links were invisible. `extractLinks` dropped every href with
|
|
37
|
+
a scheme, so a site writing `https://example.com/about` rather than `/about`
|
|
38
|
+
had its links checked not at all. They are now resolved against the site's own
|
|
39
|
+
origin like any other link.
|
|
40
|
+
|
|
9
41
|
## [0.11.0] - 2026-09-08
|
|
10
42
|
|
|
11
43
|
### Added
|
|
@@ -338,6 +370,8 @@ Initial release. `snapshot`, `check` and `audit` commands; filesystem and HTTP
|
|
|
338
370
|
crawling; diff classified by transition; absolute, cross-page and hreflang audit
|
|
339
371
|
rules; pretty, JSON, markdown, GitHub and HTML reporters.
|
|
340
372
|
|
|
373
|
+
[0.13.0]: https://github.com/shyamexe/pagetrace/compare/v0.12.0...v0.13.0
|
|
374
|
+
[0.12.0]: https://github.com/shyamexe/pagetrace/compare/v0.11.0...v0.12.0
|
|
341
375
|
[0.11.0]: https://github.com/shyamexe/pagetrace/compare/v0.10.0...v0.11.0
|
|
342
376
|
[0.10.0]: https://github.com/shyamexe/pagetrace/compare/v0.9.1...v0.10.0
|
|
343
377
|
[0.9.1]: https://github.com/shyamexe/pagetrace/compare/v0.9.0...v0.9.1
|
package/README.md
CHANGED
|
@@ -160,11 +160,20 @@ pagetrace update # install it globally
|
|
|
160
160
|
| `redirect.changed` | warn | A route redirects somewhere new |
|
|
161
161
|
| `canonical.redirects` | warn | A canonical points at a URL that redirects |
|
|
162
162
|
| `link.broken.added` | error | A page started linking to a URL that does not exist |
|
|
163
|
+
| `link.external.dead` | warn | An outbound link answers 404 (needs `--external`) |
|
|
163
164
|
| `sitemap.dead` | error | The sitemap lists a URL that answers 404 |
|
|
164
165
|
| `sitemap.redirect` | warn | The sitemap lists a URL that redirects |
|
|
165
166
|
| `og.removed` / `hreflang.removed` | warn | Social or i18n tags dropped |
|
|
166
167
|
| `title.changed` | info | Ordinary copy edit |
|
|
167
168
|
|
|
169
|
+
One page, checked on its own:
|
|
170
|
+
|
|
171
|
+
```bash
|
|
172
|
+
npx pagetrace page https://example.com/blog/my-post
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
No sitemap, no crawl: it fetches that URL and confirms every link on it with a real request, external links included by default (`--no-external` to skip them). Site-wide rules are left to `audit` — a single-page check never looks at robots.txt, so it does not get to say whether one exists.
|
|
176
|
+
|
|
168
177
|
Broken links have a command of their own, when that is the only question you have:
|
|
169
178
|
|
|
170
179
|
```bash
|
|
@@ -174,7 +183,17 @@ npx pagetrace links --dir ./out --format json
|
|
|
174
183
|
|
|
175
184
|
It crawls once, runs the same rules as `audit`, and prints only the link findings — exit 1 if any, or `No broken links found — 42 pages checked.`
|
|
176
185
|
|
|
177
|
-
Internal links are checked
|
|
186
|
+
Internal links are checked always. Outbound links are checked on request:
|
|
187
|
+
|
|
188
|
+
```bash
|
|
189
|
+
npx pagetrace links --url https://example.com --external
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
Only `404` and `410` count as dead. A `403` from a bot wall, a `429`, a timeout or a TLS failure are all reported as nothing at all, because they describe the request rather than the page — which is how link checkers turn into noise nobody reads. `HEAD` first, falling back to `GET` for servers that refuse it, one request per unique URL across the whole site, capped at 200.
|
|
193
|
+
|
|
194
|
+
Think twice before putting `--external` in `check`. A third party's bad afternoon becomes a diff in your repository and a red build you cannot fix.
|
|
195
|
+
|
|
196
|
+
Internal links are checked too. Only the broken ones are stored, so a site's navigation never lands in the lockfile: `link.broken` for a link that is already dead, `link.broken.added` for one this build broke. A `--dir` crawl is authoritative — the build directory is the whole site — while a crawl confirms each candidate with a real request first, because a sitemap routinely omits pages that are live. External links are checked only with `--external`, and only a 404 or 410 counts.
|
|
178
197
|
|
|
179
198
|
Redirects are recorded from the response itself, so they cost no extra requests. A redirect that only adds or drops a trailing slash is server configuration rather than drift and is not reported. `canonical.redirects` is only raised when the canonical's target was actually crawled, so a `--limit` run cannot invent it.
|
|
180
199
|
|
package/dist/cli.cjs
CHANGED
|
@@ -254,13 +254,24 @@ function auditPage(page, config = {}) {
|
|
|
254
254
|
return findings;
|
|
255
255
|
}
|
|
256
256
|
function auditLinks(page) {
|
|
257
|
-
return
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
257
|
+
return [
|
|
258
|
+
...(page.brokenLinks ?? []).map((target) => ({
|
|
259
|
+
code: "link.broken",
|
|
260
|
+
severity: "error",
|
|
261
|
+
route: page.route,
|
|
262
|
+
message: `Links to ${target}, which does not exist.`,
|
|
263
|
+
after: target
|
|
264
|
+
})),
|
|
265
|
+
// A warning, not an error: the page it points at belongs to somebody else,
|
|
266
|
+
// who can delete it on a Tuesday without asking you.
|
|
267
|
+
...(page.deadExternal ?? []).map((target) => ({
|
|
268
|
+
code: "link.external.dead",
|
|
269
|
+
severity: "warn",
|
|
270
|
+
route: page.route,
|
|
271
|
+
message: `Links out to ${target}, which answers 404.`,
|
|
272
|
+
after: target
|
|
273
|
+
}))
|
|
274
|
+
];
|
|
264
275
|
}
|
|
265
276
|
function auditSite(snapshot) {
|
|
266
277
|
const findings = [];
|
|
@@ -631,6 +642,12 @@ function diffPage(before, after) {
|
|
|
631
642
|
push("link.broken.removed", "info", `Link to ${target} was fixed or removed.`, {
|
|
632
643
|
before: target
|
|
633
644
|
});
|
|
645
|
+
const deadBefore = before.deadExternal ?? [];
|
|
646
|
+
const deadAfter = after.deadExternal ?? [];
|
|
647
|
+
for (const target of deadAfter.filter((t) => !deadBefore.includes(t)))
|
|
648
|
+
push("link.external.dead.added", "warn", `Links out to ${target}, which answers 404.`, {
|
|
649
|
+
after: target
|
|
650
|
+
});
|
|
634
651
|
const droppedHreflang = Object.keys(before.hreflang).filter((k) => !(k in after.hreflang));
|
|
635
652
|
if (droppedHreflang.length > 0)
|
|
636
653
|
push("hreflang.removed", "warn", "hreflang alternates were removed.", {
|
|
@@ -739,6 +756,10 @@ var GUIDANCE = {
|
|
|
739
756
|
nextjs: "A `<Link href>` pointing at a route that no longer exists. TypeScript will not catch it \u2014 typed routes are opt-in via `experimental.typedRoutes`."
|
|
740
757
|
}
|
|
741
758
|
},
|
|
759
|
+
"link.external.dead": {
|
|
760
|
+
why: "An outbound link to a page that is gone sends readers to a 404 and spends the trust the link was passing on nothing. Unlike an internal link, the page is not yours to restore.",
|
|
761
|
+
fix: "Point the link at the current URL, at an archived copy, or remove it. A link that has been dead a while is usually a citation worth replacing rather than deleting."
|
|
762
|
+
},
|
|
742
763
|
"sitemap.dead": {
|
|
743
764
|
why: "A sitemap is a list of URLs you are asking to have crawled. Entries that 404 spend crawl budget on nothing and lower the trust placed in the rest of the file.",
|
|
744
765
|
fix: "Remove the URL from the sitemap, or restore the page. If it moved, redirect it and list the destination instead.",
|
|
@@ -1520,8 +1541,8 @@ function extractLinks(html) {
|
|
|
1520
1541
|
const hrefs = /* @__PURE__ */ new Set();
|
|
1521
1542
|
for (const anchor of root.querySelectorAll("a[href]")) {
|
|
1522
1543
|
const href = anchor.getAttribute("href")?.trim();
|
|
1523
|
-
if (!href) continue;
|
|
1524
|
-
if (
|
|
1544
|
+
if (!href || href.startsWith("#")) continue;
|
|
1545
|
+
if (/^[a-z][a-z0-9+.-]*:/i.test(href) && !/^https?:/i.test(href)) continue;
|
|
1525
1546
|
hrefs.add(href);
|
|
1526
1547
|
}
|
|
1527
1548
|
return [...hrefs];
|
|
@@ -1615,6 +1636,7 @@ async function walkHtml(dir, acc = []) {
|
|
|
1615
1636
|
}
|
|
1616
1637
|
return acc;
|
|
1617
1638
|
}
|
|
1639
|
+
var USER_AGENT = "pagetrace (+https://npmjs.com/package/pagetrace)";
|
|
1618
1640
|
var DEFAULT_TIMEOUT_MS = 15e3;
|
|
1619
1641
|
var MAX_ATTEMPTS = 3;
|
|
1620
1642
|
var RETRY_BASE_MS = 300;
|
|
@@ -1630,7 +1652,7 @@ async function fetchDoc(url, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
|
1630
1652
|
try {
|
|
1631
1653
|
response = await fetch(url, {
|
|
1632
1654
|
signal: AbortSignal.timeout(timeoutMs),
|
|
1633
|
-
headers: { "user-agent":
|
|
1655
|
+
headers: { "user-agent": USER_AGENT }
|
|
1634
1656
|
});
|
|
1635
1657
|
} catch (cause) {
|
|
1636
1658
|
lastError = new Error(`Could not reach ${url}: ${cause.message}`, { cause });
|
|
@@ -1675,7 +1697,9 @@ async function snapshotFromDir(dir, config = {}) {
|
|
|
1675
1697
|
if (robots !== null) site.robotsTxt = extractRobotsTxt(robots, agents);
|
|
1676
1698
|
const llms = await (0, import_promises.readFile)((0, import_node_path.join)(dir, "llms.txt"), "utf8").catch(() => null);
|
|
1677
1699
|
if (llms !== null) site.llmsTxt = extractLlmsTxt(llms);
|
|
1678
|
-
|
|
1700
|
+
const origin = site.origin ?? "https://pagetrace.invalid";
|
|
1701
|
+
await resolveBrokenLinks(pages, links, origin);
|
|
1702
|
+
if (config.checkExternal) await resolveDeadExternal(pages, links, origin, 5);
|
|
1679
1703
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1680
1704
|
}
|
|
1681
1705
|
var ASSET_PATH = /\.(?!html?$)[a-z0-9]+$/i;
|
|
@@ -1714,6 +1738,92 @@ async function resolveBrokenLinks(pages, links, origin, verify) {
|
|
|
1714
1738
|
if (confirmed.length > 0) pages[route].brokenLinks = confirmed.sort();
|
|
1715
1739
|
}
|
|
1716
1740
|
}
|
|
1741
|
+
function externalUrl(href, from, origin) {
|
|
1742
|
+
try {
|
|
1743
|
+
const url = new URL(href, `${origin}${from === "/" ? "" : from}/`);
|
|
1744
|
+
if (url.protocol !== "http:" && url.protocol !== "https:") return null;
|
|
1745
|
+
if (url.origin === origin) return null;
|
|
1746
|
+
url.hash = "";
|
|
1747
|
+
return url.href;
|
|
1748
|
+
} catch {
|
|
1749
|
+
return null;
|
|
1750
|
+
}
|
|
1751
|
+
}
|
|
1752
|
+
var MAX_EXTERNAL_CHECKS = 200;
|
|
1753
|
+
async function findDeadExternal(urls, concurrency, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
1754
|
+
const dead = /* @__PURE__ */ new Set();
|
|
1755
|
+
const queue = urls.slice(0, MAX_EXTERNAL_CHECKS);
|
|
1756
|
+
const request = (url, method) => fetch(url, {
|
|
1757
|
+
method,
|
|
1758
|
+
redirect: "follow",
|
|
1759
|
+
signal: AbortSignal.timeout(timeoutMs),
|
|
1760
|
+
headers: { "user-agent": USER_AGENT }
|
|
1761
|
+
});
|
|
1762
|
+
await Promise.all(
|
|
1763
|
+
Array.from({ length: Math.min(concurrency, queue.length) }, async () => {
|
|
1764
|
+
while (queue.length > 0) {
|
|
1765
|
+
const url = queue.shift();
|
|
1766
|
+
try {
|
|
1767
|
+
let response = await request(url, "HEAD");
|
|
1768
|
+
if (response.status === 405 || response.status === 501) {
|
|
1769
|
+
response = await request(url, "GET");
|
|
1770
|
+
}
|
|
1771
|
+
if (response.status === 404 || response.status === 410) dead.add(url);
|
|
1772
|
+
} catch {
|
|
1773
|
+
}
|
|
1774
|
+
}
|
|
1775
|
+
})
|
|
1776
|
+
);
|
|
1777
|
+
return dead;
|
|
1778
|
+
}
|
|
1779
|
+
async function resolveDeadExternal(pages, links, origin, concurrency, timeoutMs) {
|
|
1780
|
+
const perPage = /* @__PURE__ */ new Map();
|
|
1781
|
+
const unique = /* @__PURE__ */ new Set();
|
|
1782
|
+
for (const [route, hrefs] of Object.entries(links)) {
|
|
1783
|
+
const external = [
|
|
1784
|
+
...new Set(
|
|
1785
|
+
hrefs.map((href) => externalUrl(href, route, origin)).filter((url) => url !== null)
|
|
1786
|
+
)
|
|
1787
|
+
];
|
|
1788
|
+
if (external.length === 0) continue;
|
|
1789
|
+
perPage.set(route, external);
|
|
1790
|
+
for (const url of external) unique.add(url);
|
|
1791
|
+
}
|
|
1792
|
+
const dead = await findDeadExternal([...unique], concurrency, timeoutMs);
|
|
1793
|
+
for (const [route, external] of perPage) {
|
|
1794
|
+
const found = external.filter((url) => dead.has(url)).sort();
|
|
1795
|
+
if (found.length > 0) pages[route].deadExternal = found;
|
|
1796
|
+
}
|
|
1797
|
+
}
|
|
1798
|
+
async function snapshotFromPage(pageUrl, options = {}) {
|
|
1799
|
+
const target = new URL(pageUrl);
|
|
1800
|
+
const site = {
|
|
1801
|
+
origin: options.siteUrl ? new URL(options.siteUrl).origin : target.origin,
|
|
1802
|
+
robotsTxt: null,
|
|
1803
|
+
llmsTxt: null,
|
|
1804
|
+
sitemap: null
|
|
1805
|
+
};
|
|
1806
|
+
const doc = await fetchDoc(target.href, options.timeout);
|
|
1807
|
+
if (doc === null) throw new Error(`${target.href} answered 404 \u2014 nothing to check.`);
|
|
1808
|
+
const route = routeFromUrl(target.href);
|
|
1809
|
+
const page = extractPage(doc.text, route);
|
|
1810
|
+
const landed = !doc.url ? route : originOf2(doc.url) === target.origin ? routeFromUrl(doc.url) : doc.url;
|
|
1811
|
+
page.redirectsTo = landed === route ? null : landed;
|
|
1812
|
+
const pages = { [route]: page };
|
|
1813
|
+
const links = { [route]: extractLinks(doc.text) };
|
|
1814
|
+
const origin = site.origin ?? target.origin;
|
|
1815
|
+
await resolveBrokenLinks(pages, links, origin, async (candidate) => {
|
|
1816
|
+
try {
|
|
1817
|
+
return await fetchDoc(new URL(candidate, target.origin).href, options.timeout) === null;
|
|
1818
|
+
} catch {
|
|
1819
|
+
return false;
|
|
1820
|
+
}
|
|
1821
|
+
});
|
|
1822
|
+
if (options.checkExternal) {
|
|
1823
|
+
await resolveDeadExternal(pages, links, origin, Math.max(1, options.concurrency ?? 5), options.timeout);
|
|
1824
|
+
}
|
|
1825
|
+
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1826
|
+
}
|
|
1717
1827
|
var MAX_NESTED_SITEMAPS = 50;
|
|
1718
1828
|
async function snapshotFromOrigin(origin, options = {}) {
|
|
1719
1829
|
const base = new URL(origin);
|
|
@@ -1802,6 +1912,9 @@ async function snapshotFromOrigin(origin, options = {}) {
|
|
|
1802
1912
|
return false;
|
|
1803
1913
|
}
|
|
1804
1914
|
});
|
|
1915
|
+
if (options.checkExternal) {
|
|
1916
|
+
await resolveDeadExternal(pages, links, base.origin, concurrency, timeout);
|
|
1917
|
+
}
|
|
1805
1918
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1806
1919
|
}
|
|
1807
1920
|
|
|
@@ -1854,13 +1967,15 @@ async function loadConfig(path = DEFAULT_CONFIG) {
|
|
|
1854
1967
|
}
|
|
1855
1968
|
}
|
|
1856
1969
|
async function build(flags, config) {
|
|
1857
|
-
|
|
1970
|
+
const checkExternal = flags.external ?? config.checkExternal;
|
|
1971
|
+
if (flags.dir) return snapshotFromDir(flags.dir, { ...config, checkExternal });
|
|
1858
1972
|
if (flags.url)
|
|
1859
1973
|
return snapshotFromOrigin(flags.url, {
|
|
1860
1974
|
...config,
|
|
1861
1975
|
limit: flags.limit,
|
|
1862
1976
|
concurrency: flags.concurrency,
|
|
1863
|
-
ignoreRobots: flags.ignoreRobots ?? config.ignoreRobots
|
|
1977
|
+
ignoreRobots: flags.ignoreRobots ?? config.ignoreRobots,
|
|
1978
|
+
checkExternal
|
|
1864
1979
|
});
|
|
1865
1980
|
throw new Error("Provide a source: --dir <build directory> or --url <origin>.");
|
|
1866
1981
|
}
|
|
@@ -1879,7 +1994,7 @@ function render(findings, format) {
|
|
|
1879
1994
|
}
|
|
1880
1995
|
}
|
|
1881
1996
|
var cli = (0, import_cac.cac)("pagetrace");
|
|
1882
|
-
cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
1997
|
+
cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
1883
1998
|
const config = await loadConfig(flags.config);
|
|
1884
1999
|
const snapshot = await build(flags, config);
|
|
1885
2000
|
const written = await writeLockfile(flags.out, snapshot);
|
|
@@ -1888,7 +2003,7 @@ cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").opti
|
|
|
1888
2003
|
written ? import_picocolors2.default.green(`Wrote ${flags.out} \u2014 ${count} page${count === 1 ? "" : "s"}.`) : import_picocolors2.default.dim(`${flags.out} is already up to date \u2014 ${count} page${count === 1 ? "" : "s"}.`)
|
|
1889
2004
|
);
|
|
1890
2005
|
});
|
|
1891
|
-
cli.command("check", "Compare the current surface against the lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--lockfile <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | github | sarif", { default: "pretty" }).option("--fail-on <severity>", "error | warn | info", { default: "error" }).option("--audit", "Also run absolute rules, not just the diff", { default: true }).option("--update", "Write the new state to the lockfile after reporting").option("--baseline-branch <ref>", "Read the baseline lockfile from a git ref instead of disk").action(async (flags) => {
|
|
2006
|
+
cli.command("check", "Compare the current surface against the lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--lockfile <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | github | sarif", { default: "pretty" }).option("--fail-on <severity>", "error | warn | info", { default: "error" }).option("--audit", "Also run absolute rules, not just the diff", { default: true }).option("--update", "Write the new state to the lockfile after reporting").option("--baseline-branch <ref>", "Read the baseline lockfile from a git ref instead of disk").action(async (flags) => {
|
|
1892
2007
|
const failOn = parseFailOn(flags.failOn, false);
|
|
1893
2008
|
const config = await loadConfig(flags.config);
|
|
1894
2009
|
const next = await build(flags, config);
|
|
@@ -1918,7 +2033,7 @@ Failing: ${summary.error} error, ${summary.warn} warning (--fail-on ${failOn}).`
|
|
|
1918
2033
|
process.exitCode = EXIT_FINDINGS;
|
|
1919
2034
|
}
|
|
1920
2035
|
});
|
|
1921
|
-
cli.command("audit", "Audit a site as it stands, with explanations and fixes").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | html", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "never" }).action(async (flags) => {
|
|
2036
|
+
cli.command("audit", "Audit a site as it stands, with explanations and fixes").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | html", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "never" }).action(async (flags) => {
|
|
1922
2037
|
const failOn = parseFailOn(flags.failOn, true);
|
|
1923
2038
|
const config = await loadConfig(flags.config);
|
|
1924
2039
|
const snapshot = await build(flags, config);
|
|
@@ -1958,7 +2073,43 @@ cli.command("audit", "Audit a site as it stands, with explanations and fixes").o
|
|
|
1958
2073
|
process.exitCode = EXIT_FINDINGS;
|
|
1959
2074
|
}
|
|
1960
2075
|
});
|
|
1961
|
-
cli.command("
|
|
2076
|
+
cli.command("page <url>", "Check one page: every link on it, and its own surface").option("--external", "Check links that leave the site", { default: true }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "error" }).action(async (url, flags) => {
|
|
2077
|
+
const failOn = parseFailOn(flags.failOn, true);
|
|
2078
|
+
const config = await loadConfig(flags.config);
|
|
2079
|
+
const snapshot = await snapshotFromPage(url, {
|
|
2080
|
+
...config,
|
|
2081
|
+
concurrency: flags.concurrency,
|
|
2082
|
+
checkExternal: flags.external
|
|
2083
|
+
});
|
|
2084
|
+
const [page] = Object.values(snapshot.pages);
|
|
2085
|
+
const platform = detectPlatform([page.generator], Object.values(page.og));
|
|
2086
|
+
const findings = applyConfig(auditSnapshot(snapshot, config), config).filter((f) => f.route !== null).map((f) => withGuidance(f, platform));
|
|
2087
|
+
if (findings.length === 0 && !flags.out) {
|
|
2088
|
+
const checked = (page.brokenLinks?.length ?? 0) + (page.deadExternal?.length ?? 0);
|
|
2089
|
+
console.log(import_picocolors2.default.green(`${url} looks sound \u2014 no findings, no dead links.`));
|
|
2090
|
+
if (checked > 0) console.log(import_picocolors2.default.dim("(unreachable links were treated as unknown, not dead)"));
|
|
2091
|
+
return;
|
|
2092
|
+
}
|
|
2093
|
+
const groups = aggregate(findings);
|
|
2094
|
+
const meta = {
|
|
2095
|
+
target: url,
|
|
2096
|
+
platform,
|
|
2097
|
+
pageCount: 1,
|
|
2098
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString().slice(0, 10)
|
|
2099
|
+
};
|
|
2100
|
+
const output = flags.format === "json" ? JSON.stringify({ schemaVersion: 1, meta, summary: summarize(findings), groups }, null, 2) : flags.format === "markdown" ? formatAuditMarkdown(groups, meta) : formatAuditPretty(groups, meta, process.stdout.columns);
|
|
2101
|
+
if (flags.out) {
|
|
2102
|
+
await (0, import_promises2.writeFile)(flags.out, `${output}
|
|
2103
|
+
`, "utf8");
|
|
2104
|
+
console.log(import_picocolors2.default.green(`Wrote ${flags.out}.`));
|
|
2105
|
+
} else {
|
|
2106
|
+
console.log(output);
|
|
2107
|
+
}
|
|
2108
|
+
if (failOn !== "never" && shouldFail(findings, failOn)) {
|
|
2109
|
+
process.exitCode = EXIT_FINDINGS;
|
|
2110
|
+
}
|
|
2111
|
+
});
|
|
2112
|
+
cli.command("links", "Find internal links that point at no page").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "error" }).action(async (flags) => {
|
|
1962
2113
|
const failOn = parseFailOn(flags.failOn, true);
|
|
1963
2114
|
const config = await loadConfig(flags.config);
|
|
1964
2115
|
const snapshot = await build(flags, config);
|
|
@@ -2005,7 +2156,7 @@ cli.command("links", "Find internal links that point at no page").option("--url
|
|
|
2005
2156
|
process.exitCode = EXIT_FINDINGS;
|
|
2006
2157
|
}
|
|
2007
2158
|
});
|
|
2008
|
-
cli.command("init", "Write a config file and take the first snapshot").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
2159
|
+
cli.command("init", "Write a config file and take the first snapshot").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
2009
2160
|
const existing = await (0, import_promises2.readFile)(flags.config, "utf8").catch(() => null);
|
|
2010
2161
|
if (existing === null) {
|
|
2011
2162
|
const config = flags.url ? { siteUrl: new URL(flags.url).origin } : {};
|
|
@@ -2028,11 +2179,11 @@ Commit both files, then run \`pagetrace check ${flags.dir ? `--dir ${flags.dir}`
|
|
|
2028
2179
|
});
|
|
2029
2180
|
cli.command("update", "Check npm for a newer pagetrace and install it").option("--check", "Only report whether an update exists").action(async (flags) => {
|
|
2030
2181
|
const latest = await latestVersion("pagetrace");
|
|
2031
|
-
if (!isNewer(latest, "0.
|
|
2032
|
-
console.log(import_picocolors2.default.green(`pagetrace ${"0.
|
|
2182
|
+
if (!isNewer(latest, "0.13.0")) {
|
|
2183
|
+
console.log(import_picocolors2.default.green(`pagetrace ${"0.13.0"} is the latest version.`));
|
|
2033
2184
|
return;
|
|
2034
2185
|
}
|
|
2035
|
-
console.log(import_picocolors2.default.yellow(`Update available: ${"0.
|
|
2186
|
+
console.log(import_picocolors2.default.yellow(`Update available: ${"0.13.0"} \u2192 ${latest}`));
|
|
2036
2187
|
if (flags.check) return;
|
|
2037
2188
|
if (process.argv[1]?.startsWith(process.cwd())) {
|
|
2038
2189
|
console.log(
|
|
@@ -2050,7 +2201,7 @@ cli.command("update", "Check npm for a newer pagetrace and install it").option("
|
|
|
2050
2201
|
console.log(import_picocolors2.default.green(`Updated to pagetrace ${latest}.`));
|
|
2051
2202
|
});
|
|
2052
2203
|
cli.help();
|
|
2053
|
-
cli.version("0.
|
|
2204
|
+
cli.version("0.13.0");
|
|
2054
2205
|
async function main() {
|
|
2055
2206
|
try {
|
|
2056
2207
|
cli.parse(process.argv, { run: false });
|
package/dist/cli.js
CHANGED
|
@@ -231,13 +231,24 @@ function auditPage(page, config = {}) {
|
|
|
231
231
|
return findings;
|
|
232
232
|
}
|
|
233
233
|
function auditLinks(page) {
|
|
234
|
-
return
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
234
|
+
return [
|
|
235
|
+
...(page.brokenLinks ?? []).map((target) => ({
|
|
236
|
+
code: "link.broken",
|
|
237
|
+
severity: "error",
|
|
238
|
+
route: page.route,
|
|
239
|
+
message: `Links to ${target}, which does not exist.`,
|
|
240
|
+
after: target
|
|
241
|
+
})),
|
|
242
|
+
// A warning, not an error: the page it points at belongs to somebody else,
|
|
243
|
+
// who can delete it on a Tuesday without asking you.
|
|
244
|
+
...(page.deadExternal ?? []).map((target) => ({
|
|
245
|
+
code: "link.external.dead",
|
|
246
|
+
severity: "warn",
|
|
247
|
+
route: page.route,
|
|
248
|
+
message: `Links out to ${target}, which answers 404.`,
|
|
249
|
+
after: target
|
|
250
|
+
}))
|
|
251
|
+
];
|
|
241
252
|
}
|
|
242
253
|
function auditSite(snapshot) {
|
|
243
254
|
const findings = [];
|
|
@@ -608,6 +619,12 @@ function diffPage(before, after) {
|
|
|
608
619
|
push("link.broken.removed", "info", `Link to ${target} was fixed or removed.`, {
|
|
609
620
|
before: target
|
|
610
621
|
});
|
|
622
|
+
const deadBefore = before.deadExternal ?? [];
|
|
623
|
+
const deadAfter = after.deadExternal ?? [];
|
|
624
|
+
for (const target of deadAfter.filter((t) => !deadBefore.includes(t)))
|
|
625
|
+
push("link.external.dead.added", "warn", `Links out to ${target}, which answers 404.`, {
|
|
626
|
+
after: target
|
|
627
|
+
});
|
|
611
628
|
const droppedHreflang = Object.keys(before.hreflang).filter((k) => !(k in after.hreflang));
|
|
612
629
|
if (droppedHreflang.length > 0)
|
|
613
630
|
push("hreflang.removed", "warn", "hreflang alternates were removed.", {
|
|
@@ -716,6 +733,10 @@ var GUIDANCE = {
|
|
|
716
733
|
nextjs: "A `<Link href>` pointing at a route that no longer exists. TypeScript will not catch it \u2014 typed routes are opt-in via `experimental.typedRoutes`."
|
|
717
734
|
}
|
|
718
735
|
},
|
|
736
|
+
"link.external.dead": {
|
|
737
|
+
why: "An outbound link to a page that is gone sends readers to a 404 and spends the trust the link was passing on nothing. Unlike an internal link, the page is not yours to restore.",
|
|
738
|
+
fix: "Point the link at the current URL, at an archived copy, or remove it. A link that has been dead a while is usually a citation worth replacing rather than deleting."
|
|
739
|
+
},
|
|
719
740
|
"sitemap.dead": {
|
|
720
741
|
why: "A sitemap is a list of URLs you are asking to have crawled. Entries that 404 spend crawl budget on nothing and lower the trust placed in the rest of the file.",
|
|
721
742
|
fix: "Remove the URL from the sitemap, or restore the page. If it moved, redirect it and list the destination instead.",
|
|
@@ -1497,8 +1518,8 @@ function extractLinks(html) {
|
|
|
1497
1518
|
const hrefs = /* @__PURE__ */ new Set();
|
|
1498
1519
|
for (const anchor of root.querySelectorAll("a[href]")) {
|
|
1499
1520
|
const href = anchor.getAttribute("href")?.trim();
|
|
1500
|
-
if (!href) continue;
|
|
1501
|
-
if (
|
|
1521
|
+
if (!href || href.startsWith("#")) continue;
|
|
1522
|
+
if (/^[a-z][a-z0-9+.-]*:/i.test(href) && !/^https?:/i.test(href)) continue;
|
|
1502
1523
|
hrefs.add(href);
|
|
1503
1524
|
}
|
|
1504
1525
|
return [...hrefs];
|
|
@@ -1592,6 +1613,7 @@ async function walkHtml(dir, acc = []) {
|
|
|
1592
1613
|
}
|
|
1593
1614
|
return acc;
|
|
1594
1615
|
}
|
|
1616
|
+
var USER_AGENT = "pagetrace (+https://npmjs.com/package/pagetrace)";
|
|
1595
1617
|
var DEFAULT_TIMEOUT_MS = 15e3;
|
|
1596
1618
|
var MAX_ATTEMPTS = 3;
|
|
1597
1619
|
var RETRY_BASE_MS = 300;
|
|
@@ -1607,7 +1629,7 @@ async function fetchDoc(url, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
|
1607
1629
|
try {
|
|
1608
1630
|
response = await fetch(url, {
|
|
1609
1631
|
signal: AbortSignal.timeout(timeoutMs),
|
|
1610
|
-
headers: { "user-agent":
|
|
1632
|
+
headers: { "user-agent": USER_AGENT }
|
|
1611
1633
|
});
|
|
1612
1634
|
} catch (cause) {
|
|
1613
1635
|
lastError = new Error(`Could not reach ${url}: ${cause.message}`, { cause });
|
|
@@ -1652,7 +1674,9 @@ async function snapshotFromDir(dir, config = {}) {
|
|
|
1652
1674
|
if (robots !== null) site.robotsTxt = extractRobotsTxt(robots, agents);
|
|
1653
1675
|
const llms = await readFile(join(dir, "llms.txt"), "utf8").catch(() => null);
|
|
1654
1676
|
if (llms !== null) site.llmsTxt = extractLlmsTxt(llms);
|
|
1655
|
-
|
|
1677
|
+
const origin = site.origin ?? "https://pagetrace.invalid";
|
|
1678
|
+
await resolveBrokenLinks(pages, links, origin);
|
|
1679
|
+
if (config.checkExternal) await resolveDeadExternal(pages, links, origin, 5);
|
|
1656
1680
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1657
1681
|
}
|
|
1658
1682
|
var ASSET_PATH = /\.(?!html?$)[a-z0-9]+$/i;
|
|
@@ -1691,6 +1715,92 @@ async function resolveBrokenLinks(pages, links, origin, verify) {
|
|
|
1691
1715
|
if (confirmed.length > 0) pages[route].brokenLinks = confirmed.sort();
|
|
1692
1716
|
}
|
|
1693
1717
|
}
|
|
1718
|
+
function externalUrl(href, from, origin) {
|
|
1719
|
+
try {
|
|
1720
|
+
const url = new URL(href, `${origin}${from === "/" ? "" : from}/`);
|
|
1721
|
+
if (url.protocol !== "http:" && url.protocol !== "https:") return null;
|
|
1722
|
+
if (url.origin === origin) return null;
|
|
1723
|
+
url.hash = "";
|
|
1724
|
+
return url.href;
|
|
1725
|
+
} catch {
|
|
1726
|
+
return null;
|
|
1727
|
+
}
|
|
1728
|
+
}
|
|
1729
|
+
var MAX_EXTERNAL_CHECKS = 200;
|
|
1730
|
+
async function findDeadExternal(urls, concurrency, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
1731
|
+
const dead = /* @__PURE__ */ new Set();
|
|
1732
|
+
const queue = urls.slice(0, MAX_EXTERNAL_CHECKS);
|
|
1733
|
+
const request = (url, method) => fetch(url, {
|
|
1734
|
+
method,
|
|
1735
|
+
redirect: "follow",
|
|
1736
|
+
signal: AbortSignal.timeout(timeoutMs),
|
|
1737
|
+
headers: { "user-agent": USER_AGENT }
|
|
1738
|
+
});
|
|
1739
|
+
await Promise.all(
|
|
1740
|
+
Array.from({ length: Math.min(concurrency, queue.length) }, async () => {
|
|
1741
|
+
while (queue.length > 0) {
|
|
1742
|
+
const url = queue.shift();
|
|
1743
|
+
try {
|
|
1744
|
+
let response = await request(url, "HEAD");
|
|
1745
|
+
if (response.status === 405 || response.status === 501) {
|
|
1746
|
+
response = await request(url, "GET");
|
|
1747
|
+
}
|
|
1748
|
+
if (response.status === 404 || response.status === 410) dead.add(url);
|
|
1749
|
+
} catch {
|
|
1750
|
+
}
|
|
1751
|
+
}
|
|
1752
|
+
})
|
|
1753
|
+
);
|
|
1754
|
+
return dead;
|
|
1755
|
+
}
|
|
1756
|
+
async function resolveDeadExternal(pages, links, origin, concurrency, timeoutMs) {
|
|
1757
|
+
const perPage = /* @__PURE__ */ new Map();
|
|
1758
|
+
const unique = /* @__PURE__ */ new Set();
|
|
1759
|
+
for (const [route, hrefs] of Object.entries(links)) {
|
|
1760
|
+
const external = [
|
|
1761
|
+
...new Set(
|
|
1762
|
+
hrefs.map((href) => externalUrl(href, route, origin)).filter((url) => url !== null)
|
|
1763
|
+
)
|
|
1764
|
+
];
|
|
1765
|
+
if (external.length === 0) continue;
|
|
1766
|
+
perPage.set(route, external);
|
|
1767
|
+
for (const url of external) unique.add(url);
|
|
1768
|
+
}
|
|
1769
|
+
const dead = await findDeadExternal([...unique], concurrency, timeoutMs);
|
|
1770
|
+
for (const [route, external] of perPage) {
|
|
1771
|
+
const found = external.filter((url) => dead.has(url)).sort();
|
|
1772
|
+
if (found.length > 0) pages[route].deadExternal = found;
|
|
1773
|
+
}
|
|
1774
|
+
}
|
|
1775
|
+
async function snapshotFromPage(pageUrl, options = {}) {
|
|
1776
|
+
const target = new URL(pageUrl);
|
|
1777
|
+
const site = {
|
|
1778
|
+
origin: options.siteUrl ? new URL(options.siteUrl).origin : target.origin,
|
|
1779
|
+
robotsTxt: null,
|
|
1780
|
+
llmsTxt: null,
|
|
1781
|
+
sitemap: null
|
|
1782
|
+
};
|
|
1783
|
+
const doc = await fetchDoc(target.href, options.timeout);
|
|
1784
|
+
if (doc === null) throw new Error(`${target.href} answered 404 \u2014 nothing to check.`);
|
|
1785
|
+
const route = routeFromUrl(target.href);
|
|
1786
|
+
const page = extractPage(doc.text, route);
|
|
1787
|
+
const landed = !doc.url ? route : originOf2(doc.url) === target.origin ? routeFromUrl(doc.url) : doc.url;
|
|
1788
|
+
page.redirectsTo = landed === route ? null : landed;
|
|
1789
|
+
const pages = { [route]: page };
|
|
1790
|
+
const links = { [route]: extractLinks(doc.text) };
|
|
1791
|
+
const origin = site.origin ?? target.origin;
|
|
1792
|
+
await resolveBrokenLinks(pages, links, origin, async (candidate) => {
|
|
1793
|
+
try {
|
|
1794
|
+
return await fetchDoc(new URL(candidate, target.origin).href, options.timeout) === null;
|
|
1795
|
+
} catch {
|
|
1796
|
+
return false;
|
|
1797
|
+
}
|
|
1798
|
+
});
|
|
1799
|
+
if (options.checkExternal) {
|
|
1800
|
+
await resolveDeadExternal(pages, links, origin, Math.max(1, options.concurrency ?? 5), options.timeout);
|
|
1801
|
+
}
|
|
1802
|
+
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1803
|
+
}
|
|
1694
1804
|
var MAX_NESTED_SITEMAPS = 50;
|
|
1695
1805
|
async function snapshotFromOrigin(origin, options = {}) {
|
|
1696
1806
|
const base = new URL(origin);
|
|
@@ -1779,6 +1889,9 @@ async function snapshotFromOrigin(origin, options = {}) {
|
|
|
1779
1889
|
return false;
|
|
1780
1890
|
}
|
|
1781
1891
|
});
|
|
1892
|
+
if (options.checkExternal) {
|
|
1893
|
+
await resolveDeadExternal(pages, links, base.origin, concurrency, timeout);
|
|
1894
|
+
}
|
|
1782
1895
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1783
1896
|
}
|
|
1784
1897
|
|
|
@@ -1831,13 +1944,15 @@ async function loadConfig(path = DEFAULT_CONFIG) {
|
|
|
1831
1944
|
}
|
|
1832
1945
|
}
|
|
1833
1946
|
async function build(flags, config) {
|
|
1834
|
-
|
|
1947
|
+
const checkExternal = flags.external ?? config.checkExternal;
|
|
1948
|
+
if (flags.dir) return snapshotFromDir(flags.dir, { ...config, checkExternal });
|
|
1835
1949
|
if (flags.url)
|
|
1836
1950
|
return snapshotFromOrigin(flags.url, {
|
|
1837
1951
|
...config,
|
|
1838
1952
|
limit: flags.limit,
|
|
1839
1953
|
concurrency: flags.concurrency,
|
|
1840
|
-
ignoreRobots: flags.ignoreRobots ?? config.ignoreRobots
|
|
1954
|
+
ignoreRobots: flags.ignoreRobots ?? config.ignoreRobots,
|
|
1955
|
+
checkExternal
|
|
1841
1956
|
});
|
|
1842
1957
|
throw new Error("Provide a source: --dir <build directory> or --url <origin>.");
|
|
1843
1958
|
}
|
|
@@ -1856,7 +1971,7 @@ function render(findings, format) {
|
|
|
1856
1971
|
}
|
|
1857
1972
|
}
|
|
1858
1973
|
var cli = cac("pagetrace");
|
|
1859
|
-
cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
1974
|
+
cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
1860
1975
|
const config = await loadConfig(flags.config);
|
|
1861
1976
|
const snapshot = await build(flags, config);
|
|
1862
1977
|
const written = await writeLockfile(flags.out, snapshot);
|
|
@@ -1865,7 +1980,7 @@ cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").opti
|
|
|
1865
1980
|
written ? pc2.green(`Wrote ${flags.out} \u2014 ${count} page${count === 1 ? "" : "s"}.`) : pc2.dim(`${flags.out} is already up to date \u2014 ${count} page${count === 1 ? "" : "s"}.`)
|
|
1866
1981
|
);
|
|
1867
1982
|
});
|
|
1868
|
-
cli.command("check", "Compare the current surface against the lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--lockfile <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | github | sarif", { default: "pretty" }).option("--fail-on <severity>", "error | warn | info", { default: "error" }).option("--audit", "Also run absolute rules, not just the diff", { default: true }).option("--update", "Write the new state to the lockfile after reporting").option("--baseline-branch <ref>", "Read the baseline lockfile from a git ref instead of disk").action(async (flags) => {
|
|
1983
|
+
cli.command("check", "Compare the current surface against the lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--lockfile <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | github | sarif", { default: "pretty" }).option("--fail-on <severity>", "error | warn | info", { default: "error" }).option("--audit", "Also run absolute rules, not just the diff", { default: true }).option("--update", "Write the new state to the lockfile after reporting").option("--baseline-branch <ref>", "Read the baseline lockfile from a git ref instead of disk").action(async (flags) => {
|
|
1869
1984
|
const failOn = parseFailOn(flags.failOn, false);
|
|
1870
1985
|
const config = await loadConfig(flags.config);
|
|
1871
1986
|
const next = await build(flags, config);
|
|
@@ -1895,7 +2010,7 @@ Failing: ${summary.error} error, ${summary.warn} warning (--fail-on ${failOn}).`
|
|
|
1895
2010
|
process.exitCode = EXIT_FINDINGS;
|
|
1896
2011
|
}
|
|
1897
2012
|
});
|
|
1898
|
-
cli.command("audit", "Audit a site as it stands, with explanations and fixes").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | html", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "never" }).action(async (flags) => {
|
|
2013
|
+
cli.command("audit", "Audit a site as it stands, with explanations and fixes").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | html", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "never" }).action(async (flags) => {
|
|
1899
2014
|
const failOn = parseFailOn(flags.failOn, true);
|
|
1900
2015
|
const config = await loadConfig(flags.config);
|
|
1901
2016
|
const snapshot = await build(flags, config);
|
|
@@ -1935,7 +2050,43 @@ cli.command("audit", "Audit a site as it stands, with explanations and fixes").o
|
|
|
1935
2050
|
process.exitCode = EXIT_FINDINGS;
|
|
1936
2051
|
}
|
|
1937
2052
|
});
|
|
1938
|
-
cli.command("
|
|
2053
|
+
cli.command("page <url>", "Check one page: every link on it, and its own surface").option("--external", "Check links that leave the site", { default: true }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "error" }).action(async (url, flags) => {
|
|
2054
|
+
const failOn = parseFailOn(flags.failOn, true);
|
|
2055
|
+
const config = await loadConfig(flags.config);
|
|
2056
|
+
const snapshot = await snapshotFromPage(url, {
|
|
2057
|
+
...config,
|
|
2058
|
+
concurrency: flags.concurrency,
|
|
2059
|
+
checkExternal: flags.external
|
|
2060
|
+
});
|
|
2061
|
+
const [page] = Object.values(snapshot.pages);
|
|
2062
|
+
const platform = detectPlatform([page.generator], Object.values(page.og));
|
|
2063
|
+
const findings = applyConfig(auditSnapshot(snapshot, config), config).filter((f) => f.route !== null).map((f) => withGuidance(f, platform));
|
|
2064
|
+
if (findings.length === 0 && !flags.out) {
|
|
2065
|
+
const checked = (page.brokenLinks?.length ?? 0) + (page.deadExternal?.length ?? 0);
|
|
2066
|
+
console.log(pc2.green(`${url} looks sound \u2014 no findings, no dead links.`));
|
|
2067
|
+
if (checked > 0) console.log(pc2.dim("(unreachable links were treated as unknown, not dead)"));
|
|
2068
|
+
return;
|
|
2069
|
+
}
|
|
2070
|
+
const groups = aggregate(findings);
|
|
2071
|
+
const meta = {
|
|
2072
|
+
target: url,
|
|
2073
|
+
platform,
|
|
2074
|
+
pageCount: 1,
|
|
2075
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString().slice(0, 10)
|
|
2076
|
+
};
|
|
2077
|
+
const output = flags.format === "json" ? JSON.stringify({ schemaVersion: 1, meta, summary: summarize(findings), groups }, null, 2) : flags.format === "markdown" ? formatAuditMarkdown(groups, meta) : formatAuditPretty(groups, meta, process.stdout.columns);
|
|
2078
|
+
if (flags.out) {
|
|
2079
|
+
await writeFile(flags.out, `${output}
|
|
2080
|
+
`, "utf8");
|
|
2081
|
+
console.log(pc2.green(`Wrote ${flags.out}.`));
|
|
2082
|
+
} else {
|
|
2083
|
+
console.log(output);
|
|
2084
|
+
}
|
|
2085
|
+
if (failOn !== "never" && shouldFail(findings, failOn)) {
|
|
2086
|
+
process.exitCode = EXIT_FINDINGS;
|
|
2087
|
+
}
|
|
2088
|
+
});
|
|
2089
|
+
cli.command("links", "Find internal links that point at no page").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "error" }).action(async (flags) => {
|
|
1939
2090
|
const failOn = parseFailOn(flags.failOn, true);
|
|
1940
2091
|
const config = await loadConfig(flags.config);
|
|
1941
2092
|
const snapshot = await build(flags, config);
|
|
@@ -1982,7 +2133,7 @@ cli.command("links", "Find internal links that point at no page").option("--url
|
|
|
1982
2133
|
process.exitCode = EXIT_FINDINGS;
|
|
1983
2134
|
}
|
|
1984
2135
|
});
|
|
1985
|
-
cli.command("init", "Write a config file and take the first snapshot").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
2136
|
+
cli.command("init", "Write a config file and take the first snapshot").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
1986
2137
|
const existing = await readFile2(flags.config, "utf8").catch(() => null);
|
|
1987
2138
|
if (existing === null) {
|
|
1988
2139
|
const config = flags.url ? { siteUrl: new URL(flags.url).origin } : {};
|
|
@@ -2005,11 +2156,11 @@ Commit both files, then run \`pagetrace check ${flags.dir ? `--dir ${flags.dir}`
|
|
|
2005
2156
|
});
|
|
2006
2157
|
cli.command("update", "Check npm for a newer pagetrace and install it").option("--check", "Only report whether an update exists").action(async (flags) => {
|
|
2007
2158
|
const latest = await latestVersion("pagetrace");
|
|
2008
|
-
if (!isNewer(latest, "0.
|
|
2009
|
-
console.log(pc2.green(`pagetrace ${"0.
|
|
2159
|
+
if (!isNewer(latest, "0.13.0")) {
|
|
2160
|
+
console.log(pc2.green(`pagetrace ${"0.13.0"} is the latest version.`));
|
|
2010
2161
|
return;
|
|
2011
2162
|
}
|
|
2012
|
-
console.log(pc2.yellow(`Update available: ${"0.
|
|
2163
|
+
console.log(pc2.yellow(`Update available: ${"0.13.0"} \u2192 ${latest}`));
|
|
2013
2164
|
if (flags.check) return;
|
|
2014
2165
|
if (process.argv[1]?.startsWith(process.cwd())) {
|
|
2015
2166
|
console.log(
|
|
@@ -2027,7 +2178,7 @@ cli.command("update", "Check npm for a newer pagetrace and install it").option("
|
|
|
2027
2178
|
console.log(pc2.green(`Updated to pagetrace ${latest}.`));
|
|
2028
2179
|
});
|
|
2029
2180
|
cli.help();
|
|
2030
|
-
cli.version("0.
|
|
2181
|
+
cli.version("0.13.0");
|
|
2031
2182
|
async function main() {
|
|
2032
2183
|
try {
|
|
2033
2184
|
cli.parse(process.argv, { run: false });
|
package/dist/index.cjs
CHANGED
|
@@ -67,6 +67,7 @@ __export(src_exports, {
|
|
|
67
67
|
snapshotFromDir: () => snapshotFromDir,
|
|
68
68
|
snapshotFromGitRef: () => snapshotFromGitRef,
|
|
69
69
|
snapshotFromOrigin: () => snapshotFromOrigin,
|
|
70
|
+
snapshotFromPage: () => snapshotFromPage,
|
|
70
71
|
summarize: () => summarize,
|
|
71
72
|
withGuidance: () => withGuidance
|
|
72
73
|
});
|
|
@@ -297,13 +298,24 @@ function auditPage(page, config = {}) {
|
|
|
297
298
|
return findings;
|
|
298
299
|
}
|
|
299
300
|
function auditLinks(page) {
|
|
300
|
-
return
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
301
|
+
return [
|
|
302
|
+
...(page.brokenLinks ?? []).map((target) => ({
|
|
303
|
+
code: "link.broken",
|
|
304
|
+
severity: "error",
|
|
305
|
+
route: page.route,
|
|
306
|
+
message: `Links to ${target}, which does not exist.`,
|
|
307
|
+
after: target
|
|
308
|
+
})),
|
|
309
|
+
// A warning, not an error: the page it points at belongs to somebody else,
|
|
310
|
+
// who can delete it on a Tuesday without asking you.
|
|
311
|
+
...(page.deadExternal ?? []).map((target) => ({
|
|
312
|
+
code: "link.external.dead",
|
|
313
|
+
severity: "warn",
|
|
314
|
+
route: page.route,
|
|
315
|
+
message: `Links out to ${target}, which answers 404.`,
|
|
316
|
+
after: target
|
|
317
|
+
}))
|
|
318
|
+
];
|
|
307
319
|
}
|
|
308
320
|
function auditSite(snapshot) {
|
|
309
321
|
const findings = [];
|
|
@@ -674,6 +686,12 @@ function diffPage(before, after) {
|
|
|
674
686
|
push("link.broken.removed", "info", `Link to ${target} was fixed or removed.`, {
|
|
675
687
|
before: target
|
|
676
688
|
});
|
|
689
|
+
const deadBefore = before.deadExternal ?? [];
|
|
690
|
+
const deadAfter = after.deadExternal ?? [];
|
|
691
|
+
for (const target of deadAfter.filter((t) => !deadBefore.includes(t)))
|
|
692
|
+
push("link.external.dead.added", "warn", `Links out to ${target}, which answers 404.`, {
|
|
693
|
+
after: target
|
|
694
|
+
});
|
|
677
695
|
const droppedHreflang = Object.keys(before.hreflang).filter((k) => !(k in after.hreflang));
|
|
678
696
|
if (droppedHreflang.length > 0)
|
|
679
697
|
push("hreflang.removed", "warn", "hreflang alternates were removed.", {
|
|
@@ -962,8 +980,8 @@ function extractLinks(html) {
|
|
|
962
980
|
const hrefs = /* @__PURE__ */ new Set();
|
|
963
981
|
for (const anchor of root.querySelectorAll("a[href]")) {
|
|
964
982
|
const href = anchor.getAttribute("href")?.trim();
|
|
965
|
-
if (!href) continue;
|
|
966
|
-
if (
|
|
983
|
+
if (!href || href.startsWith("#")) continue;
|
|
984
|
+
if (/^[a-z][a-z0-9+.-]*:/i.test(href) && !/^https?:/i.test(href)) continue;
|
|
967
985
|
hrefs.add(href);
|
|
968
986
|
}
|
|
969
987
|
return [...hrefs];
|
|
@@ -1021,6 +1039,10 @@ var GUIDANCE = {
|
|
|
1021
1039
|
nextjs: "A `<Link href>` pointing at a route that no longer exists. TypeScript will not catch it \u2014 typed routes are opt-in via `experimental.typedRoutes`."
|
|
1022
1040
|
}
|
|
1023
1041
|
},
|
|
1042
|
+
"link.external.dead": {
|
|
1043
|
+
why: "An outbound link to a page that is gone sends readers to a 404 and spends the trust the link was passing on nothing. Unlike an internal link, the page is not yours to restore.",
|
|
1044
|
+
fix: "Point the link at the current URL, at an archived copy, or remove it. A link that has been dead a while is usually a citation worth replacing rather than deleting."
|
|
1045
|
+
},
|
|
1024
1046
|
"sitemap.dead": {
|
|
1025
1047
|
why: "A sitemap is a list of URLs you are asking to have crawled. Entries that 404 spend crawl budget on nothing and lower the trust placed in the rest of the file.",
|
|
1026
1048
|
fix: "Remove the URL from the sitemap, or restore the page. If it moved, redirect it and list the destination instead.",
|
|
@@ -1656,6 +1678,7 @@ async function walkHtml(dir, acc = []) {
|
|
|
1656
1678
|
}
|
|
1657
1679
|
return acc;
|
|
1658
1680
|
}
|
|
1681
|
+
var USER_AGENT = "pagetrace (+https://npmjs.com/package/pagetrace)";
|
|
1659
1682
|
var DEFAULT_TIMEOUT_MS = 15e3;
|
|
1660
1683
|
var MAX_ATTEMPTS = 3;
|
|
1661
1684
|
var RETRY_BASE_MS = 300;
|
|
@@ -1671,7 +1694,7 @@ async function fetchDoc(url, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
|
1671
1694
|
try {
|
|
1672
1695
|
response = await fetch(url, {
|
|
1673
1696
|
signal: AbortSignal.timeout(timeoutMs),
|
|
1674
|
-
headers: { "user-agent":
|
|
1697
|
+
headers: { "user-agent": USER_AGENT }
|
|
1675
1698
|
});
|
|
1676
1699
|
} catch (cause) {
|
|
1677
1700
|
lastError = new Error(`Could not reach ${url}: ${cause.message}`, { cause });
|
|
@@ -1716,7 +1739,9 @@ async function snapshotFromDir(dir, config = {}) {
|
|
|
1716
1739
|
if (robots !== null) site.robotsTxt = extractRobotsTxt(robots, agents);
|
|
1717
1740
|
const llms = await (0, import_promises.readFile)((0, import_node_path.join)(dir, "llms.txt"), "utf8").catch(() => null);
|
|
1718
1741
|
if (llms !== null) site.llmsTxt = extractLlmsTxt(llms);
|
|
1719
|
-
|
|
1742
|
+
const origin = site.origin ?? "https://pagetrace.invalid";
|
|
1743
|
+
await resolveBrokenLinks(pages, links, origin);
|
|
1744
|
+
if (config.checkExternal) await resolveDeadExternal(pages, links, origin, 5);
|
|
1720
1745
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1721
1746
|
}
|
|
1722
1747
|
var ASSET_PATH = /\.(?!html?$)[a-z0-9]+$/i;
|
|
@@ -1755,6 +1780,92 @@ async function resolveBrokenLinks(pages, links, origin, verify) {
|
|
|
1755
1780
|
if (confirmed.length > 0) pages[route].brokenLinks = confirmed.sort();
|
|
1756
1781
|
}
|
|
1757
1782
|
}
|
|
1783
|
+
function externalUrl(href, from, origin) {
|
|
1784
|
+
try {
|
|
1785
|
+
const url = new URL(href, `${origin}${from === "/" ? "" : from}/`);
|
|
1786
|
+
if (url.protocol !== "http:" && url.protocol !== "https:") return null;
|
|
1787
|
+
if (url.origin === origin) return null;
|
|
1788
|
+
url.hash = "";
|
|
1789
|
+
return url.href;
|
|
1790
|
+
} catch {
|
|
1791
|
+
return null;
|
|
1792
|
+
}
|
|
1793
|
+
}
|
|
1794
|
+
var MAX_EXTERNAL_CHECKS = 200;
|
|
1795
|
+
async function findDeadExternal(urls, concurrency, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
1796
|
+
const dead = /* @__PURE__ */ new Set();
|
|
1797
|
+
const queue = urls.slice(0, MAX_EXTERNAL_CHECKS);
|
|
1798
|
+
const request = (url, method) => fetch(url, {
|
|
1799
|
+
method,
|
|
1800
|
+
redirect: "follow",
|
|
1801
|
+
signal: AbortSignal.timeout(timeoutMs),
|
|
1802
|
+
headers: { "user-agent": USER_AGENT }
|
|
1803
|
+
});
|
|
1804
|
+
await Promise.all(
|
|
1805
|
+
Array.from({ length: Math.min(concurrency, queue.length) }, async () => {
|
|
1806
|
+
while (queue.length > 0) {
|
|
1807
|
+
const url = queue.shift();
|
|
1808
|
+
try {
|
|
1809
|
+
let response = await request(url, "HEAD");
|
|
1810
|
+
if (response.status === 405 || response.status === 501) {
|
|
1811
|
+
response = await request(url, "GET");
|
|
1812
|
+
}
|
|
1813
|
+
if (response.status === 404 || response.status === 410) dead.add(url);
|
|
1814
|
+
} catch {
|
|
1815
|
+
}
|
|
1816
|
+
}
|
|
1817
|
+
})
|
|
1818
|
+
);
|
|
1819
|
+
return dead;
|
|
1820
|
+
}
|
|
1821
|
+
async function resolveDeadExternal(pages, links, origin, concurrency, timeoutMs) {
|
|
1822
|
+
const perPage = /* @__PURE__ */ new Map();
|
|
1823
|
+
const unique = /* @__PURE__ */ new Set();
|
|
1824
|
+
for (const [route, hrefs] of Object.entries(links)) {
|
|
1825
|
+
const external = [
|
|
1826
|
+
...new Set(
|
|
1827
|
+
hrefs.map((href) => externalUrl(href, route, origin)).filter((url) => url !== null)
|
|
1828
|
+
)
|
|
1829
|
+
];
|
|
1830
|
+
if (external.length === 0) continue;
|
|
1831
|
+
perPage.set(route, external);
|
|
1832
|
+
for (const url of external) unique.add(url);
|
|
1833
|
+
}
|
|
1834
|
+
const dead = await findDeadExternal([...unique], concurrency, timeoutMs);
|
|
1835
|
+
for (const [route, external] of perPage) {
|
|
1836
|
+
const found = external.filter((url) => dead.has(url)).sort();
|
|
1837
|
+
if (found.length > 0) pages[route].deadExternal = found;
|
|
1838
|
+
}
|
|
1839
|
+
}
|
|
1840
|
+
async function snapshotFromPage(pageUrl, options = {}) {
|
|
1841
|
+
const target = new URL(pageUrl);
|
|
1842
|
+
const site = {
|
|
1843
|
+
origin: options.siteUrl ? new URL(options.siteUrl).origin : target.origin,
|
|
1844
|
+
robotsTxt: null,
|
|
1845
|
+
llmsTxt: null,
|
|
1846
|
+
sitemap: null
|
|
1847
|
+
};
|
|
1848
|
+
const doc = await fetchDoc(target.href, options.timeout);
|
|
1849
|
+
if (doc === null) throw new Error(`${target.href} answered 404 \u2014 nothing to check.`);
|
|
1850
|
+
const route = routeFromUrl(target.href);
|
|
1851
|
+
const page = extractPage(doc.text, route);
|
|
1852
|
+
const landed = !doc.url ? route : originOf2(doc.url) === target.origin ? routeFromUrl(doc.url) : doc.url;
|
|
1853
|
+
page.redirectsTo = landed === route ? null : landed;
|
|
1854
|
+
const pages = { [route]: page };
|
|
1855
|
+
const links = { [route]: extractLinks(doc.text) };
|
|
1856
|
+
const origin = site.origin ?? target.origin;
|
|
1857
|
+
await resolveBrokenLinks(pages, links, origin, async (candidate) => {
|
|
1858
|
+
try {
|
|
1859
|
+
return await fetchDoc(new URL(candidate, target.origin).href, options.timeout) === null;
|
|
1860
|
+
} catch {
|
|
1861
|
+
return false;
|
|
1862
|
+
}
|
|
1863
|
+
});
|
|
1864
|
+
if (options.checkExternal) {
|
|
1865
|
+
await resolveDeadExternal(pages, links, origin, Math.max(1, options.concurrency ?? 5), options.timeout);
|
|
1866
|
+
}
|
|
1867
|
+
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1868
|
+
}
|
|
1758
1869
|
var MAX_NESTED_SITEMAPS = 50;
|
|
1759
1870
|
async function snapshotFromOrigin(origin, options = {}) {
|
|
1760
1871
|
const base = new URL(origin);
|
|
@@ -1843,6 +1954,9 @@ async function snapshotFromOrigin(origin, options = {}) {
|
|
|
1843
1954
|
return false;
|
|
1844
1955
|
}
|
|
1845
1956
|
});
|
|
1957
|
+
if (options.checkExternal) {
|
|
1958
|
+
await resolveDeadExternal(pages, links, base.origin, concurrency, timeout);
|
|
1959
|
+
}
|
|
1846
1960
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1847
1961
|
}
|
|
1848
1962
|
// Annotate the CommonJS export names for ESM import in node:
|
|
@@ -1884,6 +1998,7 @@ async function snapshotFromOrigin(origin, options = {}) {
|
|
|
1884
1998
|
snapshotFromDir,
|
|
1885
1999
|
snapshotFromGitRef,
|
|
1886
2000
|
snapshotFromOrigin,
|
|
2001
|
+
snapshotFromPage,
|
|
1887
2002
|
summarize,
|
|
1888
2003
|
withGuidance
|
|
1889
2004
|
});
|
package/dist/index.d.cts
CHANGED
|
@@ -50,6 +50,12 @@ interface PageFingerprint {
|
|
|
50
50
|
* property the lockfile has to have.
|
|
51
51
|
*/
|
|
52
52
|
brokenLinks?: string[];
|
|
53
|
+
/**
|
|
54
|
+
* External links that answered 404 or 410. Only present when the crawl was
|
|
55
|
+
* asked to check them: it means a request per unique external URL, to hosts
|
|
56
|
+
* you do not control.
|
|
57
|
+
*/
|
|
58
|
+
deadExternal?: string[];
|
|
53
59
|
}
|
|
54
60
|
/** Site-wide signals that live outside any single page. */
|
|
55
61
|
interface SiteFingerprint {
|
|
@@ -139,6 +145,12 @@ interface Config {
|
|
|
139
145
|
* worse than crawling a site you already own.
|
|
140
146
|
*/
|
|
141
147
|
ignoreRobots?: boolean;
|
|
148
|
+
/**
|
|
149
|
+
* Also check links that leave the site. Off by default: it fans out to hosts
|
|
150
|
+
* you do not control, and a lockfile that records their availability turns a
|
|
151
|
+
* third party's bad afternoon into a diff in your repository.
|
|
152
|
+
*/
|
|
153
|
+
checkExternal?: boolean;
|
|
142
154
|
}
|
|
143
155
|
|
|
144
156
|
/**
|
|
@@ -186,12 +198,12 @@ declare function isCrawlable(path: string, rules: {
|
|
|
186
198
|
allow?: string[];
|
|
187
199
|
} | null | undefined): boolean;
|
|
188
200
|
/**
|
|
189
|
-
*
|
|
201
|
+
* Every href on the page worth resolving, deduplicated, in document order.
|
|
202
|
+
* Both internal and external: snapshot.ts is the only layer that knows the
|
|
203
|
+
* page's own URL, so classification happens there.
|
|
190
204
|
*
|
|
191
|
-
*
|
|
192
|
-
*
|
|
193
|
-
* else's problem: an external link checker fans out to hosts you do not
|
|
194
|
-
* control, where a Cloudflare 403 and a rate limit both look like a dead page.
|
|
205
|
+
* Dropped here because they can never be dead: fragments, mailto:, tel:,
|
|
206
|
+
* javascript: and other non-http schemes.
|
|
195
207
|
*/
|
|
196
208
|
declare function extractLinks(html: string): string[];
|
|
197
209
|
/** Parse llms.txt, capturing section headings so truncation is detectable. */
|
|
@@ -305,6 +317,14 @@ declare function sameSurface(a: Snapshot, b: Snapshot): boolean;
|
|
|
305
317
|
declare function shouldIgnore(route: string, patterns?: string[]): boolean;
|
|
306
318
|
/** Build a snapshot from a directory of pre-rendered HTML (next export, dist, out). */
|
|
307
319
|
declare function snapshotFromDir(dir: string, config?: Config): Promise<Snapshot>;
|
|
320
|
+
/**
|
|
321
|
+
* One page, checked properly. No sitemap, no route discovery: the snapshot
|
|
322
|
+
* holds a single page, so every link on it is a candidate and every candidate
|
|
323
|
+
* is confirmed with a real request. That is the opposite trade from a crawl,
|
|
324
|
+
* and the right one here — one page's worth of links is a bounded cost, and
|
|
325
|
+
* "is this page's linking sound" is a question a crawl answers slowly.
|
|
326
|
+
*/
|
|
327
|
+
declare function snapshotFromPage(pageUrl: string, options?: CrawlOptions): Promise<Snapshot>;
|
|
308
328
|
interface CrawlOptions extends Config {
|
|
309
329
|
/** Cap the number of pages fetched. */
|
|
310
330
|
limit?: number;
|
|
@@ -316,4 +336,4 @@ interface CrawlOptions extends Config {
|
|
|
316
336
|
/** Build a snapshot by fetching a live origin, discovering routes via sitemap. */
|
|
317
337
|
declare function snapshotFromOrigin(origin: string, options?: CrawlOptions): Promise<Snapshot>;
|
|
318
338
|
|
|
319
|
-
export { type Aggregate, type AuditMeta, type Config, DEFAULT_AI_AGENTS, type Finding, GUIDANCE, type Guidance, type JsonLdEntity, type PageFingerprint, type Platform, RICH_RESULT_RULES, type Severity, type SiteFingerprint, type Snapshot, aggregate, applyConfig, auditCrossPage, auditHreflang, auditPage, auditSite, auditSnapshot, detectPlatform, diffPage, diffSite, diffSnapshots, extractJsonLd, extractLinks, extractLlmsTxt, extractPage, extractRobotsTxt, extractSitemapUrls, formatAuditHtml, formatAuditMarkdown, formatAuditPretty, formatGithub, formatJson, formatMarkdown, formatPretty, formatSarif, isCrawlable, routeFromFilePath, routeFromUrl, sameSurface, shouldFail, shouldIgnore, snapshotFromDir, snapshotFromGitRef, snapshotFromOrigin, summarize, withGuidance };
|
|
339
|
+
export { type Aggregate, type AuditMeta, type Config, DEFAULT_AI_AGENTS, type Finding, GUIDANCE, type Guidance, type JsonLdEntity, type PageFingerprint, type Platform, RICH_RESULT_RULES, type Severity, type SiteFingerprint, type Snapshot, aggregate, applyConfig, auditCrossPage, auditHreflang, auditPage, auditSite, auditSnapshot, detectPlatform, diffPage, diffSite, diffSnapshots, extractJsonLd, extractLinks, extractLlmsTxt, extractPage, extractRobotsTxt, extractSitemapUrls, formatAuditHtml, formatAuditMarkdown, formatAuditPretty, formatGithub, formatJson, formatMarkdown, formatPretty, formatSarif, isCrawlable, routeFromFilePath, routeFromUrl, sameSurface, shouldFail, shouldIgnore, snapshotFromDir, snapshotFromGitRef, snapshotFromOrigin, snapshotFromPage, summarize, withGuidance };
|
package/dist/index.d.ts
CHANGED
|
@@ -50,6 +50,12 @@ interface PageFingerprint {
|
|
|
50
50
|
* property the lockfile has to have.
|
|
51
51
|
*/
|
|
52
52
|
brokenLinks?: string[];
|
|
53
|
+
/**
|
|
54
|
+
* External links that answered 404 or 410. Only present when the crawl was
|
|
55
|
+
* asked to check them: it means a request per unique external URL, to hosts
|
|
56
|
+
* you do not control.
|
|
57
|
+
*/
|
|
58
|
+
deadExternal?: string[];
|
|
53
59
|
}
|
|
54
60
|
/** Site-wide signals that live outside any single page. */
|
|
55
61
|
interface SiteFingerprint {
|
|
@@ -139,6 +145,12 @@ interface Config {
|
|
|
139
145
|
* worse than crawling a site you already own.
|
|
140
146
|
*/
|
|
141
147
|
ignoreRobots?: boolean;
|
|
148
|
+
/**
|
|
149
|
+
* Also check links that leave the site. Off by default: it fans out to hosts
|
|
150
|
+
* you do not control, and a lockfile that records their availability turns a
|
|
151
|
+
* third party's bad afternoon into a diff in your repository.
|
|
152
|
+
*/
|
|
153
|
+
checkExternal?: boolean;
|
|
142
154
|
}
|
|
143
155
|
|
|
144
156
|
/**
|
|
@@ -186,12 +198,12 @@ declare function isCrawlable(path: string, rules: {
|
|
|
186
198
|
allow?: string[];
|
|
187
199
|
} | null | undefined): boolean;
|
|
188
200
|
/**
|
|
189
|
-
*
|
|
201
|
+
* Every href on the page worth resolving, deduplicated, in document order.
|
|
202
|
+
* Both internal and external: snapshot.ts is the only layer that knows the
|
|
203
|
+
* page's own URL, so classification happens there.
|
|
190
204
|
*
|
|
191
|
-
*
|
|
192
|
-
*
|
|
193
|
-
* else's problem: an external link checker fans out to hosts you do not
|
|
194
|
-
* control, where a Cloudflare 403 and a rate limit both look like a dead page.
|
|
205
|
+
* Dropped here because they can never be dead: fragments, mailto:, tel:,
|
|
206
|
+
* javascript: and other non-http schemes.
|
|
195
207
|
*/
|
|
196
208
|
declare function extractLinks(html: string): string[];
|
|
197
209
|
/** Parse llms.txt, capturing section headings so truncation is detectable. */
|
|
@@ -305,6 +317,14 @@ declare function sameSurface(a: Snapshot, b: Snapshot): boolean;
|
|
|
305
317
|
declare function shouldIgnore(route: string, patterns?: string[]): boolean;
|
|
306
318
|
/** Build a snapshot from a directory of pre-rendered HTML (next export, dist, out). */
|
|
307
319
|
declare function snapshotFromDir(dir: string, config?: Config): Promise<Snapshot>;
|
|
320
|
+
/**
|
|
321
|
+
* One page, checked properly. No sitemap, no route discovery: the snapshot
|
|
322
|
+
* holds a single page, so every link on it is a candidate and every candidate
|
|
323
|
+
* is confirmed with a real request. That is the opposite trade from a crawl,
|
|
324
|
+
* and the right one here — one page's worth of links is a bounded cost, and
|
|
325
|
+
* "is this page's linking sound" is a question a crawl answers slowly.
|
|
326
|
+
*/
|
|
327
|
+
declare function snapshotFromPage(pageUrl: string, options?: CrawlOptions): Promise<Snapshot>;
|
|
308
328
|
interface CrawlOptions extends Config {
|
|
309
329
|
/** Cap the number of pages fetched. */
|
|
310
330
|
limit?: number;
|
|
@@ -316,4 +336,4 @@ interface CrawlOptions extends Config {
|
|
|
316
336
|
/** Build a snapshot by fetching a live origin, discovering routes via sitemap. */
|
|
317
337
|
declare function snapshotFromOrigin(origin: string, options?: CrawlOptions): Promise<Snapshot>;
|
|
318
338
|
|
|
319
|
-
export { type Aggregate, type AuditMeta, type Config, DEFAULT_AI_AGENTS, type Finding, GUIDANCE, type Guidance, type JsonLdEntity, type PageFingerprint, type Platform, RICH_RESULT_RULES, type Severity, type SiteFingerprint, type Snapshot, aggregate, applyConfig, auditCrossPage, auditHreflang, auditPage, auditSite, auditSnapshot, detectPlatform, diffPage, diffSite, diffSnapshots, extractJsonLd, extractLinks, extractLlmsTxt, extractPage, extractRobotsTxt, extractSitemapUrls, formatAuditHtml, formatAuditMarkdown, formatAuditPretty, formatGithub, formatJson, formatMarkdown, formatPretty, formatSarif, isCrawlable, routeFromFilePath, routeFromUrl, sameSurface, shouldFail, shouldIgnore, snapshotFromDir, snapshotFromGitRef, snapshotFromOrigin, summarize, withGuidance };
|
|
339
|
+
export { type Aggregate, type AuditMeta, type Config, DEFAULT_AI_AGENTS, type Finding, GUIDANCE, type Guidance, type JsonLdEntity, type PageFingerprint, type Platform, RICH_RESULT_RULES, type Severity, type SiteFingerprint, type Snapshot, aggregate, applyConfig, auditCrossPage, auditHreflang, auditPage, auditSite, auditSnapshot, detectPlatform, diffPage, diffSite, diffSnapshots, extractJsonLd, extractLinks, extractLlmsTxt, extractPage, extractRobotsTxt, extractSitemapUrls, formatAuditHtml, formatAuditMarkdown, formatAuditPretty, formatGithub, formatJson, formatMarkdown, formatPretty, formatSarif, isCrawlable, routeFromFilePath, routeFromUrl, sameSurface, shouldFail, shouldIgnore, snapshotFromDir, snapshotFromGitRef, snapshotFromOrigin, snapshotFromPage, summarize, withGuidance };
|
package/dist/index.js
CHANGED
|
@@ -223,13 +223,24 @@ function auditPage(page, config = {}) {
|
|
|
223
223
|
return findings;
|
|
224
224
|
}
|
|
225
225
|
function auditLinks(page) {
|
|
226
|
-
return
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
226
|
+
return [
|
|
227
|
+
...(page.brokenLinks ?? []).map((target) => ({
|
|
228
|
+
code: "link.broken",
|
|
229
|
+
severity: "error",
|
|
230
|
+
route: page.route,
|
|
231
|
+
message: `Links to ${target}, which does not exist.`,
|
|
232
|
+
after: target
|
|
233
|
+
})),
|
|
234
|
+
// A warning, not an error: the page it points at belongs to somebody else,
|
|
235
|
+
// who can delete it on a Tuesday without asking you.
|
|
236
|
+
...(page.deadExternal ?? []).map((target) => ({
|
|
237
|
+
code: "link.external.dead",
|
|
238
|
+
severity: "warn",
|
|
239
|
+
route: page.route,
|
|
240
|
+
message: `Links out to ${target}, which answers 404.`,
|
|
241
|
+
after: target
|
|
242
|
+
}))
|
|
243
|
+
];
|
|
233
244
|
}
|
|
234
245
|
function auditSite(snapshot) {
|
|
235
246
|
const findings = [];
|
|
@@ -600,6 +611,12 @@ function diffPage(before, after) {
|
|
|
600
611
|
push("link.broken.removed", "info", `Link to ${target} was fixed or removed.`, {
|
|
601
612
|
before: target
|
|
602
613
|
});
|
|
614
|
+
const deadBefore = before.deadExternal ?? [];
|
|
615
|
+
const deadAfter = after.deadExternal ?? [];
|
|
616
|
+
for (const target of deadAfter.filter((t) => !deadBefore.includes(t)))
|
|
617
|
+
push("link.external.dead.added", "warn", `Links out to ${target}, which answers 404.`, {
|
|
618
|
+
after: target
|
|
619
|
+
});
|
|
603
620
|
const droppedHreflang = Object.keys(before.hreflang).filter((k) => !(k in after.hreflang));
|
|
604
621
|
if (droppedHreflang.length > 0)
|
|
605
622
|
push("hreflang.removed", "warn", "hreflang alternates were removed.", {
|
|
@@ -888,8 +905,8 @@ function extractLinks(html) {
|
|
|
888
905
|
const hrefs = /* @__PURE__ */ new Set();
|
|
889
906
|
for (const anchor of root.querySelectorAll("a[href]")) {
|
|
890
907
|
const href = anchor.getAttribute("href")?.trim();
|
|
891
|
-
if (!href) continue;
|
|
892
|
-
if (
|
|
908
|
+
if (!href || href.startsWith("#")) continue;
|
|
909
|
+
if (/^[a-z][a-z0-9+.-]*:/i.test(href) && !/^https?:/i.test(href)) continue;
|
|
893
910
|
hrefs.add(href);
|
|
894
911
|
}
|
|
895
912
|
return [...hrefs];
|
|
@@ -947,6 +964,10 @@ var GUIDANCE = {
|
|
|
947
964
|
nextjs: "A `<Link href>` pointing at a route that no longer exists. TypeScript will not catch it \u2014 typed routes are opt-in via `experimental.typedRoutes`."
|
|
948
965
|
}
|
|
949
966
|
},
|
|
967
|
+
"link.external.dead": {
|
|
968
|
+
why: "An outbound link to a page that is gone sends readers to a 404 and spends the trust the link was passing on nothing. Unlike an internal link, the page is not yours to restore.",
|
|
969
|
+
fix: "Point the link at the current URL, at an archived copy, or remove it. A link that has been dead a while is usually a citation worth replacing rather than deleting."
|
|
970
|
+
},
|
|
950
971
|
"sitemap.dead": {
|
|
951
972
|
why: "A sitemap is a list of URLs you are asking to have crawled. Entries that 404 spend crawl budget on nothing and lower the trust placed in the rest of the file.",
|
|
952
973
|
fix: "Remove the URL from the sitemap, or restore the page. If it moved, redirect it and list the destination instead.",
|
|
@@ -1582,6 +1603,7 @@ async function walkHtml(dir, acc = []) {
|
|
|
1582
1603
|
}
|
|
1583
1604
|
return acc;
|
|
1584
1605
|
}
|
|
1606
|
+
var USER_AGENT = "pagetrace (+https://npmjs.com/package/pagetrace)";
|
|
1585
1607
|
var DEFAULT_TIMEOUT_MS = 15e3;
|
|
1586
1608
|
var MAX_ATTEMPTS = 3;
|
|
1587
1609
|
var RETRY_BASE_MS = 300;
|
|
@@ -1597,7 +1619,7 @@ async function fetchDoc(url, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
|
1597
1619
|
try {
|
|
1598
1620
|
response = await fetch(url, {
|
|
1599
1621
|
signal: AbortSignal.timeout(timeoutMs),
|
|
1600
|
-
headers: { "user-agent":
|
|
1622
|
+
headers: { "user-agent": USER_AGENT }
|
|
1601
1623
|
});
|
|
1602
1624
|
} catch (cause) {
|
|
1603
1625
|
lastError = new Error(`Could not reach ${url}: ${cause.message}`, { cause });
|
|
@@ -1642,7 +1664,9 @@ async function snapshotFromDir(dir, config = {}) {
|
|
|
1642
1664
|
if (robots !== null) site.robotsTxt = extractRobotsTxt(robots, agents);
|
|
1643
1665
|
const llms = await readFile(join(dir, "llms.txt"), "utf8").catch(() => null);
|
|
1644
1666
|
if (llms !== null) site.llmsTxt = extractLlmsTxt(llms);
|
|
1645
|
-
|
|
1667
|
+
const origin = site.origin ?? "https://pagetrace.invalid";
|
|
1668
|
+
await resolveBrokenLinks(pages, links, origin);
|
|
1669
|
+
if (config.checkExternal) await resolveDeadExternal(pages, links, origin, 5);
|
|
1646
1670
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1647
1671
|
}
|
|
1648
1672
|
var ASSET_PATH = /\.(?!html?$)[a-z0-9]+$/i;
|
|
@@ -1681,6 +1705,92 @@ async function resolveBrokenLinks(pages, links, origin, verify) {
|
|
|
1681
1705
|
if (confirmed.length > 0) pages[route].brokenLinks = confirmed.sort();
|
|
1682
1706
|
}
|
|
1683
1707
|
}
|
|
1708
|
+
function externalUrl(href, from, origin) {
|
|
1709
|
+
try {
|
|
1710
|
+
const url = new URL(href, `${origin}${from === "/" ? "" : from}/`);
|
|
1711
|
+
if (url.protocol !== "http:" && url.protocol !== "https:") return null;
|
|
1712
|
+
if (url.origin === origin) return null;
|
|
1713
|
+
url.hash = "";
|
|
1714
|
+
return url.href;
|
|
1715
|
+
} catch {
|
|
1716
|
+
return null;
|
|
1717
|
+
}
|
|
1718
|
+
}
|
|
1719
|
+
var MAX_EXTERNAL_CHECKS = 200;
|
|
1720
|
+
async function findDeadExternal(urls, concurrency, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
1721
|
+
const dead = /* @__PURE__ */ new Set();
|
|
1722
|
+
const queue = urls.slice(0, MAX_EXTERNAL_CHECKS);
|
|
1723
|
+
const request = (url, method) => fetch(url, {
|
|
1724
|
+
method,
|
|
1725
|
+
redirect: "follow",
|
|
1726
|
+
signal: AbortSignal.timeout(timeoutMs),
|
|
1727
|
+
headers: { "user-agent": USER_AGENT }
|
|
1728
|
+
});
|
|
1729
|
+
await Promise.all(
|
|
1730
|
+
Array.from({ length: Math.min(concurrency, queue.length) }, async () => {
|
|
1731
|
+
while (queue.length > 0) {
|
|
1732
|
+
const url = queue.shift();
|
|
1733
|
+
try {
|
|
1734
|
+
let response = await request(url, "HEAD");
|
|
1735
|
+
if (response.status === 405 || response.status === 501) {
|
|
1736
|
+
response = await request(url, "GET");
|
|
1737
|
+
}
|
|
1738
|
+
if (response.status === 404 || response.status === 410) dead.add(url);
|
|
1739
|
+
} catch {
|
|
1740
|
+
}
|
|
1741
|
+
}
|
|
1742
|
+
})
|
|
1743
|
+
);
|
|
1744
|
+
return dead;
|
|
1745
|
+
}
|
|
1746
|
+
async function resolveDeadExternal(pages, links, origin, concurrency, timeoutMs) {
|
|
1747
|
+
const perPage = /* @__PURE__ */ new Map();
|
|
1748
|
+
const unique = /* @__PURE__ */ new Set();
|
|
1749
|
+
for (const [route, hrefs] of Object.entries(links)) {
|
|
1750
|
+
const external = [
|
|
1751
|
+
...new Set(
|
|
1752
|
+
hrefs.map((href) => externalUrl(href, route, origin)).filter((url) => url !== null)
|
|
1753
|
+
)
|
|
1754
|
+
];
|
|
1755
|
+
if (external.length === 0) continue;
|
|
1756
|
+
perPage.set(route, external);
|
|
1757
|
+
for (const url of external) unique.add(url);
|
|
1758
|
+
}
|
|
1759
|
+
const dead = await findDeadExternal([...unique], concurrency, timeoutMs);
|
|
1760
|
+
for (const [route, external] of perPage) {
|
|
1761
|
+
const found = external.filter((url) => dead.has(url)).sort();
|
|
1762
|
+
if (found.length > 0) pages[route].deadExternal = found;
|
|
1763
|
+
}
|
|
1764
|
+
}
|
|
1765
|
+
async function snapshotFromPage(pageUrl, options = {}) {
|
|
1766
|
+
const target = new URL(pageUrl);
|
|
1767
|
+
const site = {
|
|
1768
|
+
origin: options.siteUrl ? new URL(options.siteUrl).origin : target.origin,
|
|
1769
|
+
robotsTxt: null,
|
|
1770
|
+
llmsTxt: null,
|
|
1771
|
+
sitemap: null
|
|
1772
|
+
};
|
|
1773
|
+
const doc = await fetchDoc(target.href, options.timeout);
|
|
1774
|
+
if (doc === null) throw new Error(`${target.href} answered 404 \u2014 nothing to check.`);
|
|
1775
|
+
const route = routeFromUrl(target.href);
|
|
1776
|
+
const page = extractPage(doc.text, route);
|
|
1777
|
+
const landed = !doc.url ? route : originOf2(doc.url) === target.origin ? routeFromUrl(doc.url) : doc.url;
|
|
1778
|
+
page.redirectsTo = landed === route ? null : landed;
|
|
1779
|
+
const pages = { [route]: page };
|
|
1780
|
+
const links = { [route]: extractLinks(doc.text) };
|
|
1781
|
+
const origin = site.origin ?? target.origin;
|
|
1782
|
+
await resolveBrokenLinks(pages, links, origin, async (candidate) => {
|
|
1783
|
+
try {
|
|
1784
|
+
return await fetchDoc(new URL(candidate, target.origin).href, options.timeout) === null;
|
|
1785
|
+
} catch {
|
|
1786
|
+
return false;
|
|
1787
|
+
}
|
|
1788
|
+
});
|
|
1789
|
+
if (options.checkExternal) {
|
|
1790
|
+
await resolveDeadExternal(pages, links, origin, Math.max(1, options.concurrency ?? 5), options.timeout);
|
|
1791
|
+
}
|
|
1792
|
+
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1793
|
+
}
|
|
1684
1794
|
var MAX_NESTED_SITEMAPS = 50;
|
|
1685
1795
|
async function snapshotFromOrigin(origin, options = {}) {
|
|
1686
1796
|
const base = new URL(origin);
|
|
@@ -1769,6 +1879,9 @@ async function snapshotFromOrigin(origin, options = {}) {
|
|
|
1769
1879
|
return false;
|
|
1770
1880
|
}
|
|
1771
1881
|
});
|
|
1882
|
+
if (options.checkExternal) {
|
|
1883
|
+
await resolveDeadExternal(pages, links, base.origin, concurrency, timeout);
|
|
1884
|
+
}
|
|
1772
1885
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1773
1886
|
}
|
|
1774
1887
|
export {
|
|
@@ -1809,6 +1922,7 @@ export {
|
|
|
1809
1922
|
snapshotFromDir,
|
|
1810
1923
|
snapshotFromGitRef,
|
|
1811
1924
|
snapshotFromOrigin,
|
|
1925
|
+
snapshotFromPage,
|
|
1812
1926
|
summarize,
|
|
1813
1927
|
withGuidance
|
|
1814
1928
|
};
|