pagetrace 0.12.0 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +28 -0
- package/README.md +37 -2
- package/dist/cli.cjs +122 -26
- package/dist/cli.js +123 -27
- package/dist/index.cjs +74 -15
- package/dist/index.d.cts +14 -1
- package/dist/index.d.ts +14 -1
- package/dist/index.js +74 -16
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,32 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
|
6
6
|
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
7
|
While the version is below 1.0.0, breaking changes ship in a minor release.
|
|
8
8
|
|
|
9
|
+
## [0.14.0] - 2026-09-08
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- `--verify-all` also checks links to assets — PDFs, images, archives. They are
|
|
14
|
+
never crawled as pages, so they were skipped entirely: a link to a deleted
|
|
15
|
+
whitepaper reported nothing. A `--dir` run checks them against the filesystem;
|
|
16
|
+
a crawl spends a request each, which is why it is opt-in.
|
|
17
|
+
|
|
18
|
+
### Changed
|
|
19
|
+
|
|
20
|
+
- Link verification now runs in parallel at `--concurrency`, and the cap rose
|
|
21
|
+
from 100 to 1000 unique targets. Hitting the cap now says so on stderr rather
|
|
22
|
+
than silently under-reporting, which read as a clean site.
|
|
23
|
+
|
|
24
|
+
## [0.13.0] - 2026-09-08
|
|
25
|
+
|
|
26
|
+
### Added
|
|
27
|
+
|
|
28
|
+
- `pagetrace page <url>` checks a single page: every link on it confirmed with a
|
|
29
|
+
real request, external ones included by default, plus that page's own rules.
|
|
30
|
+
A crawl trusts its route set and only verifies what is missing from it; with
|
|
31
|
+
one page there is no route set to trust, so everything is checked. Passing a
|
|
32
|
+
deep URL to `--url` had crawled the whole site from its origin and ignored the
|
|
33
|
+
path, which is not what it looked like it did.
|
|
34
|
+
|
|
9
35
|
## [0.12.0] - 2026-09-08
|
|
10
36
|
|
|
11
37
|
### Added
|
|
@@ -359,6 +385,8 @@ Initial release. `snapshot`, `check` and `audit` commands; filesystem and HTTP
|
|
|
359
385
|
crawling; diff classified by transition; absolute, cross-page and hreflang audit
|
|
360
386
|
rules; pretty, JSON, markdown, GitHub and HTML reporters.
|
|
361
387
|
|
|
388
|
+
[0.14.0]: https://github.com/shyamexe/pagetrace/compare/v0.13.0...v0.14.0
|
|
389
|
+
[0.13.0]: https://github.com/shyamexe/pagetrace/compare/v0.12.0...v0.13.0
|
|
362
390
|
[0.12.0]: https://github.com/shyamexe/pagetrace/compare/v0.11.0...v0.12.0
|
|
363
391
|
[0.11.0]: https://github.com/shyamexe/pagetrace/compare/v0.10.0...v0.11.0
|
|
364
392
|
[0.10.0]: https://github.com/shyamexe/pagetrace/compare/v0.9.1...v0.10.0
|
package/README.md
CHANGED
|
@@ -19,6 +19,33 @@ npm install -D pagetrace
|
|
|
19
19
|
|
|
20
20
|
Requires Node 20.19 or newer. No native modules, three small dependencies.
|
|
21
21
|
|
|
22
|
+
## Commands
|
|
23
|
+
|
|
24
|
+
| Command | Answers | Crawls | Exit 1 when |
|
|
25
|
+
| --- | --- | --- | --- |
|
|
26
|
+
| `init` | "get me set up" | once, to write the first lockfile | never |
|
|
27
|
+
| `snapshot` | "record what the site looks like now" | whole site | never |
|
|
28
|
+
| `check` | "what did this deploy change?" | whole site | findings at or above `--fail-on` (default `error`) |
|
|
29
|
+
| `audit` | "what is wrong with this site?" | whole site | `--fail-on` (default `never`) |
|
|
30
|
+
| `links` | "are any links dead?" | whole site, links only in the report | any broken link |
|
|
31
|
+
| `page <url>` | "is this one page sound?" | that URL alone | `--fail-on` (default `error`) |
|
|
32
|
+
| `update` | "am I on the latest pagetrace?" | nothing | never |
|
|
33
|
+
|
|
34
|
+
Every crawling command takes `--dir <build>` or `--url <origin>`, plus:
|
|
35
|
+
|
|
36
|
+
| Flag | Default | Effect |
|
|
37
|
+
| --- | --- | --- |
|
|
38
|
+
| `--limit <n>` | 200 | Stop after this many pages |
|
|
39
|
+
| `--concurrency <n>` | 5 | Parallel requests |
|
|
40
|
+
| `--external` | off (on for `page`) | Also check links that leave the site |
|
|
41
|
+
| `--verify-all` | off | Also check links to assets — PDFs, images, archives |
|
|
42
|
+
| `--ignore-robots` | off | Crawl paths `robots.txt` disallows |
|
|
43
|
+
| `--fail-on <severity>` | varies | `error`, `warn`, `info` or `never` |
|
|
44
|
+
| `--format <format>` | `pretty` | `pretty`, `json`, `markdown`; `github` and `sarif` on `check` |
|
|
45
|
+
| `--config <file>` | `pagetrace.config.json` | Config file |
|
|
46
|
+
|
|
47
|
+
Exit codes are the same everywhere: `0` clean, `1` findings at or above `--fail-on`, `2` the run itself failed — bad flags, an unreadable build, an unreachable origin. CI can tell "the site regressed" from "the tool broke".
|
|
48
|
+
|
|
22
49
|
## Use
|
|
23
50
|
|
|
24
51
|
Set up a config file and the first baseline in one step:
|
|
@@ -56,7 +83,7 @@ npx pagetrace check --dir ./out
|
|
|
56
83
|
5 error, 9 warning, 2 info
|
|
57
84
|
```
|
|
58
85
|
|
|
59
|
-
|
|
86
|
+
`--fail-on` rejects an unrecognised value rather than quietly letting the build pass.
|
|
60
87
|
|
|
61
88
|
Note that `check` runs the absolute rules as well as the diff, so it can fail on a problem your build did not introduce. Use `--no-audit` for a pure regression gate.
|
|
62
89
|
|
|
@@ -166,6 +193,14 @@ pagetrace update # install it globally
|
|
|
166
193
|
| `og.removed` / `hreflang.removed` | warn | Social or i18n tags dropped |
|
|
167
194
|
| `title.changed` | info | Ordinary copy edit |
|
|
168
195
|
|
|
196
|
+
One page, checked on its own:
|
|
197
|
+
|
|
198
|
+
```bash
|
|
199
|
+
npx pagetrace page https://example.com/blog/my-post
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
No sitemap, no crawl: it fetches that URL and confirms every link on it with a real request, external links included by default (`--no-external` to skip them). Site-wide rules are left to `audit` — a single-page check never looks at robots.txt, so it does not get to say whether one exists.
|
|
203
|
+
|
|
169
204
|
Broken links have a command of their own, when that is the only question you have:
|
|
170
205
|
|
|
171
206
|
```bash
|
|
@@ -185,7 +220,7 @@ Only `404` and `410` count as dead. A `403` from a bot wall, a `429`, a timeout
|
|
|
185
220
|
|
|
186
221
|
Think twice before putting `--external` in `check`. A third party's bad afternoon becomes a diff in your repository and a red build you cannot fix.
|
|
187
222
|
|
|
188
|
-
Internal links are checked too. Only the broken ones are stored, so a site's navigation never lands in the lockfile: `link.broken` for a link that is already dead, `link.broken.added` for one this build broke. A `--dir` crawl is authoritative — the build directory is the whole site — while a crawl confirms each candidate with a real request first, because a sitemap routinely omits pages that are live. External links are checked only with `--external`, and only a 404 or 410 counts.
|
|
223
|
+
Internal links are checked too. Only the broken ones are stored, so a site's navigation never lands in the lockfile: `link.broken` for a link that is already dead, `link.broken.added` for one this build broke. A `--dir` crawl is authoritative — the build directory is the whole site — while a crawl confirms each candidate with a real request first, because a sitemap routinely omits pages that are live. External links are checked only with `--external`, and only a 404 or 410 counts. Links to assets — PDFs, images, archives — are skipped unless `--verify-all`, since each one costs a request on a crawl (a `--dir` run checks them against the filesystem instead).
|
|
189
224
|
|
|
190
225
|
Redirects are recorded from the response itself, so they cost no extra requests. A redirect that only adds or drops a trailing slash is server configuration rather than drift and is not reported. `canonical.redirects` is only raised when the canonical's target was actually crawled, so a `--limit` run cannot invent it.
|
|
191
226
|
|
package/dist/cli.cjs
CHANGED
|
@@ -1698,12 +1698,20 @@ async function snapshotFromDir(dir, config = {}) {
|
|
|
1698
1698
|
const llms = await (0, import_promises.readFile)((0, import_node_path.join)(dir, "llms.txt"), "utf8").catch(() => null);
|
|
1699
1699
|
if (llms !== null) site.llmsTxt = extractLlmsTxt(llms);
|
|
1700
1700
|
const origin = site.origin ?? "https://pagetrace.invalid";
|
|
1701
|
-
await resolveBrokenLinks(pages, links, origin
|
|
1701
|
+
await resolveBrokenLinks(pages, links, origin, {
|
|
1702
|
+
includeAssets: config.verifyAll,
|
|
1703
|
+
// The HTML route set is authoritative here, so a missing page is simply
|
|
1704
|
+
// broken. An asset is a file this crawl never walked, so it gets a stat.
|
|
1705
|
+
verify: config.verifyAll ? async (target) => ASSET_PATH.test(target) ? !await (0, import_promises.access)((0, import_node_path.join)(dir, target)).then(
|
|
1706
|
+
() => true,
|
|
1707
|
+
() => false
|
|
1708
|
+
) : true : void 0
|
|
1709
|
+
});
|
|
1702
1710
|
if (config.checkExternal) await resolveDeadExternal(pages, links, origin, 5);
|
|
1703
1711
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1704
1712
|
}
|
|
1705
1713
|
var ASSET_PATH = /\.(?!html?$)[a-z0-9]+$/i;
|
|
1706
|
-
function linkTarget(href, from, origin) {
|
|
1714
|
+
function linkTarget(href, from, origin, includeAssets = false) {
|
|
1707
1715
|
let url;
|
|
1708
1716
|
try {
|
|
1709
1717
|
url = new URL(href, `${origin}${from === "/" ? "" : from}/`);
|
|
@@ -1711,27 +1719,39 @@ function linkTarget(href, from, origin) {
|
|
|
1711
1719
|
return null;
|
|
1712
1720
|
}
|
|
1713
1721
|
if (url.origin !== origin) return null;
|
|
1714
|
-
if (ASSET_PATH.test(url.pathname)) return null;
|
|
1715
|
-
return routeFromUrl(url.href);
|
|
1722
|
+
if (!includeAssets && ASSET_PATH.test(url.pathname)) return null;
|
|
1723
|
+
return ASSET_PATH.test(url.pathname) ? url.pathname : routeFromUrl(url.href);
|
|
1716
1724
|
}
|
|
1717
|
-
var MAX_LINK_CHECKS =
|
|
1718
|
-
async function resolveBrokenLinks(pages, links, origin,
|
|
1725
|
+
var MAX_LINK_CHECKS = 1e3;
|
|
1726
|
+
async function resolveBrokenLinks(pages, links, origin, options = {}) {
|
|
1727
|
+
const { includeAssets = false, verify, concurrency = 5 } = options;
|
|
1719
1728
|
const known = new Set(Object.keys(pages));
|
|
1720
1729
|
const candidates = /* @__PURE__ */ new Map();
|
|
1721
1730
|
for (const [route, hrefs] of Object.entries(links)) {
|
|
1722
1731
|
const missing = [
|
|
1723
1732
|
...new Set(
|
|
1724
|
-
hrefs.map((href) => linkTarget(href, route, origin)).filter((target) => target !== null && !known.has(target))
|
|
1733
|
+
hrefs.map((href) => linkTarget(href, route, origin, includeAssets)).filter((target) => target !== null && !known.has(target))
|
|
1725
1734
|
)
|
|
1726
1735
|
];
|
|
1727
1736
|
if (missing.length > 0) candidates.set(route, missing);
|
|
1728
1737
|
}
|
|
1729
1738
|
const broken = /* @__PURE__ */ new Set();
|
|
1730
1739
|
if (verify) {
|
|
1731
|
-
const
|
|
1732
|
-
|
|
1733
|
-
|
|
1740
|
+
const unique = [...new Set([...candidates.values()].flat())];
|
|
1741
|
+
const queue = unique.slice(0, MAX_LINK_CHECKS);
|
|
1742
|
+
if (unique.length > queue.length) {
|
|
1743
|
+
console.error(
|
|
1744
|
+
`pagetrace: ${unique.length} link targets to confirm, checking the first ${MAX_LINK_CHECKS}.`
|
|
1745
|
+
);
|
|
1734
1746
|
}
|
|
1747
|
+
await Promise.all(
|
|
1748
|
+
Array.from({ length: Math.min(Math.max(1, concurrency), queue.length) }, async () => {
|
|
1749
|
+
while (queue.length > 0) {
|
|
1750
|
+
const target = queue.shift();
|
|
1751
|
+
if (await verify(target)) broken.add(target);
|
|
1752
|
+
}
|
|
1753
|
+
})
|
|
1754
|
+
);
|
|
1735
1755
|
}
|
|
1736
1756
|
for (const [route, missing] of candidates) {
|
|
1737
1757
|
const confirmed = verify ? missing.filter((target) => broken.has(target)) : missing;
|
|
@@ -1795,6 +1815,39 @@ async function resolveDeadExternal(pages, links, origin, concurrency, timeoutMs)
|
|
|
1795
1815
|
if (found.length > 0) pages[route].deadExternal = found;
|
|
1796
1816
|
}
|
|
1797
1817
|
}
|
|
1818
|
+
async function snapshotFromPage(pageUrl, options = {}) {
|
|
1819
|
+
const target = new URL(pageUrl);
|
|
1820
|
+
const site = {
|
|
1821
|
+
origin: options.siteUrl ? new URL(options.siteUrl).origin : target.origin,
|
|
1822
|
+
robotsTxt: null,
|
|
1823
|
+
llmsTxt: null,
|
|
1824
|
+
sitemap: null
|
|
1825
|
+
};
|
|
1826
|
+
const doc = await fetchDoc(target.href, options.timeout);
|
|
1827
|
+
if (doc === null) throw new Error(`${target.href} answered 404 \u2014 nothing to check.`);
|
|
1828
|
+
const route = routeFromUrl(target.href);
|
|
1829
|
+
const page = extractPage(doc.text, route);
|
|
1830
|
+
const landed = !doc.url ? route : originOf2(doc.url) === target.origin ? routeFromUrl(doc.url) : doc.url;
|
|
1831
|
+
page.redirectsTo = landed === route ? null : landed;
|
|
1832
|
+
const pages = { [route]: page };
|
|
1833
|
+
const links = { [route]: extractLinks(doc.text) };
|
|
1834
|
+
const origin = site.origin ?? target.origin;
|
|
1835
|
+
await resolveBrokenLinks(pages, links, origin, {
|
|
1836
|
+
includeAssets: options.verifyAll,
|
|
1837
|
+
concurrency: options.concurrency,
|
|
1838
|
+
verify: async (candidate) => {
|
|
1839
|
+
try {
|
|
1840
|
+
return await fetchDoc(new URL(candidate, target.origin).href, options.timeout) === null;
|
|
1841
|
+
} catch {
|
|
1842
|
+
return false;
|
|
1843
|
+
}
|
|
1844
|
+
}
|
|
1845
|
+
});
|
|
1846
|
+
if (options.checkExternal) {
|
|
1847
|
+
await resolveDeadExternal(pages, links, origin, Math.max(1, options.concurrency ?? 5), options.timeout);
|
|
1848
|
+
}
|
|
1849
|
+
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1850
|
+
}
|
|
1798
1851
|
var MAX_NESTED_SITEMAPS = 50;
|
|
1799
1852
|
async function snapshotFromOrigin(origin, options = {}) {
|
|
1800
1853
|
const base = new URL(origin);
|
|
@@ -1876,11 +1929,15 @@ async function snapshotFromOrigin(origin, options = {}) {
|
|
|
1876
1929
|
}
|
|
1877
1930
|
})
|
|
1878
1931
|
);
|
|
1879
|
-
await resolveBrokenLinks(pages, links, base.origin,
|
|
1880
|
-
|
|
1881
|
-
|
|
1882
|
-
|
|
1883
|
-
|
|
1932
|
+
await resolveBrokenLinks(pages, links, base.origin, {
|
|
1933
|
+
includeAssets: options.verifyAll,
|
|
1934
|
+
concurrency,
|
|
1935
|
+
verify: async (route) => {
|
|
1936
|
+
try {
|
|
1937
|
+
return await fetchDoc(new URL(route, base).href, timeout) === null;
|
|
1938
|
+
} catch {
|
|
1939
|
+
return false;
|
|
1940
|
+
}
|
|
1884
1941
|
}
|
|
1885
1942
|
});
|
|
1886
1943
|
if (options.checkExternal) {
|
|
@@ -1939,14 +1996,16 @@ async function loadConfig(path = DEFAULT_CONFIG) {
|
|
|
1939
1996
|
}
|
|
1940
1997
|
async function build(flags, config) {
|
|
1941
1998
|
const checkExternal = flags.external ?? config.checkExternal;
|
|
1942
|
-
|
|
1999
|
+
const verifyAll = flags.verifyAll ?? config.verifyAll;
|
|
2000
|
+
if (flags.dir) return snapshotFromDir(flags.dir, { ...config, checkExternal, verifyAll });
|
|
1943
2001
|
if (flags.url)
|
|
1944
2002
|
return snapshotFromOrigin(flags.url, {
|
|
1945
2003
|
...config,
|
|
1946
2004
|
limit: flags.limit,
|
|
1947
2005
|
concurrency: flags.concurrency,
|
|
1948
2006
|
ignoreRobots: flags.ignoreRobots ?? config.ignoreRobots,
|
|
1949
|
-
checkExternal
|
|
2007
|
+
checkExternal,
|
|
2008
|
+
verifyAll
|
|
1950
2009
|
});
|
|
1951
2010
|
throw new Error("Provide a source: --dir <build directory> or --url <origin>.");
|
|
1952
2011
|
}
|
|
@@ -1965,7 +2024,7 @@ function render(findings, format) {
|
|
|
1965
2024
|
}
|
|
1966
2025
|
}
|
|
1967
2026
|
var cli = (0, import_cac.cac)("pagetrace");
|
|
1968
|
-
cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
2027
|
+
cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--verify-all", "Also check links to assets (PDFs, images, archives)").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
1969
2028
|
const config = await loadConfig(flags.config);
|
|
1970
2029
|
const snapshot = await build(flags, config);
|
|
1971
2030
|
const written = await writeLockfile(flags.out, snapshot);
|
|
@@ -1974,7 +2033,7 @@ cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").opti
|
|
|
1974
2033
|
written ? import_picocolors2.default.green(`Wrote ${flags.out} \u2014 ${count} page${count === 1 ? "" : "s"}.`) : import_picocolors2.default.dim(`${flags.out} is already up to date \u2014 ${count} page${count === 1 ? "" : "s"}.`)
|
|
1975
2034
|
);
|
|
1976
2035
|
});
|
|
1977
|
-
cli.command("check", "Compare the current surface against the lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--lockfile <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | github | sarif", { default: "pretty" }).option("--fail-on <severity>", "error | warn | info", { default: "error" }).option("--audit", "Also run absolute rules, not just the diff", { default: true }).option("--update", "Write the new state to the lockfile after reporting").option("--baseline-branch <ref>", "Read the baseline lockfile from a git ref instead of disk").action(async (flags) => {
|
|
2036
|
+
cli.command("check", "Compare the current surface against the lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--verify-all", "Also check links to assets (PDFs, images, archives)").option("--lockfile <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | github | sarif", { default: "pretty" }).option("--fail-on <severity>", "error | warn | info", { default: "error" }).option("--audit", "Also run absolute rules, not just the diff", { default: true }).option("--update", "Write the new state to the lockfile after reporting").option("--baseline-branch <ref>", "Read the baseline lockfile from a git ref instead of disk").action(async (flags) => {
|
|
1978
2037
|
const failOn = parseFailOn(flags.failOn, false);
|
|
1979
2038
|
const config = await loadConfig(flags.config);
|
|
1980
2039
|
const next = await build(flags, config);
|
|
@@ -2004,7 +2063,7 @@ Failing: ${summary.error} error, ${summary.warn} warning (--fail-on ${failOn}).`
|
|
|
2004
2063
|
process.exitCode = EXIT_FINDINGS;
|
|
2005
2064
|
}
|
|
2006
2065
|
});
|
|
2007
|
-
cli.command("audit", "Audit a site as it stands, with explanations and fixes").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | html", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "never" }).action(async (flags) => {
|
|
2066
|
+
cli.command("audit", "Audit a site as it stands, with explanations and fixes").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--verify-all", "Also check links to assets (PDFs, images, archives)").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | html", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "never" }).action(async (flags) => {
|
|
2008
2067
|
const failOn = parseFailOn(flags.failOn, true);
|
|
2009
2068
|
const config = await loadConfig(flags.config);
|
|
2010
2069
|
const snapshot = await build(flags, config);
|
|
@@ -2044,7 +2103,44 @@ cli.command("audit", "Audit a site as it stands, with explanations and fixes").o
|
|
|
2044
2103
|
process.exitCode = EXIT_FINDINGS;
|
|
2045
2104
|
}
|
|
2046
2105
|
});
|
|
2047
|
-
cli.command("
|
|
2106
|
+
cli.command("page <url>", "Check one page: every link on it, and its own surface").option("--external", "Check links that leave the site", { default: true }).option("--verify-all", "Also check links to assets (PDFs, images, archives)").option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "error" }).action(async (url, flags) => {
|
|
2107
|
+
const failOn = parseFailOn(flags.failOn, true);
|
|
2108
|
+
const config = await loadConfig(flags.config);
|
|
2109
|
+
const snapshot = await snapshotFromPage(url, {
|
|
2110
|
+
...config,
|
|
2111
|
+
concurrency: flags.concurrency,
|
|
2112
|
+
checkExternal: flags.external,
|
|
2113
|
+
verifyAll: flags.verifyAll ?? config.verifyAll
|
|
2114
|
+
});
|
|
2115
|
+
const [page] = Object.values(snapshot.pages);
|
|
2116
|
+
const platform = detectPlatform([page.generator], Object.values(page.og));
|
|
2117
|
+
const findings = applyConfig(auditSnapshot(snapshot, config), config).filter((f) => f.route !== null).map((f) => withGuidance(f, platform));
|
|
2118
|
+
if (findings.length === 0 && !flags.out) {
|
|
2119
|
+
const checked = (page.brokenLinks?.length ?? 0) + (page.deadExternal?.length ?? 0);
|
|
2120
|
+
console.log(import_picocolors2.default.green(`${url} looks sound \u2014 no findings, no dead links.`));
|
|
2121
|
+
if (checked > 0) console.log(import_picocolors2.default.dim("(unreachable links were treated as unknown, not dead)"));
|
|
2122
|
+
return;
|
|
2123
|
+
}
|
|
2124
|
+
const groups = aggregate(findings);
|
|
2125
|
+
const meta = {
|
|
2126
|
+
target: url,
|
|
2127
|
+
platform,
|
|
2128
|
+
pageCount: 1,
|
|
2129
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString().slice(0, 10)
|
|
2130
|
+
};
|
|
2131
|
+
const output = flags.format === "json" ? JSON.stringify({ schemaVersion: 1, meta, summary: summarize(findings), groups }, null, 2) : flags.format === "markdown" ? formatAuditMarkdown(groups, meta) : formatAuditPretty(groups, meta, process.stdout.columns);
|
|
2132
|
+
if (flags.out) {
|
|
2133
|
+
await (0, import_promises2.writeFile)(flags.out, `${output}
|
|
2134
|
+
`, "utf8");
|
|
2135
|
+
console.log(import_picocolors2.default.green(`Wrote ${flags.out}.`));
|
|
2136
|
+
} else {
|
|
2137
|
+
console.log(output);
|
|
2138
|
+
}
|
|
2139
|
+
if (failOn !== "never" && shouldFail(findings, failOn)) {
|
|
2140
|
+
process.exitCode = EXIT_FINDINGS;
|
|
2141
|
+
}
|
|
2142
|
+
});
|
|
2143
|
+
cli.command("links", "Find internal links that point at no page").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--verify-all", "Also check links to assets (PDFs, images, archives)").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "error" }).action(async (flags) => {
|
|
2048
2144
|
const failOn = parseFailOn(flags.failOn, true);
|
|
2049
2145
|
const config = await loadConfig(flags.config);
|
|
2050
2146
|
const snapshot = await build(flags, config);
|
|
@@ -2091,7 +2187,7 @@ cli.command("links", "Find internal links that point at no page").option("--url
|
|
|
2091
2187
|
process.exitCode = EXIT_FINDINGS;
|
|
2092
2188
|
}
|
|
2093
2189
|
});
|
|
2094
|
-
cli.command("init", "Write a config file and take the first snapshot").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
2190
|
+
cli.command("init", "Write a config file and take the first snapshot").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--verify-all", "Also check links to assets (PDFs, images, archives)").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
2095
2191
|
const existing = await (0, import_promises2.readFile)(flags.config, "utf8").catch(() => null);
|
|
2096
2192
|
if (existing === null) {
|
|
2097
2193
|
const config = flags.url ? { siteUrl: new URL(flags.url).origin } : {};
|
|
@@ -2114,11 +2210,11 @@ Commit both files, then run \`pagetrace check ${flags.dir ? `--dir ${flags.dir}`
|
|
|
2114
2210
|
});
|
|
2115
2211
|
cli.command("update", "Check npm for a newer pagetrace and install it").option("--check", "Only report whether an update exists").action(async (flags) => {
|
|
2116
2212
|
const latest = await latestVersion("pagetrace");
|
|
2117
|
-
if (!isNewer(latest, "0.
|
|
2118
|
-
console.log(import_picocolors2.default.green(`pagetrace ${"0.
|
|
2213
|
+
if (!isNewer(latest, "0.14.0")) {
|
|
2214
|
+
console.log(import_picocolors2.default.green(`pagetrace ${"0.14.0"} is the latest version.`));
|
|
2119
2215
|
return;
|
|
2120
2216
|
}
|
|
2121
|
-
console.log(import_picocolors2.default.yellow(`Update available: ${"0.
|
|
2217
|
+
console.log(import_picocolors2.default.yellow(`Update available: ${"0.14.0"} \u2192 ${latest}`));
|
|
2122
2218
|
if (flags.check) return;
|
|
2123
2219
|
if (process.argv[1]?.startsWith(process.cwd())) {
|
|
2124
2220
|
console.log(
|
|
@@ -2136,7 +2232,7 @@ cli.command("update", "Check npm for a newer pagetrace and install it").option("
|
|
|
2136
2232
|
console.log(import_picocolors2.default.green(`Updated to pagetrace ${latest}.`));
|
|
2137
2233
|
});
|
|
2138
2234
|
cli.help();
|
|
2139
|
-
cli.version("0.
|
|
2235
|
+
cli.version("0.14.0");
|
|
2140
2236
|
async function main() {
|
|
2141
2237
|
try {
|
|
2142
2238
|
cli.parse(process.argv, { run: false });
|
package/dist/cli.js
CHANGED
|
@@ -1304,7 +1304,7 @@ ${cards || "<p>No issues found.</p>"}
|
|
|
1304
1304
|
|
|
1305
1305
|
// src/snapshot.ts
|
|
1306
1306
|
import { execFile } from "child_process";
|
|
1307
|
-
import { readdir, readFile } from "fs/promises";
|
|
1307
|
+
import { access, readdir, readFile } from "fs/promises";
|
|
1308
1308
|
import { join, relative, sep } from "path";
|
|
1309
1309
|
import { promisify } from "util";
|
|
1310
1310
|
|
|
@@ -1675,12 +1675,20 @@ async function snapshotFromDir(dir, config = {}) {
|
|
|
1675
1675
|
const llms = await readFile(join(dir, "llms.txt"), "utf8").catch(() => null);
|
|
1676
1676
|
if (llms !== null) site.llmsTxt = extractLlmsTxt(llms);
|
|
1677
1677
|
const origin = site.origin ?? "https://pagetrace.invalid";
|
|
1678
|
-
await resolveBrokenLinks(pages, links, origin
|
|
1678
|
+
await resolveBrokenLinks(pages, links, origin, {
|
|
1679
|
+
includeAssets: config.verifyAll,
|
|
1680
|
+
// The HTML route set is authoritative here, so a missing page is simply
|
|
1681
|
+
// broken. An asset is a file this crawl never walked, so it gets a stat.
|
|
1682
|
+
verify: config.verifyAll ? async (target) => ASSET_PATH.test(target) ? !await access(join(dir, target)).then(
|
|
1683
|
+
() => true,
|
|
1684
|
+
() => false
|
|
1685
|
+
) : true : void 0
|
|
1686
|
+
});
|
|
1679
1687
|
if (config.checkExternal) await resolveDeadExternal(pages, links, origin, 5);
|
|
1680
1688
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1681
1689
|
}
|
|
1682
1690
|
var ASSET_PATH = /\.(?!html?$)[a-z0-9]+$/i;
|
|
1683
|
-
function linkTarget(href, from, origin) {
|
|
1691
|
+
function linkTarget(href, from, origin, includeAssets = false) {
|
|
1684
1692
|
let url;
|
|
1685
1693
|
try {
|
|
1686
1694
|
url = new URL(href, `${origin}${from === "/" ? "" : from}/`);
|
|
@@ -1688,27 +1696,39 @@ function linkTarget(href, from, origin) {
|
|
|
1688
1696
|
return null;
|
|
1689
1697
|
}
|
|
1690
1698
|
if (url.origin !== origin) return null;
|
|
1691
|
-
if (ASSET_PATH.test(url.pathname)) return null;
|
|
1692
|
-
return routeFromUrl(url.href);
|
|
1699
|
+
if (!includeAssets && ASSET_PATH.test(url.pathname)) return null;
|
|
1700
|
+
return ASSET_PATH.test(url.pathname) ? url.pathname : routeFromUrl(url.href);
|
|
1693
1701
|
}
|
|
1694
|
-
var MAX_LINK_CHECKS =
|
|
1695
|
-
async function resolveBrokenLinks(pages, links, origin,
|
|
1702
|
+
var MAX_LINK_CHECKS = 1e3;
|
|
1703
|
+
async function resolveBrokenLinks(pages, links, origin, options = {}) {
|
|
1704
|
+
const { includeAssets = false, verify, concurrency = 5 } = options;
|
|
1696
1705
|
const known = new Set(Object.keys(pages));
|
|
1697
1706
|
const candidates = /* @__PURE__ */ new Map();
|
|
1698
1707
|
for (const [route, hrefs] of Object.entries(links)) {
|
|
1699
1708
|
const missing = [
|
|
1700
1709
|
...new Set(
|
|
1701
|
-
hrefs.map((href) => linkTarget(href, route, origin)).filter((target) => target !== null && !known.has(target))
|
|
1710
|
+
hrefs.map((href) => linkTarget(href, route, origin, includeAssets)).filter((target) => target !== null && !known.has(target))
|
|
1702
1711
|
)
|
|
1703
1712
|
];
|
|
1704
1713
|
if (missing.length > 0) candidates.set(route, missing);
|
|
1705
1714
|
}
|
|
1706
1715
|
const broken = /* @__PURE__ */ new Set();
|
|
1707
1716
|
if (verify) {
|
|
1708
|
-
const
|
|
1709
|
-
|
|
1710
|
-
|
|
1717
|
+
const unique = [...new Set([...candidates.values()].flat())];
|
|
1718
|
+
const queue = unique.slice(0, MAX_LINK_CHECKS);
|
|
1719
|
+
if (unique.length > queue.length) {
|
|
1720
|
+
console.error(
|
|
1721
|
+
`pagetrace: ${unique.length} link targets to confirm, checking the first ${MAX_LINK_CHECKS}.`
|
|
1722
|
+
);
|
|
1711
1723
|
}
|
|
1724
|
+
await Promise.all(
|
|
1725
|
+
Array.from({ length: Math.min(Math.max(1, concurrency), queue.length) }, async () => {
|
|
1726
|
+
while (queue.length > 0) {
|
|
1727
|
+
const target = queue.shift();
|
|
1728
|
+
if (await verify(target)) broken.add(target);
|
|
1729
|
+
}
|
|
1730
|
+
})
|
|
1731
|
+
);
|
|
1712
1732
|
}
|
|
1713
1733
|
for (const [route, missing] of candidates) {
|
|
1714
1734
|
const confirmed = verify ? missing.filter((target) => broken.has(target)) : missing;
|
|
@@ -1772,6 +1792,39 @@ async function resolveDeadExternal(pages, links, origin, concurrency, timeoutMs)
|
|
|
1772
1792
|
if (found.length > 0) pages[route].deadExternal = found;
|
|
1773
1793
|
}
|
|
1774
1794
|
}
|
|
1795
|
+
async function snapshotFromPage(pageUrl, options = {}) {
|
|
1796
|
+
const target = new URL(pageUrl);
|
|
1797
|
+
const site = {
|
|
1798
|
+
origin: options.siteUrl ? new URL(options.siteUrl).origin : target.origin,
|
|
1799
|
+
robotsTxt: null,
|
|
1800
|
+
llmsTxt: null,
|
|
1801
|
+
sitemap: null
|
|
1802
|
+
};
|
|
1803
|
+
const doc = await fetchDoc(target.href, options.timeout);
|
|
1804
|
+
if (doc === null) throw new Error(`${target.href} answered 404 \u2014 nothing to check.`);
|
|
1805
|
+
const route = routeFromUrl(target.href);
|
|
1806
|
+
const page = extractPage(doc.text, route);
|
|
1807
|
+
const landed = !doc.url ? route : originOf2(doc.url) === target.origin ? routeFromUrl(doc.url) : doc.url;
|
|
1808
|
+
page.redirectsTo = landed === route ? null : landed;
|
|
1809
|
+
const pages = { [route]: page };
|
|
1810
|
+
const links = { [route]: extractLinks(doc.text) };
|
|
1811
|
+
const origin = site.origin ?? target.origin;
|
|
1812
|
+
await resolveBrokenLinks(pages, links, origin, {
|
|
1813
|
+
includeAssets: options.verifyAll,
|
|
1814
|
+
concurrency: options.concurrency,
|
|
1815
|
+
verify: async (candidate) => {
|
|
1816
|
+
try {
|
|
1817
|
+
return await fetchDoc(new URL(candidate, target.origin).href, options.timeout) === null;
|
|
1818
|
+
} catch {
|
|
1819
|
+
return false;
|
|
1820
|
+
}
|
|
1821
|
+
}
|
|
1822
|
+
});
|
|
1823
|
+
if (options.checkExternal) {
|
|
1824
|
+
await resolveDeadExternal(pages, links, origin, Math.max(1, options.concurrency ?? 5), options.timeout);
|
|
1825
|
+
}
|
|
1826
|
+
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1827
|
+
}
|
|
1775
1828
|
var MAX_NESTED_SITEMAPS = 50;
|
|
1776
1829
|
async function snapshotFromOrigin(origin, options = {}) {
|
|
1777
1830
|
const base = new URL(origin);
|
|
@@ -1853,11 +1906,15 @@ async function snapshotFromOrigin(origin, options = {}) {
|
|
|
1853
1906
|
}
|
|
1854
1907
|
})
|
|
1855
1908
|
);
|
|
1856
|
-
await resolveBrokenLinks(pages, links, base.origin,
|
|
1857
|
-
|
|
1858
|
-
|
|
1859
|
-
|
|
1860
|
-
|
|
1909
|
+
await resolveBrokenLinks(pages, links, base.origin, {
|
|
1910
|
+
includeAssets: options.verifyAll,
|
|
1911
|
+
concurrency,
|
|
1912
|
+
verify: async (route) => {
|
|
1913
|
+
try {
|
|
1914
|
+
return await fetchDoc(new URL(route, base).href, timeout) === null;
|
|
1915
|
+
} catch {
|
|
1916
|
+
return false;
|
|
1917
|
+
}
|
|
1861
1918
|
}
|
|
1862
1919
|
});
|
|
1863
1920
|
if (options.checkExternal) {
|
|
@@ -1916,14 +1973,16 @@ async function loadConfig(path = DEFAULT_CONFIG) {
|
|
|
1916
1973
|
}
|
|
1917
1974
|
async function build(flags, config) {
|
|
1918
1975
|
const checkExternal = flags.external ?? config.checkExternal;
|
|
1919
|
-
|
|
1976
|
+
const verifyAll = flags.verifyAll ?? config.verifyAll;
|
|
1977
|
+
if (flags.dir) return snapshotFromDir(flags.dir, { ...config, checkExternal, verifyAll });
|
|
1920
1978
|
if (flags.url)
|
|
1921
1979
|
return snapshotFromOrigin(flags.url, {
|
|
1922
1980
|
...config,
|
|
1923
1981
|
limit: flags.limit,
|
|
1924
1982
|
concurrency: flags.concurrency,
|
|
1925
1983
|
ignoreRobots: flags.ignoreRobots ?? config.ignoreRobots,
|
|
1926
|
-
checkExternal
|
|
1984
|
+
checkExternal,
|
|
1985
|
+
verifyAll
|
|
1927
1986
|
});
|
|
1928
1987
|
throw new Error("Provide a source: --dir <build directory> or --url <origin>.");
|
|
1929
1988
|
}
|
|
@@ -1942,7 +2001,7 @@ function render(findings, format) {
|
|
|
1942
2001
|
}
|
|
1943
2002
|
}
|
|
1944
2003
|
var cli = cac("pagetrace");
|
|
1945
|
-
cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
2004
|
+
cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--verify-all", "Also check links to assets (PDFs, images, archives)").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
1946
2005
|
const config = await loadConfig(flags.config);
|
|
1947
2006
|
const snapshot = await build(flags, config);
|
|
1948
2007
|
const written = await writeLockfile(flags.out, snapshot);
|
|
@@ -1951,7 +2010,7 @@ cli.command("snapshot", "Record the current SEO/AEO surface to a lockfile").opti
|
|
|
1951
2010
|
written ? pc2.green(`Wrote ${flags.out} \u2014 ${count} page${count === 1 ? "" : "s"}.`) : pc2.dim(`${flags.out} is already up to date \u2014 ${count} page${count === 1 ? "" : "s"}.`)
|
|
1952
2011
|
);
|
|
1953
2012
|
});
|
|
1954
|
-
cli.command("check", "Compare the current surface against the lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--lockfile <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | github | sarif", { default: "pretty" }).option("--fail-on <severity>", "error | warn | info", { default: "error" }).option("--audit", "Also run absolute rules, not just the diff", { default: true }).option("--update", "Write the new state to the lockfile after reporting").option("--baseline-branch <ref>", "Read the baseline lockfile from a git ref instead of disk").action(async (flags) => {
|
|
2013
|
+
cli.command("check", "Compare the current surface against the lockfile").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--verify-all", "Also check links to assets (PDFs, images, archives)").option("--lockfile <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | github | sarif", { default: "pretty" }).option("--fail-on <severity>", "error | warn | info", { default: "error" }).option("--audit", "Also run absolute rules, not just the diff", { default: true }).option("--update", "Write the new state to the lockfile after reporting").option("--baseline-branch <ref>", "Read the baseline lockfile from a git ref instead of disk").action(async (flags) => {
|
|
1955
2014
|
const failOn = parseFailOn(flags.failOn, false);
|
|
1956
2015
|
const config = await loadConfig(flags.config);
|
|
1957
2016
|
const next = await build(flags, config);
|
|
@@ -1981,7 +2040,7 @@ Failing: ${summary.error} error, ${summary.warn} warning (--fail-on ${failOn}).`
|
|
|
1981
2040
|
process.exitCode = EXIT_FINDINGS;
|
|
1982
2041
|
}
|
|
1983
2042
|
});
|
|
1984
|
-
cli.command("audit", "Audit a site as it stands, with explanations and fixes").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | html", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "never" }).action(async (flags) => {
|
|
2043
|
+
cli.command("audit", "Audit a site as it stands, with explanations and fixes").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--verify-all", "Also check links to assets (PDFs, images, archives)").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown | html", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "never" }).action(async (flags) => {
|
|
1985
2044
|
const failOn = parseFailOn(flags.failOn, true);
|
|
1986
2045
|
const config = await loadConfig(flags.config);
|
|
1987
2046
|
const snapshot = await build(flags, config);
|
|
@@ -2021,7 +2080,44 @@ cli.command("audit", "Audit a site as it stands, with explanations and fixes").o
|
|
|
2021
2080
|
process.exitCode = EXIT_FINDINGS;
|
|
2022
2081
|
}
|
|
2023
2082
|
});
|
|
2024
|
-
cli.command("
|
|
2083
|
+
cli.command("page <url>", "Check one page: every link on it, and its own surface").option("--external", "Check links that leave the site", { default: true }).option("--verify-all", "Also check links to assets (PDFs, images, archives)").option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "error" }).action(async (url, flags) => {
|
|
2084
|
+
const failOn = parseFailOn(flags.failOn, true);
|
|
2085
|
+
const config = await loadConfig(flags.config);
|
|
2086
|
+
const snapshot = await snapshotFromPage(url, {
|
|
2087
|
+
...config,
|
|
2088
|
+
concurrency: flags.concurrency,
|
|
2089
|
+
checkExternal: flags.external,
|
|
2090
|
+
verifyAll: flags.verifyAll ?? config.verifyAll
|
|
2091
|
+
});
|
|
2092
|
+
const [page] = Object.values(snapshot.pages);
|
|
2093
|
+
const platform = detectPlatform([page.generator], Object.values(page.og));
|
|
2094
|
+
const findings = applyConfig(auditSnapshot(snapshot, config), config).filter((f) => f.route !== null).map((f) => withGuidance(f, platform));
|
|
2095
|
+
if (findings.length === 0 && !flags.out) {
|
|
2096
|
+
const checked = (page.brokenLinks?.length ?? 0) + (page.deadExternal?.length ?? 0);
|
|
2097
|
+
console.log(pc2.green(`${url} looks sound \u2014 no findings, no dead links.`));
|
|
2098
|
+
if (checked > 0) console.log(pc2.dim("(unreachable links were treated as unknown, not dead)"));
|
|
2099
|
+
return;
|
|
2100
|
+
}
|
|
2101
|
+
const groups = aggregate(findings);
|
|
2102
|
+
const meta = {
|
|
2103
|
+
target: url,
|
|
2104
|
+
platform,
|
|
2105
|
+
pageCount: 1,
|
|
2106
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString().slice(0, 10)
|
|
2107
|
+
};
|
|
2108
|
+
const output = flags.format === "json" ? JSON.stringify({ schemaVersion: 1, meta, summary: summarize(findings), groups }, null, 2) : flags.format === "markdown" ? formatAuditMarkdown(groups, meta) : formatAuditPretty(groups, meta, process.stdout.columns);
|
|
2109
|
+
if (flags.out) {
|
|
2110
|
+
await writeFile(flags.out, `${output}
|
|
2111
|
+
`, "utf8");
|
|
2112
|
+
console.log(pc2.green(`Wrote ${flags.out}.`));
|
|
2113
|
+
} else {
|
|
2114
|
+
console.log(output);
|
|
2115
|
+
}
|
|
2116
|
+
if (failOn !== "never" && shouldFail(findings, failOn)) {
|
|
2117
|
+
process.exitCode = EXIT_FINDINGS;
|
|
2118
|
+
}
|
|
2119
|
+
});
|
|
2120
|
+
cli.command("links", "Find internal links that point at no page").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--verify-all", "Also check links to assets (PDFs, images, archives)").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "error" }).action(async (flags) => {
|
|
2025
2121
|
const failOn = parseFailOn(flags.failOn, true);
|
|
2026
2122
|
const config = await loadConfig(flags.config);
|
|
2027
2123
|
const snapshot = await build(flags, config);
|
|
@@ -2068,7 +2164,7 @@ cli.command("links", "Find internal links that point at no page").option("--url
|
|
|
2068
2164
|
process.exitCode = EXIT_FINDINGS;
|
|
2069
2165
|
}
|
|
2070
2166
|
});
|
|
2071
|
-
cli.command("init", "Write a config file and take the first snapshot").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
2167
|
+
cli.command("init", "Write a config file and take the first snapshot").option("--dir <dir>", "Directory of built HTML").option("--url <origin>", "Live origin to crawl").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--verify-all", "Also check links to assets (PDFs, images, archives)").option("--out <file>", "Lockfile path", { default: DEFAULT_LOCKFILE }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).action(async (flags) => {
|
|
2072
2168
|
const existing = await readFile2(flags.config, "utf8").catch(() => null);
|
|
2073
2169
|
if (existing === null) {
|
|
2074
2170
|
const config = flags.url ? { siteUrl: new URL(flags.url).origin } : {};
|
|
@@ -2091,11 +2187,11 @@ Commit both files, then run \`pagetrace check ${flags.dir ? `--dir ${flags.dir}`
|
|
|
2091
2187
|
});
|
|
2092
2188
|
cli.command("update", "Check npm for a newer pagetrace and install it").option("--check", "Only report whether an update exists").action(async (flags) => {
|
|
2093
2189
|
const latest = await latestVersion("pagetrace");
|
|
2094
|
-
if (!isNewer(latest, "0.
|
|
2095
|
-
console.log(pc2.green(`pagetrace ${"0.
|
|
2190
|
+
if (!isNewer(latest, "0.14.0")) {
|
|
2191
|
+
console.log(pc2.green(`pagetrace ${"0.14.0"} is the latest version.`));
|
|
2096
2192
|
return;
|
|
2097
2193
|
}
|
|
2098
|
-
console.log(pc2.yellow(`Update available: ${"0.
|
|
2194
|
+
console.log(pc2.yellow(`Update available: ${"0.14.0"} \u2192 ${latest}`));
|
|
2099
2195
|
if (flags.check) return;
|
|
2100
2196
|
if (process.argv[1]?.startsWith(process.cwd())) {
|
|
2101
2197
|
console.log(
|
|
@@ -2113,7 +2209,7 @@ cli.command("update", "Check npm for a newer pagetrace and install it").option("
|
|
|
2113
2209
|
console.log(pc2.green(`Updated to pagetrace ${latest}.`));
|
|
2114
2210
|
});
|
|
2115
2211
|
cli.help();
|
|
2116
|
-
cli.version("0.
|
|
2212
|
+
cli.version("0.14.0");
|
|
2117
2213
|
async function main() {
|
|
2118
2214
|
try {
|
|
2119
2215
|
cli.parse(process.argv, { run: false });
|
package/dist/index.cjs
CHANGED
|
@@ -67,6 +67,7 @@ __export(src_exports, {
|
|
|
67
67
|
snapshotFromDir: () => snapshotFromDir,
|
|
68
68
|
snapshotFromGitRef: () => snapshotFromGitRef,
|
|
69
69
|
snapshotFromOrigin: () => snapshotFromOrigin,
|
|
70
|
+
snapshotFromPage: () => snapshotFromPage,
|
|
70
71
|
summarize: () => summarize,
|
|
71
72
|
withGuidance: () => withGuidance
|
|
72
73
|
});
|
|
@@ -1739,12 +1740,20 @@ async function snapshotFromDir(dir, config = {}) {
|
|
|
1739
1740
|
const llms = await (0, import_promises.readFile)((0, import_node_path.join)(dir, "llms.txt"), "utf8").catch(() => null);
|
|
1740
1741
|
if (llms !== null) site.llmsTxt = extractLlmsTxt(llms);
|
|
1741
1742
|
const origin = site.origin ?? "https://pagetrace.invalid";
|
|
1742
|
-
await resolveBrokenLinks(pages, links, origin
|
|
1743
|
+
await resolveBrokenLinks(pages, links, origin, {
|
|
1744
|
+
includeAssets: config.verifyAll,
|
|
1745
|
+
// The HTML route set is authoritative here, so a missing page is simply
|
|
1746
|
+
// broken. An asset is a file this crawl never walked, so it gets a stat.
|
|
1747
|
+
verify: config.verifyAll ? async (target) => ASSET_PATH.test(target) ? !await (0, import_promises.access)((0, import_node_path.join)(dir, target)).then(
|
|
1748
|
+
() => true,
|
|
1749
|
+
() => false
|
|
1750
|
+
) : true : void 0
|
|
1751
|
+
});
|
|
1743
1752
|
if (config.checkExternal) await resolveDeadExternal(pages, links, origin, 5);
|
|
1744
1753
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1745
1754
|
}
|
|
1746
1755
|
var ASSET_PATH = /\.(?!html?$)[a-z0-9]+$/i;
|
|
1747
|
-
function linkTarget(href, from, origin) {
|
|
1756
|
+
function linkTarget(href, from, origin, includeAssets = false) {
|
|
1748
1757
|
let url;
|
|
1749
1758
|
try {
|
|
1750
1759
|
url = new URL(href, `${origin}${from === "/" ? "" : from}/`);
|
|
@@ -1752,27 +1761,39 @@ function linkTarget(href, from, origin) {
|
|
|
1752
1761
|
return null;
|
|
1753
1762
|
}
|
|
1754
1763
|
if (url.origin !== origin) return null;
|
|
1755
|
-
if (ASSET_PATH.test(url.pathname)) return null;
|
|
1756
|
-
return routeFromUrl(url.href);
|
|
1764
|
+
if (!includeAssets && ASSET_PATH.test(url.pathname)) return null;
|
|
1765
|
+
return ASSET_PATH.test(url.pathname) ? url.pathname : routeFromUrl(url.href);
|
|
1757
1766
|
}
|
|
1758
|
-
var MAX_LINK_CHECKS =
|
|
1759
|
-
async function resolveBrokenLinks(pages, links, origin,
|
|
1767
|
+
var MAX_LINK_CHECKS = 1e3;
|
|
1768
|
+
async function resolveBrokenLinks(pages, links, origin, options = {}) {
|
|
1769
|
+
const { includeAssets = false, verify, concurrency = 5 } = options;
|
|
1760
1770
|
const known = new Set(Object.keys(pages));
|
|
1761
1771
|
const candidates = /* @__PURE__ */ new Map();
|
|
1762
1772
|
for (const [route, hrefs] of Object.entries(links)) {
|
|
1763
1773
|
const missing = [
|
|
1764
1774
|
...new Set(
|
|
1765
|
-
hrefs.map((href) => linkTarget(href, route, origin)).filter((target) => target !== null && !known.has(target))
|
|
1775
|
+
hrefs.map((href) => linkTarget(href, route, origin, includeAssets)).filter((target) => target !== null && !known.has(target))
|
|
1766
1776
|
)
|
|
1767
1777
|
];
|
|
1768
1778
|
if (missing.length > 0) candidates.set(route, missing);
|
|
1769
1779
|
}
|
|
1770
1780
|
const broken = /* @__PURE__ */ new Set();
|
|
1771
1781
|
if (verify) {
|
|
1772
|
-
const
|
|
1773
|
-
|
|
1774
|
-
|
|
1782
|
+
const unique = [...new Set([...candidates.values()].flat())];
|
|
1783
|
+
const queue = unique.slice(0, MAX_LINK_CHECKS);
|
|
1784
|
+
if (unique.length > queue.length) {
|
|
1785
|
+
console.error(
|
|
1786
|
+
`pagetrace: ${unique.length} link targets to confirm, checking the first ${MAX_LINK_CHECKS}.`
|
|
1787
|
+
);
|
|
1775
1788
|
}
|
|
1789
|
+
await Promise.all(
|
|
1790
|
+
Array.from({ length: Math.min(Math.max(1, concurrency), queue.length) }, async () => {
|
|
1791
|
+
while (queue.length > 0) {
|
|
1792
|
+
const target = queue.shift();
|
|
1793
|
+
if (await verify(target)) broken.add(target);
|
|
1794
|
+
}
|
|
1795
|
+
})
|
|
1796
|
+
);
|
|
1776
1797
|
}
|
|
1777
1798
|
for (const [route, missing] of candidates) {
|
|
1778
1799
|
const confirmed = verify ? missing.filter((target) => broken.has(target)) : missing;
|
|
@@ -1836,6 +1857,39 @@ async function resolveDeadExternal(pages, links, origin, concurrency, timeoutMs)
|
|
|
1836
1857
|
if (found.length > 0) pages[route].deadExternal = found;
|
|
1837
1858
|
}
|
|
1838
1859
|
}
|
|
1860
|
+
async function snapshotFromPage(pageUrl, options = {}) {
|
|
1861
|
+
const target = new URL(pageUrl);
|
|
1862
|
+
const site = {
|
|
1863
|
+
origin: options.siteUrl ? new URL(options.siteUrl).origin : target.origin,
|
|
1864
|
+
robotsTxt: null,
|
|
1865
|
+
llmsTxt: null,
|
|
1866
|
+
sitemap: null
|
|
1867
|
+
};
|
|
1868
|
+
const doc = await fetchDoc(target.href, options.timeout);
|
|
1869
|
+
if (doc === null) throw new Error(`${target.href} answered 404 \u2014 nothing to check.`);
|
|
1870
|
+
const route = routeFromUrl(target.href);
|
|
1871
|
+
const page = extractPage(doc.text, route);
|
|
1872
|
+
const landed = !doc.url ? route : originOf2(doc.url) === target.origin ? routeFromUrl(doc.url) : doc.url;
|
|
1873
|
+
page.redirectsTo = landed === route ? null : landed;
|
|
1874
|
+
const pages = { [route]: page };
|
|
1875
|
+
const links = { [route]: extractLinks(doc.text) };
|
|
1876
|
+
const origin = site.origin ?? target.origin;
|
|
1877
|
+
await resolveBrokenLinks(pages, links, origin, {
|
|
1878
|
+
includeAssets: options.verifyAll,
|
|
1879
|
+
concurrency: options.concurrency,
|
|
1880
|
+
verify: async (candidate) => {
|
|
1881
|
+
try {
|
|
1882
|
+
return await fetchDoc(new URL(candidate, target.origin).href, options.timeout) === null;
|
|
1883
|
+
} catch {
|
|
1884
|
+
return false;
|
|
1885
|
+
}
|
|
1886
|
+
}
|
|
1887
|
+
});
|
|
1888
|
+
if (options.checkExternal) {
|
|
1889
|
+
await resolveDeadExternal(pages, links, origin, Math.max(1, options.concurrency ?? 5), options.timeout);
|
|
1890
|
+
}
|
|
1891
|
+
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1892
|
+
}
|
|
1839
1893
|
var MAX_NESTED_SITEMAPS = 50;
|
|
1840
1894
|
async function snapshotFromOrigin(origin, options = {}) {
|
|
1841
1895
|
const base = new URL(origin);
|
|
@@ -1917,11 +1971,15 @@ async function snapshotFromOrigin(origin, options = {}) {
|
|
|
1917
1971
|
}
|
|
1918
1972
|
})
|
|
1919
1973
|
);
|
|
1920
|
-
await resolveBrokenLinks(pages, links, base.origin,
|
|
1921
|
-
|
|
1922
|
-
|
|
1923
|
-
|
|
1924
|
-
|
|
1974
|
+
await resolveBrokenLinks(pages, links, base.origin, {
|
|
1975
|
+
includeAssets: options.verifyAll,
|
|
1976
|
+
concurrency,
|
|
1977
|
+
verify: async (route) => {
|
|
1978
|
+
try {
|
|
1979
|
+
return await fetchDoc(new URL(route, base).href, timeout) === null;
|
|
1980
|
+
} catch {
|
|
1981
|
+
return false;
|
|
1982
|
+
}
|
|
1925
1983
|
}
|
|
1926
1984
|
});
|
|
1927
1985
|
if (options.checkExternal) {
|
|
@@ -1968,6 +2026,7 @@ async function snapshotFromOrigin(origin, options = {}) {
|
|
|
1968
2026
|
snapshotFromDir,
|
|
1969
2027
|
snapshotFromGitRef,
|
|
1970
2028
|
snapshotFromOrigin,
|
|
2029
|
+
snapshotFromPage,
|
|
1971
2030
|
summarize,
|
|
1972
2031
|
withGuidance
|
|
1973
2032
|
});
|
package/dist/index.d.cts
CHANGED
|
@@ -151,6 +151,11 @@ interface Config {
|
|
|
151
151
|
* third party's bad afternoon into a diff in your repository.
|
|
152
152
|
*/
|
|
153
153
|
checkExternal?: boolean;
|
|
154
|
+
/**
|
|
155
|
+
* Also check links to assets — PDFs, images, archives. They are never crawled
|
|
156
|
+
* as pages, so each one costs a request (or a filesystem check for --dir).
|
|
157
|
+
*/
|
|
158
|
+
verifyAll?: boolean;
|
|
154
159
|
}
|
|
155
160
|
|
|
156
161
|
/**
|
|
@@ -317,6 +322,14 @@ declare function sameSurface(a: Snapshot, b: Snapshot): boolean;
|
|
|
317
322
|
declare function shouldIgnore(route: string, patterns?: string[]): boolean;
|
|
318
323
|
/** Build a snapshot from a directory of pre-rendered HTML (next export, dist, out). */
|
|
319
324
|
declare function snapshotFromDir(dir: string, config?: Config): Promise<Snapshot>;
|
|
325
|
+
/**
|
|
326
|
+
* One page, checked properly. No sitemap, no route discovery: the snapshot
|
|
327
|
+
* holds a single page, so every link on it is a candidate and every candidate
|
|
328
|
+
* is confirmed with a real request. That is the opposite trade from a crawl,
|
|
329
|
+
* and the right one here — one page's worth of links is a bounded cost, and
|
|
330
|
+
* "is this page's linking sound" is a question a crawl answers slowly.
|
|
331
|
+
*/
|
|
332
|
+
declare function snapshotFromPage(pageUrl: string, options?: CrawlOptions): Promise<Snapshot>;
|
|
320
333
|
interface CrawlOptions extends Config {
|
|
321
334
|
/** Cap the number of pages fetched. */
|
|
322
335
|
limit?: number;
|
|
@@ -328,4 +341,4 @@ interface CrawlOptions extends Config {
|
|
|
328
341
|
/** Build a snapshot by fetching a live origin, discovering routes via sitemap. */
|
|
329
342
|
declare function snapshotFromOrigin(origin: string, options?: CrawlOptions): Promise<Snapshot>;
|
|
330
343
|
|
|
331
|
-
export { type Aggregate, type AuditMeta, type Config, DEFAULT_AI_AGENTS, type Finding, GUIDANCE, type Guidance, type JsonLdEntity, type PageFingerprint, type Platform, RICH_RESULT_RULES, type Severity, type SiteFingerprint, type Snapshot, aggregate, applyConfig, auditCrossPage, auditHreflang, auditPage, auditSite, auditSnapshot, detectPlatform, diffPage, diffSite, diffSnapshots, extractJsonLd, extractLinks, extractLlmsTxt, extractPage, extractRobotsTxt, extractSitemapUrls, formatAuditHtml, formatAuditMarkdown, formatAuditPretty, formatGithub, formatJson, formatMarkdown, formatPretty, formatSarif, isCrawlable, routeFromFilePath, routeFromUrl, sameSurface, shouldFail, shouldIgnore, snapshotFromDir, snapshotFromGitRef, snapshotFromOrigin, summarize, withGuidance };
|
|
344
|
+
export { type Aggregate, type AuditMeta, type Config, DEFAULT_AI_AGENTS, type Finding, GUIDANCE, type Guidance, type JsonLdEntity, type PageFingerprint, type Platform, RICH_RESULT_RULES, type Severity, type SiteFingerprint, type Snapshot, aggregate, applyConfig, auditCrossPage, auditHreflang, auditPage, auditSite, auditSnapshot, detectPlatform, diffPage, diffSite, diffSnapshots, extractJsonLd, extractLinks, extractLlmsTxt, extractPage, extractRobotsTxt, extractSitemapUrls, formatAuditHtml, formatAuditMarkdown, formatAuditPretty, formatGithub, formatJson, formatMarkdown, formatPretty, formatSarif, isCrawlable, routeFromFilePath, routeFromUrl, sameSurface, shouldFail, shouldIgnore, snapshotFromDir, snapshotFromGitRef, snapshotFromOrigin, snapshotFromPage, summarize, withGuidance };
|
package/dist/index.d.ts
CHANGED
|
@@ -151,6 +151,11 @@ interface Config {
|
|
|
151
151
|
* third party's bad afternoon into a diff in your repository.
|
|
152
152
|
*/
|
|
153
153
|
checkExternal?: boolean;
|
|
154
|
+
/**
|
|
155
|
+
* Also check links to assets — PDFs, images, archives. They are never crawled
|
|
156
|
+
* as pages, so each one costs a request (or a filesystem check for --dir).
|
|
157
|
+
*/
|
|
158
|
+
verifyAll?: boolean;
|
|
154
159
|
}
|
|
155
160
|
|
|
156
161
|
/**
|
|
@@ -317,6 +322,14 @@ declare function sameSurface(a: Snapshot, b: Snapshot): boolean;
|
|
|
317
322
|
declare function shouldIgnore(route: string, patterns?: string[]): boolean;
|
|
318
323
|
/** Build a snapshot from a directory of pre-rendered HTML (next export, dist, out). */
|
|
319
324
|
declare function snapshotFromDir(dir: string, config?: Config): Promise<Snapshot>;
|
|
325
|
+
/**
|
|
326
|
+
* One page, checked properly. No sitemap, no route discovery: the snapshot
|
|
327
|
+
* holds a single page, so every link on it is a candidate and every candidate
|
|
328
|
+
* is confirmed with a real request. That is the opposite trade from a crawl,
|
|
329
|
+
* and the right one here — one page's worth of links is a bounded cost, and
|
|
330
|
+
* "is this page's linking sound" is a question a crawl answers slowly.
|
|
331
|
+
*/
|
|
332
|
+
declare function snapshotFromPage(pageUrl: string, options?: CrawlOptions): Promise<Snapshot>;
|
|
320
333
|
interface CrawlOptions extends Config {
|
|
321
334
|
/** Cap the number of pages fetched. */
|
|
322
335
|
limit?: number;
|
|
@@ -328,4 +341,4 @@ interface CrawlOptions extends Config {
|
|
|
328
341
|
/** Build a snapshot by fetching a live origin, discovering routes via sitemap. */
|
|
329
342
|
declare function snapshotFromOrigin(origin: string, options?: CrawlOptions): Promise<Snapshot>;
|
|
330
343
|
|
|
331
|
-
export { type Aggregate, type AuditMeta, type Config, DEFAULT_AI_AGENTS, type Finding, GUIDANCE, type Guidance, type JsonLdEntity, type PageFingerprint, type Platform, RICH_RESULT_RULES, type Severity, type SiteFingerprint, type Snapshot, aggregate, applyConfig, auditCrossPage, auditHreflang, auditPage, auditSite, auditSnapshot, detectPlatform, diffPage, diffSite, diffSnapshots, extractJsonLd, extractLinks, extractLlmsTxt, extractPage, extractRobotsTxt, extractSitemapUrls, formatAuditHtml, formatAuditMarkdown, formatAuditPretty, formatGithub, formatJson, formatMarkdown, formatPretty, formatSarif, isCrawlable, routeFromFilePath, routeFromUrl, sameSurface, shouldFail, shouldIgnore, snapshotFromDir, snapshotFromGitRef, snapshotFromOrigin, summarize, withGuidance };
|
|
344
|
+
export { type Aggregate, type AuditMeta, type Config, DEFAULT_AI_AGENTS, type Finding, GUIDANCE, type Guidance, type JsonLdEntity, type PageFingerprint, type Platform, RICH_RESULT_RULES, type Severity, type SiteFingerprint, type Snapshot, aggregate, applyConfig, auditCrossPage, auditHreflang, auditPage, auditSite, auditSnapshot, detectPlatform, diffPage, diffSite, diffSnapshots, extractJsonLd, extractLinks, extractLlmsTxt, extractPage, extractRobotsTxt, extractSitemapUrls, formatAuditHtml, formatAuditMarkdown, formatAuditPretty, formatGithub, formatJson, formatMarkdown, formatPretty, formatSarif, isCrawlable, routeFromFilePath, routeFromUrl, sameSurface, shouldFail, shouldIgnore, snapshotFromDir, snapshotFromGitRef, snapshotFromOrigin, snapshotFromPage, summarize, withGuidance };
|
package/dist/index.js
CHANGED
|
@@ -1535,7 +1535,7 @@ ${cards || "<p>No issues found.</p>"}
|
|
|
1535
1535
|
|
|
1536
1536
|
// src/snapshot.ts
|
|
1537
1537
|
import { execFile } from "child_process";
|
|
1538
|
-
import { readdir, readFile } from "fs/promises";
|
|
1538
|
+
import { access, readdir, readFile } from "fs/promises";
|
|
1539
1539
|
import { join, relative, sep } from "path";
|
|
1540
1540
|
import { promisify } from "util";
|
|
1541
1541
|
function routeFromFilePath(root, filePath) {
|
|
@@ -1665,12 +1665,20 @@ async function snapshotFromDir(dir, config = {}) {
|
|
|
1665
1665
|
const llms = await readFile(join(dir, "llms.txt"), "utf8").catch(() => null);
|
|
1666
1666
|
if (llms !== null) site.llmsTxt = extractLlmsTxt(llms);
|
|
1667
1667
|
const origin = site.origin ?? "https://pagetrace.invalid";
|
|
1668
|
-
await resolveBrokenLinks(pages, links, origin
|
|
1668
|
+
await resolveBrokenLinks(pages, links, origin, {
|
|
1669
|
+
includeAssets: config.verifyAll,
|
|
1670
|
+
// The HTML route set is authoritative here, so a missing page is simply
|
|
1671
|
+
// broken. An asset is a file this crawl never walked, so it gets a stat.
|
|
1672
|
+
verify: config.verifyAll ? async (target) => ASSET_PATH.test(target) ? !await access(join(dir, target)).then(
|
|
1673
|
+
() => true,
|
|
1674
|
+
() => false
|
|
1675
|
+
) : true : void 0
|
|
1676
|
+
});
|
|
1669
1677
|
if (config.checkExternal) await resolveDeadExternal(pages, links, origin, 5);
|
|
1670
1678
|
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1671
1679
|
}
|
|
1672
1680
|
var ASSET_PATH = /\.(?!html?$)[a-z0-9]+$/i;
|
|
1673
|
-
function linkTarget(href, from, origin) {
|
|
1681
|
+
function linkTarget(href, from, origin, includeAssets = false) {
|
|
1674
1682
|
let url;
|
|
1675
1683
|
try {
|
|
1676
1684
|
url = new URL(href, `${origin}${from === "/" ? "" : from}/`);
|
|
@@ -1678,27 +1686,39 @@ function linkTarget(href, from, origin) {
|
|
|
1678
1686
|
return null;
|
|
1679
1687
|
}
|
|
1680
1688
|
if (url.origin !== origin) return null;
|
|
1681
|
-
if (ASSET_PATH.test(url.pathname)) return null;
|
|
1682
|
-
return routeFromUrl(url.href);
|
|
1689
|
+
if (!includeAssets && ASSET_PATH.test(url.pathname)) return null;
|
|
1690
|
+
return ASSET_PATH.test(url.pathname) ? url.pathname : routeFromUrl(url.href);
|
|
1683
1691
|
}
|
|
1684
|
-
var MAX_LINK_CHECKS =
|
|
1685
|
-
async function resolveBrokenLinks(pages, links, origin,
|
|
1692
|
+
var MAX_LINK_CHECKS = 1e3;
|
|
1693
|
+
async function resolveBrokenLinks(pages, links, origin, options = {}) {
|
|
1694
|
+
const { includeAssets = false, verify, concurrency = 5 } = options;
|
|
1686
1695
|
const known = new Set(Object.keys(pages));
|
|
1687
1696
|
const candidates = /* @__PURE__ */ new Map();
|
|
1688
1697
|
for (const [route, hrefs] of Object.entries(links)) {
|
|
1689
1698
|
const missing = [
|
|
1690
1699
|
...new Set(
|
|
1691
|
-
hrefs.map((href) => linkTarget(href, route, origin)).filter((target) => target !== null && !known.has(target))
|
|
1700
|
+
hrefs.map((href) => linkTarget(href, route, origin, includeAssets)).filter((target) => target !== null && !known.has(target))
|
|
1692
1701
|
)
|
|
1693
1702
|
];
|
|
1694
1703
|
if (missing.length > 0) candidates.set(route, missing);
|
|
1695
1704
|
}
|
|
1696
1705
|
const broken = /* @__PURE__ */ new Set();
|
|
1697
1706
|
if (verify) {
|
|
1698
|
-
const
|
|
1699
|
-
|
|
1700
|
-
|
|
1707
|
+
const unique = [...new Set([...candidates.values()].flat())];
|
|
1708
|
+
const queue = unique.slice(0, MAX_LINK_CHECKS);
|
|
1709
|
+
if (unique.length > queue.length) {
|
|
1710
|
+
console.error(
|
|
1711
|
+
`pagetrace: ${unique.length} link targets to confirm, checking the first ${MAX_LINK_CHECKS}.`
|
|
1712
|
+
);
|
|
1701
1713
|
}
|
|
1714
|
+
await Promise.all(
|
|
1715
|
+
Array.from({ length: Math.min(Math.max(1, concurrency), queue.length) }, async () => {
|
|
1716
|
+
while (queue.length > 0) {
|
|
1717
|
+
const target = queue.shift();
|
|
1718
|
+
if (await verify(target)) broken.add(target);
|
|
1719
|
+
}
|
|
1720
|
+
})
|
|
1721
|
+
);
|
|
1702
1722
|
}
|
|
1703
1723
|
for (const [route, missing] of candidates) {
|
|
1704
1724
|
const confirmed = verify ? missing.filter((target) => broken.has(target)) : missing;
|
|
@@ -1762,6 +1782,39 @@ async function resolveDeadExternal(pages, links, origin, concurrency, timeoutMs)
|
|
|
1762
1782
|
if (found.length > 0) pages[route].deadExternal = found;
|
|
1763
1783
|
}
|
|
1764
1784
|
}
|
|
1785
|
+
async function snapshotFromPage(pageUrl, options = {}) {
|
|
1786
|
+
const target = new URL(pageUrl);
|
|
1787
|
+
const site = {
|
|
1788
|
+
origin: options.siteUrl ? new URL(options.siteUrl).origin : target.origin,
|
|
1789
|
+
robotsTxt: null,
|
|
1790
|
+
llmsTxt: null,
|
|
1791
|
+
sitemap: null
|
|
1792
|
+
};
|
|
1793
|
+
const doc = await fetchDoc(target.href, options.timeout);
|
|
1794
|
+
if (doc === null) throw new Error(`${target.href} answered 404 \u2014 nothing to check.`);
|
|
1795
|
+
const route = routeFromUrl(target.href);
|
|
1796
|
+
const page = extractPage(doc.text, route);
|
|
1797
|
+
const landed = !doc.url ? route : originOf2(doc.url) === target.origin ? routeFromUrl(doc.url) : doc.url;
|
|
1798
|
+
page.redirectsTo = landed === route ? null : landed;
|
|
1799
|
+
const pages = { [route]: page };
|
|
1800
|
+
const links = { [route]: extractLinks(doc.text) };
|
|
1801
|
+
const origin = site.origin ?? target.origin;
|
|
1802
|
+
await resolveBrokenLinks(pages, links, origin, {
|
|
1803
|
+
includeAssets: options.verifyAll,
|
|
1804
|
+
concurrency: options.concurrency,
|
|
1805
|
+
verify: async (candidate) => {
|
|
1806
|
+
try {
|
|
1807
|
+
return await fetchDoc(new URL(candidate, target.origin).href, options.timeout) === null;
|
|
1808
|
+
} catch {
|
|
1809
|
+
return false;
|
|
1810
|
+
}
|
|
1811
|
+
}
|
|
1812
|
+
});
|
|
1813
|
+
if (options.checkExternal) {
|
|
1814
|
+
await resolveDeadExternal(pages, links, origin, Math.max(1, options.concurrency ?? 5), options.timeout);
|
|
1815
|
+
}
|
|
1816
|
+
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1817
|
+
}
|
|
1765
1818
|
var MAX_NESTED_SITEMAPS = 50;
|
|
1766
1819
|
async function snapshotFromOrigin(origin, options = {}) {
|
|
1767
1820
|
const base = new URL(origin);
|
|
@@ -1843,11 +1896,15 @@ async function snapshotFromOrigin(origin, options = {}) {
|
|
|
1843
1896
|
}
|
|
1844
1897
|
})
|
|
1845
1898
|
);
|
|
1846
|
-
await resolveBrokenLinks(pages, links, base.origin,
|
|
1847
|
-
|
|
1848
|
-
|
|
1849
|
-
|
|
1850
|
-
|
|
1899
|
+
await resolveBrokenLinks(pages, links, base.origin, {
|
|
1900
|
+
includeAssets: options.verifyAll,
|
|
1901
|
+
concurrency,
|
|
1902
|
+
verify: async (route) => {
|
|
1903
|
+
try {
|
|
1904
|
+
return await fetchDoc(new URL(route, base).href, timeout) === null;
|
|
1905
|
+
} catch {
|
|
1906
|
+
return false;
|
|
1907
|
+
}
|
|
1851
1908
|
}
|
|
1852
1909
|
});
|
|
1853
1910
|
if (options.checkExternal) {
|
|
@@ -1893,6 +1950,7 @@ export {
|
|
|
1893
1950
|
snapshotFromDir,
|
|
1894
1951
|
snapshotFromGitRef,
|
|
1895
1952
|
snapshotFromOrigin,
|
|
1953
|
+
snapshotFromPage,
|
|
1896
1954
|
summarize,
|
|
1897
1955
|
withGuidance
|
|
1898
1956
|
};
|