pagetrace 0.12.0 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/README.md +8 -0
- package/dist/cli.cjs +69 -4
- package/dist/cli.js +69 -4
- package/dist/index.cjs +31 -0
- package/dist/index.d.cts +9 -1
- package/dist/index.d.ts +9 -1
- package/dist/index.js +30 -0
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,17 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
|
6
6
|
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
7
|
While the version is below 1.0.0, breaking changes ship in a minor release.
|
|
8
8
|
|
|
9
|
+
## [0.13.0] - 2026-09-08
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- `pagetrace page <url>` checks a single page: every link on it confirmed with a
|
|
14
|
+
real request, external ones included by default, plus that page's own rules.
|
|
15
|
+
A crawl trusts its route set and only verifies what is missing from it; with
|
|
16
|
+
one page there is no route set to trust, so everything is checked. Passing a
|
|
17
|
+
deep URL to `--url` had crawled the whole site from its origin and ignored the
|
|
18
|
+
path, which is not what it looked like it did.
|
|
19
|
+
|
|
9
20
|
## [0.12.0] - 2026-09-08
|
|
10
21
|
|
|
11
22
|
### Added
|
|
@@ -359,6 +370,7 @@ Initial release. `snapshot`, `check` and `audit` commands; filesystem and HTTP
|
|
|
359
370
|
crawling; diff classified by transition; absolute, cross-page and hreflang audit
|
|
360
371
|
rules; pretty, JSON, markdown, GitHub and HTML reporters.
|
|
361
372
|
|
|
373
|
+
[0.13.0]: https://github.com/shyamexe/pagetrace/compare/v0.12.0...v0.13.0
|
|
362
374
|
[0.12.0]: https://github.com/shyamexe/pagetrace/compare/v0.11.0...v0.12.0
|
|
363
375
|
[0.11.0]: https://github.com/shyamexe/pagetrace/compare/v0.10.0...v0.11.0
|
|
364
376
|
[0.10.0]: https://github.com/shyamexe/pagetrace/compare/v0.9.1...v0.10.0
|
package/README.md
CHANGED
|
@@ -166,6 +166,14 @@ pagetrace update # install it globally
|
|
|
166
166
|
| `og.removed` / `hreflang.removed` | warn | Social or i18n tags dropped |
|
|
167
167
|
| `title.changed` | info | Ordinary copy edit |
|
|
168
168
|
|
|
169
|
+
One page, checked on its own:
|
|
170
|
+
|
|
171
|
+
```bash
|
|
172
|
+
npx pagetrace page https://example.com/blog/my-post
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
No sitemap, no crawl: it fetches that URL and confirms every link on it with a real request, external links included by default (`--no-external` to skip them). Site-wide rules are left to `audit` — a single-page check never looks at robots.txt, so it does not get to say whether one exists.
|
|
176
|
+
|
|
169
177
|
Broken links have a command of their own, when that is the only question you have:
|
|
170
178
|
|
|
171
179
|
```bash
|
package/dist/cli.cjs
CHANGED
|
@@ -1795,6 +1795,35 @@ async function resolveDeadExternal(pages, links, origin, concurrency, timeoutMs)
|
|
|
1795
1795
|
if (found.length > 0) pages[route].deadExternal = found;
|
|
1796
1796
|
}
|
|
1797
1797
|
}
|
|
1798
|
+
async function snapshotFromPage(pageUrl, options = {}) {
|
|
1799
|
+
const target = new URL(pageUrl);
|
|
1800
|
+
const site = {
|
|
1801
|
+
origin: options.siteUrl ? new URL(options.siteUrl).origin : target.origin,
|
|
1802
|
+
robotsTxt: null,
|
|
1803
|
+
llmsTxt: null,
|
|
1804
|
+
sitemap: null
|
|
1805
|
+
};
|
|
1806
|
+
const doc = await fetchDoc(target.href, options.timeout);
|
|
1807
|
+
if (doc === null) throw new Error(`${target.href} answered 404 \u2014 nothing to check.`);
|
|
1808
|
+
const route = routeFromUrl(target.href);
|
|
1809
|
+
const page = extractPage(doc.text, route);
|
|
1810
|
+
const landed = !doc.url ? route : originOf2(doc.url) === target.origin ? routeFromUrl(doc.url) : doc.url;
|
|
1811
|
+
page.redirectsTo = landed === route ? null : landed;
|
|
1812
|
+
const pages = { [route]: page };
|
|
1813
|
+
const links = { [route]: extractLinks(doc.text) };
|
|
1814
|
+
const origin = site.origin ?? target.origin;
|
|
1815
|
+
await resolveBrokenLinks(pages, links, origin, async (candidate) => {
|
|
1816
|
+
try {
|
|
1817
|
+
return await fetchDoc(new URL(candidate, target.origin).href, options.timeout) === null;
|
|
1818
|
+
} catch {
|
|
1819
|
+
return false;
|
|
1820
|
+
}
|
|
1821
|
+
});
|
|
1822
|
+
if (options.checkExternal) {
|
|
1823
|
+
await resolveDeadExternal(pages, links, origin, Math.max(1, options.concurrency ?? 5), options.timeout);
|
|
1824
|
+
}
|
|
1825
|
+
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1826
|
+
}
|
|
1798
1827
|
var MAX_NESTED_SITEMAPS = 50;
|
|
1799
1828
|
async function snapshotFromOrigin(origin, options = {}) {
|
|
1800
1829
|
const base = new URL(origin);
|
|
@@ -2044,6 +2073,42 @@ cli.command("audit", "Audit a site as it stands, with explanations and fixes").o
|
|
|
2044
2073
|
process.exitCode = EXIT_FINDINGS;
|
|
2045
2074
|
}
|
|
2046
2075
|
});
|
|
2076
|
+
cli.command("page <url>", "Check one page: every link on it, and its own surface").option("--external", "Check links that leave the site", { default: true }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "error" }).action(async (url, flags) => {
|
|
2077
|
+
const failOn = parseFailOn(flags.failOn, true);
|
|
2078
|
+
const config = await loadConfig(flags.config);
|
|
2079
|
+
const snapshot = await snapshotFromPage(url, {
|
|
2080
|
+
...config,
|
|
2081
|
+
concurrency: flags.concurrency,
|
|
2082
|
+
checkExternal: flags.external
|
|
2083
|
+
});
|
|
2084
|
+
const [page] = Object.values(snapshot.pages);
|
|
2085
|
+
const platform = detectPlatform([page.generator], Object.values(page.og));
|
|
2086
|
+
const findings = applyConfig(auditSnapshot(snapshot, config), config).filter((f) => f.route !== null).map((f) => withGuidance(f, platform));
|
|
2087
|
+
if (findings.length === 0 && !flags.out) {
|
|
2088
|
+
const checked = (page.brokenLinks?.length ?? 0) + (page.deadExternal?.length ?? 0);
|
|
2089
|
+
console.log(import_picocolors2.default.green(`${url} looks sound \u2014 no findings, no dead links.`));
|
|
2090
|
+
if (checked > 0) console.log(import_picocolors2.default.dim("(unreachable links were treated as unknown, not dead)"));
|
|
2091
|
+
return;
|
|
2092
|
+
}
|
|
2093
|
+
const groups = aggregate(findings);
|
|
2094
|
+
const meta = {
|
|
2095
|
+
target: url,
|
|
2096
|
+
platform,
|
|
2097
|
+
pageCount: 1,
|
|
2098
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString().slice(0, 10)
|
|
2099
|
+
};
|
|
2100
|
+
const output = flags.format === "json" ? JSON.stringify({ schemaVersion: 1, meta, summary: summarize(findings), groups }, null, 2) : flags.format === "markdown" ? formatAuditMarkdown(groups, meta) : formatAuditPretty(groups, meta, process.stdout.columns);
|
|
2101
|
+
if (flags.out) {
|
|
2102
|
+
await (0, import_promises2.writeFile)(flags.out, `${output}
|
|
2103
|
+
`, "utf8");
|
|
2104
|
+
console.log(import_picocolors2.default.green(`Wrote ${flags.out}.`));
|
|
2105
|
+
} else {
|
|
2106
|
+
console.log(output);
|
|
2107
|
+
}
|
|
2108
|
+
if (failOn !== "never" && shouldFail(findings, failOn)) {
|
|
2109
|
+
process.exitCode = EXIT_FINDINGS;
|
|
2110
|
+
}
|
|
2111
|
+
});
|
|
2047
2112
|
cli.command("links", "Find internal links that point at no page").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "error" }).action(async (flags) => {
|
|
2048
2113
|
const failOn = parseFailOn(flags.failOn, true);
|
|
2049
2114
|
const config = await loadConfig(flags.config);
|
|
@@ -2114,11 +2179,11 @@ Commit both files, then run \`pagetrace check ${flags.dir ? `--dir ${flags.dir}`
|
|
|
2114
2179
|
});
|
|
2115
2180
|
cli.command("update", "Check npm for a newer pagetrace and install it").option("--check", "Only report whether an update exists").action(async (flags) => {
|
|
2116
2181
|
const latest = await latestVersion("pagetrace");
|
|
2117
|
-
if (!isNewer(latest, "0.
|
|
2118
|
-
console.log(import_picocolors2.default.green(`pagetrace ${"0.
|
|
2182
|
+
if (!isNewer(latest, "0.13.0")) {
|
|
2183
|
+
console.log(import_picocolors2.default.green(`pagetrace ${"0.13.0"} is the latest version.`));
|
|
2119
2184
|
return;
|
|
2120
2185
|
}
|
|
2121
|
-
console.log(import_picocolors2.default.yellow(`Update available: ${"0.
|
|
2186
|
+
console.log(import_picocolors2.default.yellow(`Update available: ${"0.13.0"} \u2192 ${latest}`));
|
|
2122
2187
|
if (flags.check) return;
|
|
2123
2188
|
if (process.argv[1]?.startsWith(process.cwd())) {
|
|
2124
2189
|
console.log(
|
|
@@ -2136,7 +2201,7 @@ cli.command("update", "Check npm for a newer pagetrace and install it").option("
|
|
|
2136
2201
|
console.log(import_picocolors2.default.green(`Updated to pagetrace ${latest}.`));
|
|
2137
2202
|
});
|
|
2138
2203
|
cli.help();
|
|
2139
|
-
cli.version("0.
|
|
2204
|
+
cli.version("0.13.0");
|
|
2140
2205
|
async function main() {
|
|
2141
2206
|
try {
|
|
2142
2207
|
cli.parse(process.argv, { run: false });
|
package/dist/cli.js
CHANGED
|
@@ -1772,6 +1772,35 @@ async function resolveDeadExternal(pages, links, origin, concurrency, timeoutMs)
|
|
|
1772
1772
|
if (found.length > 0) pages[route].deadExternal = found;
|
|
1773
1773
|
}
|
|
1774
1774
|
}
|
|
1775
|
+
async function snapshotFromPage(pageUrl, options = {}) {
|
|
1776
|
+
const target = new URL(pageUrl);
|
|
1777
|
+
const site = {
|
|
1778
|
+
origin: options.siteUrl ? new URL(options.siteUrl).origin : target.origin,
|
|
1779
|
+
robotsTxt: null,
|
|
1780
|
+
llmsTxt: null,
|
|
1781
|
+
sitemap: null
|
|
1782
|
+
};
|
|
1783
|
+
const doc = await fetchDoc(target.href, options.timeout);
|
|
1784
|
+
if (doc === null) throw new Error(`${target.href} answered 404 \u2014 nothing to check.`);
|
|
1785
|
+
const route = routeFromUrl(target.href);
|
|
1786
|
+
const page = extractPage(doc.text, route);
|
|
1787
|
+
const landed = !doc.url ? route : originOf2(doc.url) === target.origin ? routeFromUrl(doc.url) : doc.url;
|
|
1788
|
+
page.redirectsTo = landed === route ? null : landed;
|
|
1789
|
+
const pages = { [route]: page };
|
|
1790
|
+
const links = { [route]: extractLinks(doc.text) };
|
|
1791
|
+
const origin = site.origin ?? target.origin;
|
|
1792
|
+
await resolveBrokenLinks(pages, links, origin, async (candidate) => {
|
|
1793
|
+
try {
|
|
1794
|
+
return await fetchDoc(new URL(candidate, target.origin).href, options.timeout) === null;
|
|
1795
|
+
} catch {
|
|
1796
|
+
return false;
|
|
1797
|
+
}
|
|
1798
|
+
});
|
|
1799
|
+
if (options.checkExternal) {
|
|
1800
|
+
await resolveDeadExternal(pages, links, origin, Math.max(1, options.concurrency ?? 5), options.timeout);
|
|
1801
|
+
}
|
|
1802
|
+
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1803
|
+
}
|
|
1775
1804
|
var MAX_NESTED_SITEMAPS = 50;
|
|
1776
1805
|
async function snapshotFromOrigin(origin, options = {}) {
|
|
1777
1806
|
const base = new URL(origin);
|
|
@@ -2021,6 +2050,42 @@ cli.command("audit", "Audit a site as it stands, with explanations and fixes").o
|
|
|
2021
2050
|
process.exitCode = EXIT_FINDINGS;
|
|
2022
2051
|
}
|
|
2023
2052
|
});
|
|
2053
|
+
cli.command("page <url>", "Check one page: every link on it, and its own surface").option("--external", "Check links that leave the site", { default: true }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "error" }).action(async (url, flags) => {
|
|
2054
|
+
const failOn = parseFailOn(flags.failOn, true);
|
|
2055
|
+
const config = await loadConfig(flags.config);
|
|
2056
|
+
const snapshot = await snapshotFromPage(url, {
|
|
2057
|
+
...config,
|
|
2058
|
+
concurrency: flags.concurrency,
|
|
2059
|
+
checkExternal: flags.external
|
|
2060
|
+
});
|
|
2061
|
+
const [page] = Object.values(snapshot.pages);
|
|
2062
|
+
const platform = detectPlatform([page.generator], Object.values(page.og));
|
|
2063
|
+
const findings = applyConfig(auditSnapshot(snapshot, config), config).filter((f) => f.route !== null).map((f) => withGuidance(f, platform));
|
|
2064
|
+
if (findings.length === 0 && !flags.out) {
|
|
2065
|
+
const checked = (page.brokenLinks?.length ?? 0) + (page.deadExternal?.length ?? 0);
|
|
2066
|
+
console.log(pc2.green(`${url} looks sound \u2014 no findings, no dead links.`));
|
|
2067
|
+
if (checked > 0) console.log(pc2.dim("(unreachable links were treated as unknown, not dead)"));
|
|
2068
|
+
return;
|
|
2069
|
+
}
|
|
2070
|
+
const groups = aggregate(findings);
|
|
2071
|
+
const meta = {
|
|
2072
|
+
target: url,
|
|
2073
|
+
platform,
|
|
2074
|
+
pageCount: 1,
|
|
2075
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString().slice(0, 10)
|
|
2076
|
+
};
|
|
2077
|
+
const output = flags.format === "json" ? JSON.stringify({ schemaVersion: 1, meta, summary: summarize(findings), groups }, null, 2) : flags.format === "markdown" ? formatAuditMarkdown(groups, meta) : formatAuditPretty(groups, meta, process.stdout.columns);
|
|
2078
|
+
if (flags.out) {
|
|
2079
|
+
await writeFile(flags.out, `${output}
|
|
2080
|
+
`, "utf8");
|
|
2081
|
+
console.log(pc2.green(`Wrote ${flags.out}.`));
|
|
2082
|
+
} else {
|
|
2083
|
+
console.log(output);
|
|
2084
|
+
}
|
|
2085
|
+
if (failOn !== "never" && shouldFail(findings, failOn)) {
|
|
2086
|
+
process.exitCode = EXIT_FINDINGS;
|
|
2087
|
+
}
|
|
2088
|
+
});
|
|
2024
2089
|
cli.command("links", "Find internal links that point at no page").option("--url <origin>", "Live origin to crawl").option("--dir <dir>", "Directory of built HTML").option("--limit <n>", "Max pages to crawl", { default: 200 }).option("--concurrency <n>", "Parallel requests", { default: 5 }).option("--ignore-robots", "Crawl paths that robots.txt disallows").option("--external", "Also check links that leave the site").option("--config <file>", "Config file", { default: DEFAULT_CONFIG }).option("--format <format>", "pretty | json | markdown", { default: "pretty" }).option("--out <file>", "Write the report to a file instead of stdout").option("--fail-on <severity>", "error | warn | info | never", { default: "error" }).action(async (flags) => {
|
|
2025
2090
|
const failOn = parseFailOn(flags.failOn, true);
|
|
2026
2091
|
const config = await loadConfig(flags.config);
|
|
@@ -2091,11 +2156,11 @@ Commit both files, then run \`pagetrace check ${flags.dir ? `--dir ${flags.dir}`
|
|
|
2091
2156
|
});
|
|
2092
2157
|
cli.command("update", "Check npm for a newer pagetrace and install it").option("--check", "Only report whether an update exists").action(async (flags) => {
|
|
2093
2158
|
const latest = await latestVersion("pagetrace");
|
|
2094
|
-
if (!isNewer(latest, "0.
|
|
2095
|
-
console.log(pc2.green(`pagetrace ${"0.
|
|
2159
|
+
if (!isNewer(latest, "0.13.0")) {
|
|
2160
|
+
console.log(pc2.green(`pagetrace ${"0.13.0"} is the latest version.`));
|
|
2096
2161
|
return;
|
|
2097
2162
|
}
|
|
2098
|
-
console.log(pc2.yellow(`Update available: ${"0.
|
|
2163
|
+
console.log(pc2.yellow(`Update available: ${"0.13.0"} \u2192 ${latest}`));
|
|
2099
2164
|
if (flags.check) return;
|
|
2100
2165
|
if (process.argv[1]?.startsWith(process.cwd())) {
|
|
2101
2166
|
console.log(
|
|
@@ -2113,7 +2178,7 @@ cli.command("update", "Check npm for a newer pagetrace and install it").option("
|
|
|
2113
2178
|
console.log(pc2.green(`Updated to pagetrace ${latest}.`));
|
|
2114
2179
|
});
|
|
2115
2180
|
cli.help();
|
|
2116
|
-
cli.version("0.
|
|
2181
|
+
cli.version("0.13.0");
|
|
2117
2182
|
async function main() {
|
|
2118
2183
|
try {
|
|
2119
2184
|
cli.parse(process.argv, { run: false });
|
package/dist/index.cjs
CHANGED
|
@@ -67,6 +67,7 @@ __export(src_exports, {
|
|
|
67
67
|
snapshotFromDir: () => snapshotFromDir,
|
|
68
68
|
snapshotFromGitRef: () => snapshotFromGitRef,
|
|
69
69
|
snapshotFromOrigin: () => snapshotFromOrigin,
|
|
70
|
+
snapshotFromPage: () => snapshotFromPage,
|
|
70
71
|
summarize: () => summarize,
|
|
71
72
|
withGuidance: () => withGuidance
|
|
72
73
|
});
|
|
@@ -1836,6 +1837,35 @@ async function resolveDeadExternal(pages, links, origin, concurrency, timeoutMs)
|
|
|
1836
1837
|
if (found.length > 0) pages[route].deadExternal = found;
|
|
1837
1838
|
}
|
|
1838
1839
|
}
|
|
1840
|
+
async function snapshotFromPage(pageUrl, options = {}) {
|
|
1841
|
+
const target = new URL(pageUrl);
|
|
1842
|
+
const site = {
|
|
1843
|
+
origin: options.siteUrl ? new URL(options.siteUrl).origin : target.origin,
|
|
1844
|
+
robotsTxt: null,
|
|
1845
|
+
llmsTxt: null,
|
|
1846
|
+
sitemap: null
|
|
1847
|
+
};
|
|
1848
|
+
const doc = await fetchDoc(target.href, options.timeout);
|
|
1849
|
+
if (doc === null) throw new Error(`${target.href} answered 404 \u2014 nothing to check.`);
|
|
1850
|
+
const route = routeFromUrl(target.href);
|
|
1851
|
+
const page = extractPage(doc.text, route);
|
|
1852
|
+
const landed = !doc.url ? route : originOf2(doc.url) === target.origin ? routeFromUrl(doc.url) : doc.url;
|
|
1853
|
+
page.redirectsTo = landed === route ? null : landed;
|
|
1854
|
+
const pages = { [route]: page };
|
|
1855
|
+
const links = { [route]: extractLinks(doc.text) };
|
|
1856
|
+
const origin = site.origin ?? target.origin;
|
|
1857
|
+
await resolveBrokenLinks(pages, links, origin, async (candidate) => {
|
|
1858
|
+
try {
|
|
1859
|
+
return await fetchDoc(new URL(candidate, target.origin).href, options.timeout) === null;
|
|
1860
|
+
} catch {
|
|
1861
|
+
return false;
|
|
1862
|
+
}
|
|
1863
|
+
});
|
|
1864
|
+
if (options.checkExternal) {
|
|
1865
|
+
await resolveDeadExternal(pages, links, origin, Math.max(1, options.concurrency ?? 5), options.timeout);
|
|
1866
|
+
}
|
|
1867
|
+
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1868
|
+
}
|
|
1839
1869
|
var MAX_NESTED_SITEMAPS = 50;
|
|
1840
1870
|
async function snapshotFromOrigin(origin, options = {}) {
|
|
1841
1871
|
const base = new URL(origin);
|
|
@@ -1968,6 +1998,7 @@ async function snapshotFromOrigin(origin, options = {}) {
|
|
|
1968
1998
|
snapshotFromDir,
|
|
1969
1999
|
snapshotFromGitRef,
|
|
1970
2000
|
snapshotFromOrigin,
|
|
2001
|
+
snapshotFromPage,
|
|
1971
2002
|
summarize,
|
|
1972
2003
|
withGuidance
|
|
1973
2004
|
});
|
package/dist/index.d.cts
CHANGED
|
@@ -317,6 +317,14 @@ declare function sameSurface(a: Snapshot, b: Snapshot): boolean;
|
|
|
317
317
|
declare function shouldIgnore(route: string, patterns?: string[]): boolean;
|
|
318
318
|
/** Build a snapshot from a directory of pre-rendered HTML (next export, dist, out). */
|
|
319
319
|
declare function snapshotFromDir(dir: string, config?: Config): Promise<Snapshot>;
|
|
320
|
+
/**
|
|
321
|
+
* One page, checked properly. No sitemap, no route discovery: the snapshot
|
|
322
|
+
* holds a single page, so every link on it is a candidate and every candidate
|
|
323
|
+
* is confirmed with a real request. That is the opposite trade from a crawl,
|
|
324
|
+
* and the right one here — one page's worth of links is a bounded cost, and
|
|
325
|
+
* "is this page's linking sound" is a question a crawl answers slowly.
|
|
326
|
+
*/
|
|
327
|
+
declare function snapshotFromPage(pageUrl: string, options?: CrawlOptions): Promise<Snapshot>;
|
|
320
328
|
interface CrawlOptions extends Config {
|
|
321
329
|
/** Cap the number of pages fetched. */
|
|
322
330
|
limit?: number;
|
|
@@ -328,4 +336,4 @@ interface CrawlOptions extends Config {
|
|
|
328
336
|
/** Build a snapshot by fetching a live origin, discovering routes via sitemap. */
|
|
329
337
|
declare function snapshotFromOrigin(origin: string, options?: CrawlOptions): Promise<Snapshot>;
|
|
330
338
|
|
|
331
|
-
export { type Aggregate, type AuditMeta, type Config, DEFAULT_AI_AGENTS, type Finding, GUIDANCE, type Guidance, type JsonLdEntity, type PageFingerprint, type Platform, RICH_RESULT_RULES, type Severity, type SiteFingerprint, type Snapshot, aggregate, applyConfig, auditCrossPage, auditHreflang, auditPage, auditSite, auditSnapshot, detectPlatform, diffPage, diffSite, diffSnapshots, extractJsonLd, extractLinks, extractLlmsTxt, extractPage, extractRobotsTxt, extractSitemapUrls, formatAuditHtml, formatAuditMarkdown, formatAuditPretty, formatGithub, formatJson, formatMarkdown, formatPretty, formatSarif, isCrawlable, routeFromFilePath, routeFromUrl, sameSurface, shouldFail, shouldIgnore, snapshotFromDir, snapshotFromGitRef, snapshotFromOrigin, summarize, withGuidance };
|
|
339
|
+
export { type Aggregate, type AuditMeta, type Config, DEFAULT_AI_AGENTS, type Finding, GUIDANCE, type Guidance, type JsonLdEntity, type PageFingerprint, type Platform, RICH_RESULT_RULES, type Severity, type SiteFingerprint, type Snapshot, aggregate, applyConfig, auditCrossPage, auditHreflang, auditPage, auditSite, auditSnapshot, detectPlatform, diffPage, diffSite, diffSnapshots, extractJsonLd, extractLinks, extractLlmsTxt, extractPage, extractRobotsTxt, extractSitemapUrls, formatAuditHtml, formatAuditMarkdown, formatAuditPretty, formatGithub, formatJson, formatMarkdown, formatPretty, formatSarif, isCrawlable, routeFromFilePath, routeFromUrl, sameSurface, shouldFail, shouldIgnore, snapshotFromDir, snapshotFromGitRef, snapshotFromOrigin, snapshotFromPage, summarize, withGuidance };
|
package/dist/index.d.ts
CHANGED
|
@@ -317,6 +317,14 @@ declare function sameSurface(a: Snapshot, b: Snapshot): boolean;
|
|
|
317
317
|
declare function shouldIgnore(route: string, patterns?: string[]): boolean;
|
|
318
318
|
/** Build a snapshot from a directory of pre-rendered HTML (next export, dist, out). */
|
|
319
319
|
declare function snapshotFromDir(dir: string, config?: Config): Promise<Snapshot>;
|
|
320
|
+
/**
|
|
321
|
+
* One page, checked properly. No sitemap, no route discovery: the snapshot
|
|
322
|
+
* holds a single page, so every link on it is a candidate and every candidate
|
|
323
|
+
* is confirmed with a real request. That is the opposite trade from a crawl,
|
|
324
|
+
* and the right one here — one page's worth of links is a bounded cost, and
|
|
325
|
+
* "is this page's linking sound" is a question a crawl answers slowly.
|
|
326
|
+
*/
|
|
327
|
+
declare function snapshotFromPage(pageUrl: string, options?: CrawlOptions): Promise<Snapshot>;
|
|
320
328
|
interface CrawlOptions extends Config {
|
|
321
329
|
/** Cap the number of pages fetched. */
|
|
322
330
|
limit?: number;
|
|
@@ -328,4 +336,4 @@ interface CrawlOptions extends Config {
|
|
|
328
336
|
/** Build a snapshot by fetching a live origin, discovering routes via sitemap. */
|
|
329
337
|
declare function snapshotFromOrigin(origin: string, options?: CrawlOptions): Promise<Snapshot>;
|
|
330
338
|
|
|
331
|
-
export { type Aggregate, type AuditMeta, type Config, DEFAULT_AI_AGENTS, type Finding, GUIDANCE, type Guidance, type JsonLdEntity, type PageFingerprint, type Platform, RICH_RESULT_RULES, type Severity, type SiteFingerprint, type Snapshot, aggregate, applyConfig, auditCrossPage, auditHreflang, auditPage, auditSite, auditSnapshot, detectPlatform, diffPage, diffSite, diffSnapshots, extractJsonLd, extractLinks, extractLlmsTxt, extractPage, extractRobotsTxt, extractSitemapUrls, formatAuditHtml, formatAuditMarkdown, formatAuditPretty, formatGithub, formatJson, formatMarkdown, formatPretty, formatSarif, isCrawlable, routeFromFilePath, routeFromUrl, sameSurface, shouldFail, shouldIgnore, snapshotFromDir, snapshotFromGitRef, snapshotFromOrigin, summarize, withGuidance };
|
|
339
|
+
export { type Aggregate, type AuditMeta, type Config, DEFAULT_AI_AGENTS, type Finding, GUIDANCE, type Guidance, type JsonLdEntity, type PageFingerprint, type Platform, RICH_RESULT_RULES, type Severity, type SiteFingerprint, type Snapshot, aggregate, applyConfig, auditCrossPage, auditHreflang, auditPage, auditSite, auditSnapshot, detectPlatform, diffPage, diffSite, diffSnapshots, extractJsonLd, extractLinks, extractLlmsTxt, extractPage, extractRobotsTxt, extractSitemapUrls, formatAuditHtml, formatAuditMarkdown, formatAuditPretty, formatGithub, formatJson, formatMarkdown, formatPretty, formatSarif, isCrawlable, routeFromFilePath, routeFromUrl, sameSurface, shouldFail, shouldIgnore, snapshotFromDir, snapshotFromGitRef, snapshotFromOrigin, snapshotFromPage, summarize, withGuidance };
|
package/dist/index.js
CHANGED
|
@@ -1762,6 +1762,35 @@ async function resolveDeadExternal(pages, links, origin, concurrency, timeoutMs)
|
|
|
1762
1762
|
if (found.length > 0) pages[route].deadExternal = found;
|
|
1763
1763
|
}
|
|
1764
1764
|
}
|
|
1765
|
+
async function snapshotFromPage(pageUrl, options = {}) {
|
|
1766
|
+
const target = new URL(pageUrl);
|
|
1767
|
+
const site = {
|
|
1768
|
+
origin: options.siteUrl ? new URL(options.siteUrl).origin : target.origin,
|
|
1769
|
+
robotsTxt: null,
|
|
1770
|
+
llmsTxt: null,
|
|
1771
|
+
sitemap: null
|
|
1772
|
+
};
|
|
1773
|
+
const doc = await fetchDoc(target.href, options.timeout);
|
|
1774
|
+
if (doc === null) throw new Error(`${target.href} answered 404 \u2014 nothing to check.`);
|
|
1775
|
+
const route = routeFromUrl(target.href);
|
|
1776
|
+
const page = extractPage(doc.text, route);
|
|
1777
|
+
const landed = !doc.url ? route : originOf2(doc.url) === target.origin ? routeFromUrl(doc.url) : doc.url;
|
|
1778
|
+
page.redirectsTo = landed === route ? null : landed;
|
|
1779
|
+
const pages = { [route]: page };
|
|
1780
|
+
const links = { [route]: extractLinks(doc.text) };
|
|
1781
|
+
const origin = site.origin ?? target.origin;
|
|
1782
|
+
await resolveBrokenLinks(pages, links, origin, async (candidate) => {
|
|
1783
|
+
try {
|
|
1784
|
+
return await fetchDoc(new URL(candidate, target.origin).href, options.timeout) === null;
|
|
1785
|
+
} catch {
|
|
1786
|
+
return false;
|
|
1787
|
+
}
|
|
1788
|
+
});
|
|
1789
|
+
if (options.checkExternal) {
|
|
1790
|
+
await resolveDeadExternal(pages, links, origin, Math.max(1, options.concurrency ?? 5), options.timeout);
|
|
1791
|
+
}
|
|
1792
|
+
return { schemaVersion: 1, createdAt: (/* @__PURE__ */ new Date()).toISOString(), site, pages };
|
|
1793
|
+
}
|
|
1765
1794
|
var MAX_NESTED_SITEMAPS = 50;
|
|
1766
1795
|
async function snapshotFromOrigin(origin, options = {}) {
|
|
1767
1796
|
const base = new URL(origin);
|
|
@@ -1893,6 +1922,7 @@ export {
|
|
|
1893
1922
|
snapshotFromDir,
|
|
1894
1923
|
snapshotFromGitRef,
|
|
1895
1924
|
snapshotFromOrigin,
|
|
1925
|
+
snapshotFromPage,
|
|
1896
1926
|
summarize,
|
|
1897
1927
|
withGuidance
|
|
1898
1928
|
};
|