@pitlane/crawler 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md ADDED
@@ -0,0 +1,25 @@
1
+ # @pitlane/crawler
2
+
3
+ ## 0.1.0
4
+
5
+ Initial release.
6
+
7
+ - `crawl(router, options)` — walks an app by dispatching requests into its
8
+ router's `fetch`, yielding `{ pathname, filepath, response }` per path.
9
+ Follows `<a href>` and `<link rel="alternate">`, queues the assets a page
10
+ references, honours `rel="nofollow"` and `<meta name="robots">`, skips
11
+ cross-origin and non-navigable hrefs, and visits each path once. A redirect
12
+ yields nothing and reports through `onRedirect`, since there is no document
13
+ to write and the app still answers the path at runtime; under `spider` the
14
+ same-origin target is queued instead. Any other non-2xx response aborts the
15
+ crawl. `paths`, `spider`, `assets`, `concurrency`, `ignorePageNofollow`, and
16
+ `onRedirect` configure it.
17
+ - `staticPaths(routes)` — the paths a Remix 3 route map can serve with no
18
+ params, deduplicated and sorted. `GET` and method-agnostic routes whose
19
+ patterns declare no variables or wildcards.
20
+ - The API comes from [remix-run/remix#11150](https://github.com/remix-run/remix/pull/11150),
21
+ which was closed with the implementation kept beside the Remix docs site.
22
+ Two deliberate differences: `assets` is a new option, because a bundler has
23
+ usually emitted those files already, and the first error wins over the last
24
+ when several paths fail under concurrency.
25
+ - Tested against `remix@3.0.0-beta.10`.
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Mark Malstrom
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,101 @@
1
+ # @pitlane/crawler
2
+
3
+ Spider a [Remix 3](https://remix.run) fetch router in memory.
4
+
5
+ `crawl(router)` dispatches requests straight into `router.fetch` and yields every response it gets back, following the links each page contains. No socket, no server, no browser, no HTTP: the router is the whole transport, so an app can be walked wherever the app itself runs.
6
+
7
+ That makes prerendering a `for await` loop.
8
+
9
+ ```ts
10
+ import { crawl } from "@pitlane/crawler";
11
+ import * as fs from "node:fs/promises";
12
+ import * as path from "node:path";
13
+
14
+ import router from "./app/entry.server.ts";
15
+
16
+ for await (let { pathname, filepath, response } of crawl(router)) {
17
+ let outputPath = path.join("dist", filepath);
18
+ await fs.mkdir(path.dirname(outputPath), { recursive: true });
19
+ await fs.writeFile(outputPath, new Uint8Array(await response.arrayBuffer()));
20
+ console.log(`${pathname} -> ${outputPath}`);
21
+ }
22
+ ```
23
+
24
+ ## Install
25
+
26
+ ```sh
27
+ npm install @pitlane/crawler
28
+ # or
29
+ vp add @pitlane/crawler
30
+ ```
31
+
32
+ Requires `remix@^3.0.0-beta.10` as a peer.
33
+
34
+ Using the [`remix()` Vite plugin](https://pitlane.tools/package/dev/)? You do not need this package directly — `remix({ prerender })` runs it for you. See the [prerendering guide](https://pitlane.tools/guides/prerendering).
35
+
36
+ For everything else a walk is good for — static exports, sitemaps, link checks, render smoke tests — see the [crawling guide](https://pitlane.tools/guides/crawler).
37
+
38
+ ## `crawl(router, options?)`
39
+
40
+ Returns an async iterator of `{ pathname, filepath, response }`, one per fetched path.
41
+
42
+ - `pathname` — the path that was requested.
43
+ - `filepath` — where the response belongs on disk. HTML gets `<pathname>/index.html` so a static host serves it back for the original path; everything else keeps its own path.
44
+ - `response` — the router's response, body unread.
45
+
46
+ Results arrive in completion order, and every path is fetched at most once.
47
+
48
+ A redirect yields nothing: there is no document to write, and the app still answers the path at runtime. It reports through `onRedirect` instead, and when `spider` is on the same-origin target is queued, so a crawl seeded at a `/` that points elsewhere still finds the site. Any other non-2xx response aborts the crawl with `Crawl failed: <status> <statusText> (<pathname>)`.
49
+
50
+ | Option | Type | Default | Purpose |
51
+ | -------------------- | ------------------------------- | ------- | -------------------------------------------------------------------------------------------------------------------------------------- |
52
+ | `paths` | `string[]` | `["/"]` | Where to start. |
53
+ | `spider` | `boolean` | `true` | Follow `<a href>` and `<link rel="alternate">` to find more paths. |
54
+ | `assets` | `boolean` | `true` | Queue the `<link href>`, `<script src>`, and `<img src>` each page references. Turn it off when a bundler already emitted those files. |
55
+ | `concurrency` | `number` | `1` | How many paths to fetch at once. |
56
+ | `ignorePageNofollow` | `(pathname: string) => boolean` | — | Crawl a page's links even though the page asked robots not to follow them. |
57
+ | `onRedirect` | `(pathname, location) => void` | — | Called for a path that redirected instead of returning a document. `location` is `null` when the redirect named none. |
58
+
59
+ The first argument is anything with a `fetch(request: Request)` method: a `createRouter()` router, a built server bundle's default export, a worker-style `{ fetch }` object.
60
+
61
+ ### What spidering respects
62
+
63
+ Crawling stops where a crawler should stop, so a run over a real site does not wander:
64
+
65
+ - `rel="nofollow"` on a link, and `<meta name="robots" content="nofollow">` (or `googlebot`) on a page.
66
+ - Absolute and protocol-relative URLs, which belong to another origin.
67
+ - `#fragment`, `mailto:`, `tel:`, `javascript:`, and `data:` hrefs.
68
+
69
+ `ignorePageNofollow` is the escape hatch for the case where a page's `nofollow` is aimed at search engines rather than at you — a versioned docs tree that should not be indexed but does need to be built.
70
+
71
+ ## `staticPaths(routes)`
72
+
73
+ The question that comes before a crawl: which paths can this app serve with no params?
74
+
75
+ ```ts
76
+ import { staticPaths } from "@pitlane/crawler";
77
+ import { get, route } from "remix/routes";
78
+
79
+ let routes = route({
80
+ home: "/",
81
+ blog: get("/blog"),
82
+ post: get("/blog/:slug"),
83
+ });
84
+
85
+ staticPaths(routes); // ["/", "/blog"]
86
+ ```
87
+
88
+ A route qualifies when it answers `GET` (or any method) and its pattern declares no variables or wildcards. `/blog/:slug` is left out, because its values live outside the route map. Results are deduplicated and sorted, so a build that renders them lists its output the same way every time.
89
+
90
+ ## Provenance
91
+
92
+ The `crawl` API comes from [remix-run/remix#11150](https://github.com/remix-run/remix/pull/11150), which proposed it for `fetch-router` and was closed in favour of keeping the implementation next to the Remix docs site. This package brings it back out as something an application can install, with two changes:
93
+
94
+ - `assets` is new. Upstream always queues a page's assets, which is right for a site with no bundler and wrong for one where Vite already emitted them.
95
+ - The first error wins when several paths fail at once, rather than the last.
96
+
97
+ `staticPaths` has no upstream counterpart. It is the Remix 3 answer to React Router's `getStaticPaths`: a Remix router exposes no route table, but the route map an app builds it from is an ordinary object, and that is the thing worth reading.
98
+
99
+ ## License
100
+
101
+ MIT
@@ -0,0 +1,101 @@
1
+ import { RouteMap } from "remix/routes";
2
+ //#region src/crawl.d.ts
3
+ /** One fetched path and the response the router produced for it. */
4
+ interface CrawlResult {
5
+ /** The path that was requested, exactly as it was queued. */
6
+ pathname: string;
7
+ /**
8
+ * Where the response belongs on disk. HTML lands at `<pathname>/index.html`
9
+ * so a static host serves it for the original path; everything else keeps
10
+ * its own path.
11
+ */
12
+ filepath: string;
13
+ /** The router's response. Unconsumed: the body is still readable. */
14
+ response: Response;
15
+ }
16
+ interface CrawlOptions {
17
+ /**
18
+ * Paths to start from.
19
+ *
20
+ * @default ["/"]
21
+ */
22
+ paths?: string[];
23
+ /**
24
+ * Follow `<a href>` and `<link rel="alternate">` to discover more paths.
25
+ * Turn it off to fetch exactly the paths given.
26
+ *
27
+ * @default true
28
+ */
29
+ spider?: boolean;
30
+ /**
31
+ * Queue the `<link href>`, `<script src>`, and `<img src>` a page
32
+ * references. Turn it off when something else already emitted those files,
33
+ * as a bundler does.
34
+ *
35
+ * @default true
36
+ */
37
+ assets?: boolean;
38
+ /**
39
+ * How many paths to fetch at once.
40
+ *
41
+ * @default 1
42
+ */
43
+ concurrency?: number;
44
+ /**
45
+ * Crawl a page's links even though the page asked robots not to follow
46
+ * them. Receives the page's path.
47
+ */
48
+ ignorePageNofollow?: (pathname: string) => boolean;
49
+ /**
50
+ * Called for a path that answered with a redirect rather than a document.
51
+ * Nothing is yielded for it: there is no page to write, and the app still
52
+ * answers the path at runtime.
53
+ *
54
+ * Receives the requested path and the `Location` it pointed at, which is
55
+ * `null` for a redirect that named none.
56
+ */
57
+ onRedirect?: (pathname: string, location: string | null) => void;
58
+ }
59
+ /**
60
+ * What {@link crawl} needs from a router. A `createRouter()` router satisfies
61
+ * it, and so does anything else that answers a `Request` — a built server
62
+ * bundle's default export, a worker-style `{ fetch }` object, a test double.
63
+ */
64
+ interface CrawlTarget {
65
+ fetch(request: Request): Response | Promise<Response>;
66
+ }
67
+ /**
68
+ * Walks an app by dispatching requests straight into its router, yielding each
69
+ * response as it arrives. No socket, no server, no browser: the router's
70
+ * `fetch` is the whole transport, so this runs anywhere the app itself runs.
71
+ *
72
+ * Yields in completion order. Every path is fetched at most once. A redirect
73
+ * yields nothing and reports through {@link CrawlOptions.onRedirect}; any
74
+ * other non-2xx response aborts the crawl.
75
+ *
76
+ * @param router The router to crawl.
77
+ * @param options Crawl options.
78
+ * @returns An async iterator of results, one per fetched path.
79
+ */
80
+ declare function crawl(router: CrawlTarget, options?: CrawlOptions): AsyncIterableIterator<CrawlResult>;
81
+ //#endregion
82
+ //#region src/static-paths.d.ts
83
+ /**
84
+ * Collects every path in a route map that can be requested without params:
85
+ * the Remix 3 answer to "what pages does this app have", and the input a
86
+ * prerender pass needs before it knows any dynamic values.
87
+ *
88
+ * A route qualifies when it answers `GET` (or any method) and its pattern
89
+ * declares no variables or wildcards. `/blog` is a static path; `/blog/:slug`
90
+ * is not, because its values live outside the route map. Routes constrained to
91
+ * a protocol or hostname are skipped too, since their href is not a path.
92
+ *
93
+ * Results are deduplicated and sorted, so a build that prerenders them lists
94
+ * its output the same way every time.
95
+ *
96
+ * @param routes The route map, usually the one the app's router is built from.
97
+ * @returns The static paths, sorted.
98
+ */
99
+ declare function staticPaths(routes: RouteMap): string[];
100
+ //#endregion
101
+ export { type CrawlOptions, type CrawlResult, type CrawlTarget, crawl, staticPaths };
package/dist/index.mjs ADDED
@@ -0,0 +1,402 @@
1
+ import { getRoutePatternCaptures } from "remix/route-pattern";
2
+ //#region src/html-parser.ts
3
+ const VOID_ELEMENTS = /* @__PURE__ */ new Set([
4
+ "area",
5
+ "base",
6
+ "br",
7
+ "col",
8
+ "embed",
9
+ "hr",
10
+ "img",
11
+ "input",
12
+ "link",
13
+ "meta",
14
+ "param",
15
+ "source",
16
+ "track",
17
+ "wbr"
18
+ ]);
19
+ const RAW_TEXT_ELEMENTS = /* @__PURE__ */ new Set(["script", "style"]);
20
+ var Element = class {
21
+ name;
22
+ #attributes = /* @__PURE__ */ new Map();
23
+ #children = [];
24
+ #selfClosing;
25
+ constructor(name, selfClosing = false) {
26
+ this.name = name.toLowerCase();
27
+ this.#selfClosing = selfClosing;
28
+ }
29
+ setAttribute(name, value) {
30
+ this.#attributes.set(name.toLowerCase(), value);
31
+ }
32
+ getAttribute(name) {
33
+ return this.#attributes.get(name.toLowerCase()) ?? null;
34
+ }
35
+ get innerHTML() {
36
+ return serialize(this.#children);
37
+ }
38
+ set innerHTML(value) {
39
+ this.#children = [value];
40
+ }
41
+ appendChild(child) {
42
+ this.#children.push(child);
43
+ }
44
+ isSelfClosing() {
45
+ return this.#selfClosing;
46
+ }
47
+ toString() {
48
+ if (this.name === "#comment") return this.#attributes.get("text") ?? "";
49
+ let attrs = Array.from(this.#attributes.entries()).map(([key, value]) => value === "" ? key : `${key}="${value.replace(/"/g, "&quot;")}"`).join(" ");
50
+ let attrString = attrs ? ` ${attrs}` : "";
51
+ if (this.name.startsWith("!")) return `<${this.name}${attrString}>`;
52
+ if (VOID_ELEMENTS.has(this.name) || this.#selfClosing) return `<${this.name}${attrString} />`;
53
+ return `<${this.name}${attrString}>${serialize(this.#children)}</${this.name}>`;
54
+ }
55
+ };
56
+ /** A parsed document: the flat element list, plus the tree for serialization. */
57
+ var Document = class {
58
+ #elements;
59
+ #children;
60
+ constructor(elements, children) {
61
+ this.#elements = elements;
62
+ this.#children = children;
63
+ }
64
+ /** Every element in the document, in source order. */
65
+ get elements() {
66
+ return this.#elements;
67
+ }
68
+ toString() {
69
+ return serialize(this.#children);
70
+ }
71
+ };
72
+ function serialize(children) {
73
+ return children.map((child) => typeof child === "string" ? child : child.toString()).join("");
74
+ }
75
+ /**
76
+ * Index of the `>` that closes the tag opening at `start`, skipping any `>`
77
+ * inside a quoted attribute value. Returns -1 when the tag never closes.
78
+ */
79
+ function findTagEnd(html, start) {
80
+ let quote = null;
81
+ for (let i = start; i < html.length; i++) {
82
+ let char = html[i];
83
+ if (char === void 0) break;
84
+ if (quote) {
85
+ if (char === quote) quote = null;
86
+ continue;
87
+ }
88
+ if (char === "\"" || char === "'") {
89
+ quote = char;
90
+ continue;
91
+ }
92
+ if (char === ">") return i;
93
+ }
94
+ return -1;
95
+ }
96
+ function parseTag(content) {
97
+ let source = content.trim();
98
+ let selfClosing = false;
99
+ if (source.endsWith("/")) {
100
+ selfClosing = true;
101
+ source = source.slice(0, -1).trimEnd();
102
+ }
103
+ let nameEnd = 0;
104
+ while (nameEnd < source.length && !/\s/.test(source[nameEnd] ?? "")) nameEnd++;
105
+ let name = source.slice(0, nameEnd);
106
+ if (!name) return null;
107
+ let element = new Element(name, selfClosing);
108
+ let i = nameEnd;
109
+ while (i < source.length) {
110
+ while (i < source.length && /\s/.test(source[i] ?? "")) i++;
111
+ if (i >= source.length) break;
112
+ let attrNameStart = i;
113
+ while (i < source.length && !/[\s=]/.test(source[i] ?? "")) i++;
114
+ let attrName = source.slice(attrNameStart, i);
115
+ if (!attrName) break;
116
+ while (i < source.length && /\s/.test(source[i] ?? "")) i++;
117
+ if (source[i] !== "=") {
118
+ element.setAttribute(attrName, "");
119
+ continue;
120
+ }
121
+ i++;
122
+ while (i < source.length && /\s/.test(source[i] ?? "")) i++;
123
+ let quote = source[i];
124
+ if (quote === "\"" || quote === "'") {
125
+ i++;
126
+ let valueStart = i;
127
+ while (i < source.length && source[i] !== quote) i++;
128
+ element.setAttribute(attrName, source.slice(valueStart, i));
129
+ if (source[i] === quote) i++;
130
+ continue;
131
+ }
132
+ let valueStart = i;
133
+ while (i < source.length && !/\s/.test(source[i] ?? "")) i++;
134
+ element.setAttribute(attrName, source.slice(valueStart, i));
135
+ }
136
+ return element;
137
+ }
138
+ /**
139
+ * Parses an HTML document into a flat element list and a serializable tree.
140
+ *
141
+ * @param html The document source.
142
+ * @param options Parse options.
143
+ * @returns The parsed document.
144
+ */
145
+ function parse(html, options) {
146
+ let elements = [];
147
+ let children = [];
148
+ let stack = [];
149
+ let i = 0;
150
+ let appendChild = (child) => {
151
+ let parent = stack.at(-1);
152
+ if (parent) parent.appendChild(child);
153
+ else children.push(child);
154
+ };
155
+ let appendElement = (element) => {
156
+ elements.push(element);
157
+ appendChild(element);
158
+ };
159
+ let closeElement = (name) => {
160
+ let normalized = name.toLowerCase();
161
+ for (let index = stack.length - 1; index >= 0; index--) if (stack[index]?.name === normalized) {
162
+ stack.length = index;
163
+ return;
164
+ }
165
+ };
166
+ while (i < html.length) {
167
+ let lt = html.indexOf("<", i);
168
+ if (lt === -1) {
169
+ appendChild(html.slice(i));
170
+ break;
171
+ }
172
+ if (lt > i) appendChild(html.slice(i, lt));
173
+ if (html.startsWith("<!--", lt)) {
174
+ let end = html.indexOf("-->", lt + 4);
175
+ if (end === -1) {
176
+ appendChild(html.slice(lt));
177
+ break;
178
+ }
179
+ if (options?.comment) {
180
+ let comment = new Element("#comment");
181
+ comment.setAttribute("text", html.slice(lt, end + 3));
182
+ appendElement(comment);
183
+ }
184
+ i = end + 3;
185
+ continue;
186
+ }
187
+ if (html[lt + 1] === "/") {
188
+ let end = findTagEnd(html, lt + 2);
189
+ if (end === -1) {
190
+ appendChild(html.slice(lt));
191
+ break;
192
+ }
193
+ let tagName = html.slice(lt + 2, end).trim().split(/\s+/, 1)[0];
194
+ if (tagName) closeElement(tagName);
195
+ i = end + 1;
196
+ continue;
197
+ }
198
+ let end = findTagEnd(html, lt + 1);
199
+ if (end === -1) {
200
+ appendChild(html.slice(lt));
201
+ break;
202
+ }
203
+ let element = parseTag(html.slice(lt + 1, end));
204
+ if (!element) {
205
+ appendChild(html.slice(lt, end + 1));
206
+ i = end + 1;
207
+ continue;
208
+ }
209
+ appendElement(element);
210
+ i = end + 1;
211
+ if (element.name.startsWith("!") || VOID_ELEMENTS.has(element.name) || element.isSelfClosing()) continue;
212
+ if (RAW_TEXT_ELEMENTS.has(element.name)) {
213
+ let closeStart = html.toLowerCase().indexOf(`</${element.name}`, i);
214
+ if (closeStart === -1) {
215
+ element.appendChild(html.slice(i));
216
+ break;
217
+ }
218
+ element.appendChild(html.slice(i, closeStart));
219
+ let closeEnd = findTagEnd(html, closeStart + 2);
220
+ if (closeEnd === -1) break;
221
+ i = closeEnd + 1;
222
+ continue;
223
+ }
224
+ stack.push(element);
225
+ }
226
+ return new Document(elements, children);
227
+ }
228
+ //#endregion
229
+ //#region src/crawl.ts
230
+ /**
231
+ * Requests are dispatched straight into the router, so the origin is never
232
+ * used for anything but satisfying the `Request` constructor.
233
+ */
234
+ const BASE_URL = "http://localhost";
235
+ /**
236
+ * Walks an app by dispatching requests straight into its router, yielding each
237
+ * response as it arrives. No socket, no server, no browser: the router's
238
+ * `fetch` is the whole transport, so this runs anywhere the app itself runs.
239
+ *
240
+ * Yields in completion order. Every path is fetched at most once. A redirect
241
+ * yields nothing and reports through {@link CrawlOptions.onRedirect}; any
242
+ * other non-2xx response aborts the crawl.
243
+ *
244
+ * @param router The router to crawl.
245
+ * @param options Crawl options.
246
+ * @returns An async iterator of results, one per fetched path.
247
+ */
248
+ async function* crawl(router, options = {}) {
249
+ let { paths = ["/"], spider = true, assets = true, concurrency = 1, ignorePageNofollow, onRedirect } = options;
250
+ let queue = [];
251
+ let visited = /* @__PURE__ */ new Set();
252
+ let results = [];
253
+ let active = 0;
254
+ let error;
255
+ let notify = () => {};
256
+ let gate = new Promise((resolve) => notify = resolve);
257
+ function bump() {
258
+ let previous = notify;
259
+ gate = new Promise((resolve) => notify = resolve);
260
+ previous();
261
+ }
262
+ enqueue(paths);
263
+ while (true) {
264
+ while (active < concurrency && queue.length > 0) fetchOne(queue.shift());
265
+ if (error) throw error;
266
+ if (results.length > 0) {
267
+ yield results.shift();
268
+ continue;
269
+ }
270
+ if (active === 0 && queue.length === 0) break;
271
+ await gate;
272
+ }
273
+ function enqueue(pathnames) {
274
+ for (let pathname of pathnames) {
275
+ if (visited.has(pathname)) continue;
276
+ visited.add(pathname);
277
+ queue.push(pathname);
278
+ }
279
+ }
280
+ async function fetchOne(pathname) {
281
+ active++;
282
+ try {
283
+ let response = await router.fetch(new Request(`${BASE_URL}${pathname}`));
284
+ if (response.status >= 300 && response.status < 400) {
285
+ let location = response.headers.get("Location");
286
+ onRedirect?.(pathname, location);
287
+ await response.body?.cancel();
288
+ if (spider && location != null) enqueue(resolveAll([location], pathname));
289
+ return;
290
+ }
291
+ if (!response.ok) {
292
+ let status = [response.status, response.statusText].filter(Boolean).join(" ");
293
+ throw new Error(`Crawl failed: ${status} (${pathname})`);
294
+ }
295
+ if (!response.headers.get("Content-Type")?.includes("text/html")) {
296
+ results.push({
297
+ pathname,
298
+ filepath: pathname,
299
+ response
300
+ });
301
+ return;
302
+ }
303
+ let cloned = response.clone();
304
+ results.push({
305
+ pathname,
306
+ filepath: pathname.replace(/\/?$/, "/index.html"),
307
+ response
308
+ });
309
+ let document = parse(await cloned.text());
310
+ if (assets) enqueue(extractAssetPaths(document.elements, pathname));
311
+ if (spider && (ignorePageNofollow?.(pathname) || shouldCrawlLinks(document.elements))) enqueue(extractLinkPaths(document.elements, pathname));
312
+ } catch (thrown) {
313
+ error ??= thrown;
314
+ } finally {
315
+ active--;
316
+ bump();
317
+ }
318
+ }
319
+ }
320
+ function extractAssetPaths(elements, baseUrl) {
321
+ let hrefs = elements.filter((element) => element.name === "link" && !rel(element).includes("nofollow")).map((element) => element.getAttribute("href"));
322
+ let sources = elements.filter((element) => element.name === "script" || element.name === "img").map((element) => element.getAttribute("src"));
323
+ return resolveAll([...hrefs, ...sources], baseUrl);
324
+ }
325
+ function extractLinkPaths(elements, baseUrl) {
326
+ return resolveAll(elements.filter((element) => !rel(element).includes("nofollow") && (element.name === "a" || element.name === "link" && rel(element).includes("alternate"))).map((element) => element.getAttribute("href")), baseUrl);
327
+ }
328
+ function resolveAll(hrefs, baseUrl) {
329
+ let paths = [];
330
+ for (let href of hrefs) {
331
+ if (href == null || isNonNavigable(href) || !isRelativeUrl(href)) continue;
332
+ let resolved = resolveHref(href, baseUrl);
333
+ if (resolved != null) paths.push(resolved);
334
+ }
335
+ return paths;
336
+ }
337
+ /** Whether a page opted out of having its links followed. */
338
+ function shouldCrawlLinks(elements) {
339
+ return !elements.some((element) => {
340
+ if (element.name !== "meta") return false;
341
+ let name = element.getAttribute("name")?.toLowerCase();
342
+ if (name !== "robots" && name !== "googlebot") return false;
343
+ return (element.getAttribute("content")?.toLowerCase() ?? "").split(/[\s,]+/).includes("nofollow");
344
+ });
345
+ }
346
+ function rel(element) {
347
+ return element.getAttribute("rel")?.split(/\s+/) ?? [];
348
+ }
349
+ function isNonNavigable(href) {
350
+ return href.startsWith("#") || href.startsWith("mailto:") || href.startsWith("tel:") || href.startsWith("javascript:") || href.startsWith("data:");
351
+ }
352
+ function isRelativeUrl(href) {
353
+ return !href.startsWith("http://") && !href.startsWith("https://") && !href.startsWith("//");
354
+ }
355
+ function resolveHref(href, baseUrl) {
356
+ if (href.startsWith("/")) return href;
357
+ try {
358
+ return new URL(href, `${BASE_URL}${baseUrl}`).pathname;
359
+ } catch {
360
+ return null;
361
+ }
362
+ }
363
+ //#endregion
364
+ //#region src/static-paths.ts
365
+ function isRoute(value) {
366
+ return typeof value === "object" && value !== null && "method" in value && "pattern" in value && typeof value.href === "function";
367
+ }
368
+ /**
369
+ * Collects every path in a route map that can be requested without params:
370
+ * the Remix 3 answer to "what pages does this app have", and the input a
371
+ * prerender pass needs before it knows any dynamic values.
372
+ *
373
+ * A route qualifies when it answers `GET` (or any method) and its pattern
374
+ * declares no variables or wildcards. `/blog` is a static path; `/blog/:slug`
375
+ * is not, because its values live outside the route map. Routes constrained to
376
+ * a protocol or hostname are skipped too, since their href is not a path.
377
+ *
378
+ * Results are deduplicated and sorted, so a build that prerenders them lists
379
+ * its output the same way every time.
380
+ *
381
+ * @param routes The route map, usually the one the app's router is built from.
382
+ * @returns The static paths, sorted.
383
+ */
384
+ function staticPaths(routes) {
385
+ let paths = /* @__PURE__ */ new Set();
386
+ collect(routes, paths);
387
+ return [...paths].sort();
388
+ }
389
+ function collect(node, paths) {
390
+ for (let value of Object.values(node)) {
391
+ if (isRoute(value)) {
392
+ if (value.method !== "GET" && value.method !== "ANY") continue;
393
+ if (getRoutePatternCaptures(value.pattern).length > 0) continue;
394
+ let href = value.href();
395
+ if (href.startsWith("/")) paths.add(href);
396
+ continue;
397
+ }
398
+ if (typeof value === "object" && value !== null) collect(value, paths);
399
+ }
400
+ }
401
+ //#endregion
402
+ export { crawl, staticPaths };
package/package.json ADDED
@@ -0,0 +1,51 @@
1
+ {
2
+ "name": "@pitlane/crawler",
3
+ "version": "0.1.0",
4
+ "description": "crawl() — spider a Remix 3 fetch router in memory to prerender it, plus static-path discovery from a route map.",
5
+ "keywords": [
6
+ "crawl",
7
+ "fetch-router",
8
+ "pitlane",
9
+ "prerender",
10
+ "remix",
11
+ "ssg"
12
+ ],
13
+ "homepage": "https://pitlane.tools/package/crawler/",
14
+ "bugs": {
15
+ "url": "https://github.com/pitlane-tools/pitlane/issues"
16
+ },
17
+ "license": "MIT",
18
+ "author": "Mark Malstrom <mark@malstrom.me>",
19
+ "repository": {
20
+ "type": "git",
21
+ "url": "git+https://github.com/pitlane-tools/pitlane.git",
22
+ "directory": "packages/crawler"
23
+ },
24
+ "files": [
25
+ "dist",
26
+ "CHANGELOG.md"
27
+ ],
28
+ "type": "module",
29
+ "types": "./dist/index.d.mts",
30
+ "exports": {
31
+ ".": {
32
+ "types": "./dist/index.d.mts",
33
+ "import": "./dist/index.mjs"
34
+ }
35
+ },
36
+ "scripts": {
37
+ "prepublishOnly": "vp run build"
38
+ },
39
+ "devDependencies": {
40
+ "@types/node": "^25.5.0",
41
+ "remix": "3.0.0-beta.10",
42
+ "typescript": "^7.0.2",
43
+ "vite-plus": "^0.2.6"
44
+ },
45
+ "peerDependencies": {
46
+ "remix": "^3.0.0-beta.10"
47
+ },
48
+ "engines": {
49
+ "node": "^20.19.0 || >=22.12.0"
50
+ }
51
+ }