@johnhenry/packfile 0.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +486 -0
- package/browser.mjs +93 -0
- package/cache.mjs +61 -0
- package/compat.mjs +25 -0
- package/index.mjs +7 -0
- package/lib/blob-preview.mjs +447 -0
- package/lib/compression.browser.mjs +13 -0
- package/lib/compression.mjs +16 -0
- package/lib/create-router.mjs +79 -0
- package/lib/from-archive.mjs +23 -0
- package/lib/from-directory-lazy.mjs +61 -0
- package/lib/from-directory.mjs +79 -0
- package/lib/hash.mjs +14 -0
- package/lib/lazy-file-map.mjs +62 -0
- package/lib/mime.mjs +74 -0
- package/lib/response.mjs +41 -0
- package/lib/safe-symlink.mjs +15 -0
- package/lib/to-archive.mjs +27 -0
- package/lib/web-bundle.mjs +195 -0
- package/package.json +68 -0
- package/packfile.mjs +89 -0
- package/types.d.ts +84 -0
- package/types.ts +65 -0
package/index.mjs
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
export { fromDirectory } from "./lib/from-directory.mjs";
|
|
2
|
+
export { fromDirectoryLazy } from "./lib/from-directory-lazy.mjs";
|
|
3
|
+
export { fromArchive } from "./lib/from-archive.mjs";
|
|
4
|
+
export { toArchive } from "./lib/to-archive.mjs";
|
|
5
|
+
export { createRouter } from "./lib/create-router.mjs";
|
|
6
|
+
export { hashBuffer, hashStream } from "./lib/hash.mjs";
|
|
7
|
+
export { compileDirectory, decompileDirectory } from "./compat.mjs";
|
|
@@ -0,0 +1,447 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* blob-preview.mjs -- host a packfile `FilesMap` inside a browser tab/iframe
|
|
3
|
+
* with no server at all, by minting one `blob:` URL per file and rewriting
|
|
4
|
+
* HTML/CSS references so they resolve to the right blob URL instead of
|
|
5
|
+
* 404ing.
|
|
6
|
+
*
|
|
7
|
+
* This is "Approach B" from a prior design comparison (the heavier,
|
|
8
|
+
* real-isolation "Approach A" is a Service-Worker-based hosting mode,
|
|
9
|
+
* deferred and tracked as andbox#14 -- not implemented here). Approach B
|
|
10
|
+
* is explicitly the lighter-weight, "good enough for trusted/your-own
|
|
11
|
+
* content" path, not a general solution for arbitrary/untrusted content.
|
|
12
|
+
* See the README section this module is documented under for the full
|
|
13
|
+
* "what this does and does not solve" list.
|
|
14
|
+
*
|
|
15
|
+
* ---------------------------------------------------------------------
|
|
16
|
+
* Why blob: URLs can't just be handed to a real HTML parser and forgotten
|
|
17
|
+
* ---------------------------------------------------------------------
|
|
18
|
+
* `blob:` URLs are opaque, content-immutable handles minted by the
|
|
19
|
+
* platform at `URL.createObjectURL()` time -- their string value has no
|
|
20
|
+
* relationship to the path or content they represent, and once minted,
|
|
21
|
+
* a blob's bytes can never be changed (only revoked). That has one
|
|
22
|
+
* consequence this module has to design around directly: when two
|
|
23
|
+
* packaged files reference *each other* (the ordinary case of a
|
|
24
|
+
* multi-page site where every page links back to "home", or a page that
|
|
25
|
+
* links to itself), there is no way to mint blob A with a link to blob B
|
|
26
|
+
* baked into its bytes, and blob B with a link to blob A baked into
|
|
27
|
+
* *its* bytes, using only single-content, single-mint blobs -- one of
|
|
28
|
+
* the two references has to be decided first, and by the time the
|
|
29
|
+
* second one is minted, the first is already frozen.
|
|
30
|
+
*
|
|
31
|
+
* This module resolves that with a depth-first, mint-on-demand walk
|
|
32
|
+
* (`finalizeRewritable()` below): a file is rewritten and minted only
|
|
33
|
+
* once, only after every *acyclic* file it references has already been
|
|
34
|
+
* finalized, so genuine reference chains (A -> B -> C) resolve to real,
|
|
35
|
+
* live blob URLs end to end. The one edge that *closes* a cycle (A -> B
|
|
36
|
+
* where B (directly or transitively) already references back to A while
|
|
37
|
+
* A is still being processed, including the trivial case of a page
|
|
38
|
+
* linking to itself) is left as the original, unrewritten relative-path
|
|
39
|
+
* text for that one occurrence, and reported via `onUnresolvedReference`
|
|
40
|
+
* (default: `console.warn`) -- rewriting it to *some* blob URL would
|
|
41
|
+
* necessarily point at a stale, already-superseded version of the file,
|
|
42
|
+
* which is worse than an honestly-unrewritten link. A real, unrewritten
|
|
43
|
+
* relative link will not navigate correctly once loaded from a `blob:`
|
|
44
|
+
* URL either (confirmed directly: `new URL("./x", blobUrl)` throws
|
|
45
|
+
* `Invalid URL` -- blob URLs cannot serve as a relative-resolution base
|
|
46
|
+
* at all), so this is a real, disclosed limitation of blob-hosting a
|
|
47
|
+
* *reference cycle*, not a bug this module is hiding.
|
|
48
|
+
*
|
|
49
|
+
* ---------------------------------------------------------------------
|
|
50
|
+
* What this module does NOT rewrite
|
|
51
|
+
* ---------------------------------------------------------------------
|
|
52
|
+
* - Inline `<script>`/`<style>` block *contents* are left completely
|
|
53
|
+
* untouched (only the HTML attributes this module targets, and
|
|
54
|
+
* standalone .css files, are rewritten). An inline `<style>` block
|
|
55
|
+
* containing `url(...)` references to packaged files will not resolve.
|
|
56
|
+
* - JavaScript `import`/`import()` specifiers *inside* .js/.mjs/.cjs
|
|
57
|
+
* source text are never rewritten. Actual specifier resolution is
|
|
58
|
+
* delegated entirely to `@johnhenry/andbox`'s
|
|
59
|
+
* `createVirtualModuleRegistry()` (its `resolveSpecifier()` is exposed
|
|
60
|
+
* on the returned `registry` for advanced/manual use), per this
|
|
61
|
+
* module's brief not to reimplement that logic -- but note that even
|
|
62
|
+
* with `resolveSpecifier()` available, a plain
|
|
63
|
+
* `<script type="module" src="blob:...">` whose source contains a
|
|
64
|
+
* literal `import "./util.js"` will *not* resolve that import in a
|
|
65
|
+
* real browser: relative specifier resolution against a `blob:` base
|
|
66
|
+
* fails the same way plain HTML links would (see above), and nothing
|
|
67
|
+
* here rewrites JS source text to bake in literal blob URLs instead.
|
|
68
|
+
* Multi-file ESM graphs should be pre-bundled into a single file
|
|
69
|
+
* before packaging with packfile if they need to actually run.
|
|
70
|
+
*/
|
|
71
|
+
|
|
72
|
+
import { createVirtualModuleRegistry } from "@johnhenry/andbox";
|
|
73
|
+
import { getContentType } from "./mime.mjs";
|
|
74
|
+
|
|
75
|
+
const HTML_EXTENSIONS = new Set(["html", "htm"]);
|
|
76
|
+
const CSS_EXTENSIONS = new Set(["css"]);
|
|
77
|
+
const JS_EXTENSIONS = new Set(["js", "mjs", "cjs"]);
|
|
78
|
+
|
|
79
|
+
const TARGET_HTML_ATTRS = new Set(["href", "src", "srcset", "poster", "formaction"]);
|
|
80
|
+
|
|
81
|
+
const decoder = new TextDecoder("utf-8");
|
|
82
|
+
const encoder = new TextEncoder();
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Turn a packfile `FilesMap` into a set of `blob:` URLs suitable for hosting
|
|
86
|
+
* inside a browser tab/iframe with no server.
|
|
87
|
+
*
|
|
88
|
+
* @param {Map<string, {data: Uint8Array, size: number, hash: string}>} files
|
|
89
|
+
* Same shape `createRouter()` takes: an eager `Map`, or any object with
|
|
90
|
+
* `has()`/`get()`/`keys()` where `get()` may return a `Promise` (a
|
|
91
|
+
* `LazyFileMap`).
|
|
92
|
+
* @param {object} [options]
|
|
93
|
+
* @param {string} [options.rootPath="index.html"] - the entry-point path,
|
|
94
|
+
* must exist in `files`.
|
|
95
|
+
* @param {boolean} [options.strict=false] - throw instead of warning when
|
|
96
|
+
* a relative reference can't be resolved (missing file, or a reference
|
|
97
|
+
* cycle closing edge).
|
|
98
|
+
* @param {(info: {reason: "missing" | "cycle", targetPath: string, fromPath: string}) => void} [options.onUnresolvedReference]
|
|
99
|
+
* Called instead of the default `console.warn` for every reference this
|
|
100
|
+
* module leaves unrewritten because it couldn't (or, for cycles,
|
|
101
|
+
* shouldn't) be resolved. Ignored when `strict` is set (which throws
|
|
102
|
+
* instead).
|
|
103
|
+
* @returns {Promise<{entryUrl: string, resolve: (path: string) => string | null, dispose: () => void, registry: import("@johnhenry/andbox").VirtualModuleRegistry}>}
|
|
104
|
+
*/
|
|
105
|
+
export async function createBlobPreview(files, options = {}) {
|
|
106
|
+
const { rootPath = "index.html", strict = false, onUnresolvedReference } = options;
|
|
107
|
+
|
|
108
|
+
// Materialize every entry up front. `FilesMap.get()` may be synchronous
|
|
109
|
+
// (eager Map) or return a Promise (LazyFileMap) -- same duck-typed
|
|
110
|
+
// await lib/create-router.mjs already uses for the same reason: we need
|
|
111
|
+
// real bytes for every path before any rewriting can happen, since a
|
|
112
|
+
// relative reference from file A to file B may need B's content type
|
|
113
|
+
// (and, for HTML/CSS, B's own rewritten bytes) before A can be finalized.
|
|
114
|
+
const entries = new Map();
|
|
115
|
+
for (const path of files.keys()) {
|
|
116
|
+
let entry = files.get(path);
|
|
117
|
+
if (entry && typeof entry.then === "function") entry = await entry;
|
|
118
|
+
if (entry) entries.set(path, entry);
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
if (!entries.has(rootPath)) {
|
|
122
|
+
throw new Error(
|
|
123
|
+
`createBlobPreview: rootPath "${rootPath}" was not found in the given files map`
|
|
124
|
+
);
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
function report(reason, targetPath, fromPath) {
|
|
128
|
+
if (strict) {
|
|
129
|
+
throw new Error(
|
|
130
|
+
`createBlobPreview: cannot resolve reference from "${fromPath}" to "${targetPath}" (${reason})`
|
|
131
|
+
);
|
|
132
|
+
}
|
|
133
|
+
if (onUnresolvedReference) {
|
|
134
|
+
onUnresolvedReference({ reason, targetPath, fromPath });
|
|
135
|
+
return;
|
|
136
|
+
}
|
|
137
|
+
const explanation =
|
|
138
|
+
reason === "missing"
|
|
139
|
+
? "no such file in the packaged files (may be intentional, e.g. a path meant to load from a real network origin) -- left as-is"
|
|
140
|
+
: "resolving it would require a file that is still being finalized (a reference cycle, e.g. a page linking to itself or to a page that links back to it) -- left as the original, unrewritten path, since blob: URLs cannot express \"finalize B, using A, before A itself is finalized\" for a genuine cycle";
|
|
141
|
+
console.warn(`[packfile/blob-preview] "${fromPath}" -> "${targetPath}": ${explanation}`);
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
// --- 1. Classify every path -----------------------------------------
|
|
145
|
+
const jsSources = {};
|
|
146
|
+
const rewritablePaths = []; // html + css, processed via the DFS below
|
|
147
|
+
for (const [path, entry] of entries) {
|
|
148
|
+
const ext = extensionOf(path);
|
|
149
|
+
if (JS_EXTENSIONS.has(ext)) {
|
|
150
|
+
jsSources[path] = decoder.decode(entry.data);
|
|
151
|
+
} else if (HTML_EXTENSIONS.has(ext) || CSS_EXTENSIONS.has(ext)) {
|
|
152
|
+
rewritablePaths.push(path);
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
// --- 2. Mint JS blobs via andbox's registry (not reimplemented here) -
|
|
157
|
+
const registry = createVirtualModuleRegistry(jsSources);
|
|
158
|
+
|
|
159
|
+
// `finalized` is the single path -> blob URL table used everywhere
|
|
160
|
+
// (requirement: exactly one blob URL per path across the whole system).
|
|
161
|
+
const finalized = new Map();
|
|
162
|
+
for (const path of registry.paths()) finalized.set(path, registry.resolve(path));
|
|
163
|
+
|
|
164
|
+
// Blob URLs this module mints directly (everything except the JS ones,
|
|
165
|
+
// which are owned/revoked by `registry`), tracked so dispose() can
|
|
166
|
+
// revoke exactly what it minted.
|
|
167
|
+
const ownedBlobUrls = [];
|
|
168
|
+
|
|
169
|
+
function mintBlob(path, bytes, contentType) {
|
|
170
|
+
const blob = new Blob([bytes], { type: contentType });
|
|
171
|
+
const url = URL.createObjectURL(blob);
|
|
172
|
+
ownedBlobUrls.push(url);
|
|
173
|
+
finalized.set(path, url);
|
|
174
|
+
return url;
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
// --- 3. Mint everything that isn't HTML/CSS/JS as a direct passthrough
|
|
178
|
+
// blob (images, fonts, JSON, etc.) -- these never need rewriting, so
|
|
179
|
+
// they can be finalized immediately, before the HTML/CSS walk below
|
|
180
|
+
// needs to look any of them up.
|
|
181
|
+
for (const [path, entry] of entries) {
|
|
182
|
+
if (finalized.has(path)) continue; // already minted via the JS registry
|
|
183
|
+
const ext = extensionOf(path);
|
|
184
|
+
if (HTML_EXTENSIONS.has(ext) || CSS_EXTENSIONS.has(ext)) continue; // handled next
|
|
185
|
+
mintBlob(path, entry.data, getContentType(path));
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
// --- 4. HTML/CSS: depth-first, mint-on-demand walk --------------------
|
|
189
|
+
// A path in `inProgress` is an ancestor in the current DFS stack (i.e.
|
|
190
|
+
// "currently being rewritten, not yet minted"). See the module-level
|
|
191
|
+
// comment above for why a reference back into `inProgress` (or to the
|
|
192
|
+
// file currently being processed itself) has to degrade to
|
|
193
|
+
// "leave unrewritten" rather than pointing at a stale blob.
|
|
194
|
+
const inProgress = new Set();
|
|
195
|
+
|
|
196
|
+
function resolveReference(rawValue, fromPath) {
|
|
197
|
+
const parsed = parseReference(rawValue, fromPath);
|
|
198
|
+
if (parsed === null) return rawValue; // absolute/fragment/data/mailto/blob/etc: untouched
|
|
199
|
+
const { targetPath, suffix } = parsed;
|
|
200
|
+
|
|
201
|
+
if (!entries.has(targetPath)) {
|
|
202
|
+
report("missing", targetPath, fromPath);
|
|
203
|
+
return rawValue;
|
|
204
|
+
}
|
|
205
|
+
if (finalized.has(targetPath)) {
|
|
206
|
+
return finalized.get(targetPath) + suffix;
|
|
207
|
+
}
|
|
208
|
+
if (targetPath === fromPath || inProgress.has(targetPath)) {
|
|
209
|
+
report("cycle", targetPath, fromPath);
|
|
210
|
+
return rawValue;
|
|
211
|
+
}
|
|
212
|
+
// Not yet visited and not on the current path: finalize it now, then
|
|
213
|
+
// use its (now-real, now-permanent) blob URL.
|
|
214
|
+
finalizeRewritable(targetPath);
|
|
215
|
+
return finalized.get(targetPath) + suffix;
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
function finalizeRewritable(path) {
|
|
219
|
+
if (finalized.has(path)) return; // reached via another edge already
|
|
220
|
+
inProgress.add(path);
|
|
221
|
+
const entry = entries.get(path);
|
|
222
|
+
const text = decoder.decode(entry.data);
|
|
223
|
+
const ext = extensionOf(path);
|
|
224
|
+
const rewritten = HTML_EXTENSIONS.has(ext)
|
|
225
|
+
? rewriteHtml(text, path, resolveReference)
|
|
226
|
+
: rewriteCss(text, path, resolveReference);
|
|
227
|
+
inProgress.delete(path);
|
|
228
|
+
mintBlob(path, encoder.encode(rewritten), getContentType(path));
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
for (const path of rewritablePaths) finalizeRewritable(path);
|
|
232
|
+
|
|
233
|
+
// --- 5. Assemble the result -------------------------------------------
|
|
234
|
+
let disposed = false;
|
|
235
|
+
function dispose() {
|
|
236
|
+
if (disposed) return;
|
|
237
|
+
disposed = true;
|
|
238
|
+
registry.dispose();
|
|
239
|
+
for (const url of ownedBlobUrls) URL.revokeObjectURL(url);
|
|
240
|
+
finalized.clear();
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
return {
|
|
244
|
+
entryUrl: finalized.get(rootPath),
|
|
245
|
+
resolve: (path) => finalized.get(path) ?? null,
|
|
246
|
+
dispose,
|
|
247
|
+
registry,
|
|
248
|
+
};
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
// ---------------------------------------------------------------------------
|
|
252
|
+
// Path resolution
|
|
253
|
+
// ---------------------------------------------------------------------------
|
|
254
|
+
|
|
255
|
+
function extensionOf(path) {
|
|
256
|
+
const dot = path.lastIndexOf(".");
|
|
257
|
+
const slash = path.lastIndexOf("/");
|
|
258
|
+
if (dot === -1 || dot < slash) return "";
|
|
259
|
+
return path.slice(dot + 1).toLowerCase();
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
function directoryOf(path) {
|
|
263
|
+
const slash = path.lastIndexOf("/");
|
|
264
|
+
return slash === -1 ? "" : path.slice(0, slash + 1);
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
/**
|
|
268
|
+
* Resolve a root-relative path (no leading "/", already stripped by the
|
|
269
|
+
* caller) against the FilesMap's own root, the same way andbox's
|
|
270
|
+
* `virtual-module-registry.mjs` resolves specifiers against its own
|
|
271
|
+
* synthetic `vfs:///` base: percent-encode each segment (a path may
|
|
272
|
+
* legitimately contain literal "#"/"?" characters, which must not be
|
|
273
|
+
* parsed as a URL fragment/query delimiter and silently truncate
|
|
274
|
+
* everything after them), let the platform's own relative-URL algorithm
|
|
275
|
+
* do the "."/".." resolution, then decode back to a plain path.
|
|
276
|
+
*
|
|
277
|
+
* Excess ".." segments (more than there are directories to climb) clamp
|
|
278
|
+
* at the root rather than erroring, matching standard URL behavior --
|
|
279
|
+
* this is safe here (unlike packfile's on-disk path-safety checks elsewhere
|
|
280
|
+
* in this codebase) because nothing in this module ever touches the
|
|
281
|
+
* filesystem; the worst case is a clamped path that simply isn't in
|
|
282
|
+
* `files`, handled like any other missing reference.
|
|
283
|
+
*/
|
|
284
|
+
function resolveAgainstRoot(pathPart) {
|
|
285
|
+
const encoded = pathPart
|
|
286
|
+
.split("/")
|
|
287
|
+
.map(encodeURIComponent)
|
|
288
|
+
.join("/");
|
|
289
|
+
const resolved = new URL(`/${encoded}`, "vfs://blob-preview/");
|
|
290
|
+
return decodeURIComponent(resolved.pathname.slice(1));
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
/**
|
|
294
|
+
* Classify a raw HTML attribute / CSS url() value and, if it's a genuine
|
|
295
|
+
* local reference, resolve it to a `files` map key.
|
|
296
|
+
*
|
|
297
|
+
* Returns `null` for anything this module leaves untouched: absolute
|
|
298
|
+
* `http(s)://`/other-scheme URLs (`mailto:`, `data:`, `blob:`, `tel:`,
|
|
299
|
+
* `javascript:`, ...), protocol-relative `//host/...`, fragment-only
|
|
300
|
+
* `#...`, and query-only `?...` references.
|
|
301
|
+
*
|
|
302
|
+
* A leading-`/` path (root-relative) *is* resolved -- against the
|
|
303
|
+
* `files` map's own root, not left untouched. Unlike a real absolute
|
|
304
|
+
* path resolving against a live page origin (the kind of runtime
|
|
305
|
+
* resolution Approach A's Service-Worker mode exists for), a root-
|
|
306
|
+
* relative reference in *static* markup has a knowable target the
|
|
307
|
+
* moment this module runs (the FilesMap root), so rewriting it here is
|
|
308
|
+
* both possible and correct -- the "absolute paths need a real origin"
|
|
309
|
+
* limitation this module does NOT attempt to solve is specifically about
|
|
310
|
+
* paths a script *constructs or requests at runtime* (e.g. `fetch("/api")`
|
|
311
|
+
* inside JS, which this module never rewrites at all).
|
|
312
|
+
*/
|
|
313
|
+
function parseReference(rawValue, fromPath) {
|
|
314
|
+
if (!rawValue) return null;
|
|
315
|
+
if (/^[a-zA-Z][a-zA-Z0-9+.-]*:/.test(rawValue)) return null; // has a scheme
|
|
316
|
+
if (rawValue.startsWith("//")) return null; // protocol-relative
|
|
317
|
+
if (rawValue.startsWith("#")) return null; // fragment-only
|
|
318
|
+
|
|
319
|
+
const match = /^([^?#]*)([?#].*)?$/.exec(rawValue);
|
|
320
|
+
const pathPart = match[1];
|
|
321
|
+
const suffix = match[2] || "";
|
|
322
|
+
if (!pathPart) return null; // query-only, e.g. "?tab=2"
|
|
323
|
+
|
|
324
|
+
const targetPath = pathPart.startsWith("/")
|
|
325
|
+
? resolveAgainstRoot(pathPart.slice(1))
|
|
326
|
+
: resolveAgainstRoot(directoryOf(fromPath) + pathPart);
|
|
327
|
+
|
|
328
|
+
return { targetPath, suffix };
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
// ---------------------------------------------------------------------------
|
|
332
|
+
// HTML rewriting
|
|
333
|
+
//
|
|
334
|
+
// Deliberately not a full HTML parser (none is a dependency of this
|
|
335
|
+
// package, and the family convention here is to avoid adding one without
|
|
336
|
+
// strong justification -- see the module header). This is a two-level
|
|
337
|
+
// regex scan: first split the document into HTML comments (left
|
|
338
|
+
// untouched) and tags (attributes scanned), then scan each tag's text for
|
|
339
|
+
// `name=value` pairs and rewrite the handful of attributes that carry
|
|
340
|
+
// path references.
|
|
341
|
+
//
|
|
342
|
+
// Known, accepted gaps (not fixed by this module):
|
|
343
|
+
// - A literal ">" inside a quoted attribute value (e.g.
|
|
344
|
+
// `<a href="a>b">`) will end the tag match early, since tag boundaries
|
|
345
|
+
// are found with a plain `[^>]*` scan rather than a real tokenizer.
|
|
346
|
+
// - Comment detection is a non-greedy scan to the first "-->" -- correct
|
|
347
|
+
// for ordinary comments, not a guarantee against every pathological or
|
|
348
|
+
// malformed input.
|
|
349
|
+
// - Inline `<script>`/`<style>` block *contents* are never scanned (see
|
|
350
|
+
// the module header).
|
|
351
|
+
// ---------------------------------------------------------------------------
|
|
352
|
+
|
|
353
|
+
const TAG_OR_COMMENT_RE = /<!--[\s\S]*?-->|<[a-zA-Z!/][^>]*>/g;
|
|
354
|
+
const ATTR_RE = /([a-zA-Z_:][-a-zA-Z0-9_:.]*)(\s*=\s*)("([^"]*)"|'([^']*)'|([^\s"'=<>`]+))/g;
|
|
355
|
+
|
|
356
|
+
function rewriteHtml(text, fromPath, resolveReference) {
|
|
357
|
+
return text.replace(TAG_OR_COMMENT_RE, (tagOrComment) => {
|
|
358
|
+
if (tagOrComment.startsWith("<!--")) return tagOrComment; // comments untouched
|
|
359
|
+
return tagOrComment.replace(
|
|
360
|
+
ATTR_RE,
|
|
361
|
+
(full, name, equals, _quoted, doubleQuoted, singleQuoted, bare) => {
|
|
362
|
+
const lower = name.toLowerCase();
|
|
363
|
+
if (!TARGET_HTML_ATTRS.has(lower)) return full;
|
|
364
|
+
|
|
365
|
+
let value;
|
|
366
|
+
let quoteChar = null;
|
|
367
|
+
if (doubleQuoted !== undefined) {
|
|
368
|
+
value = doubleQuoted;
|
|
369
|
+
quoteChar = '"';
|
|
370
|
+
} else if (singleQuoted !== undefined) {
|
|
371
|
+
value = singleQuoted;
|
|
372
|
+
quoteChar = "'";
|
|
373
|
+
} else {
|
|
374
|
+
value = bare;
|
|
375
|
+
}
|
|
376
|
+
|
|
377
|
+
const newValue =
|
|
378
|
+
lower === "srcset"
|
|
379
|
+
? rewriteSrcset(value, fromPath, resolveReference)
|
|
380
|
+
: resolveReference(value, fromPath);
|
|
381
|
+
|
|
382
|
+
if (newValue === value) return full;
|
|
383
|
+
return quoteChar
|
|
384
|
+
? `${name}${equals}${quoteChar}${newValue}${quoteChar}`
|
|
385
|
+
: `${name}${equals}${newValue}`;
|
|
386
|
+
}
|
|
387
|
+
);
|
|
388
|
+
});
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
/**
|
|
392
|
+
* `srcset` is a comma-separated list of "url descriptor?" candidates
|
|
393
|
+
* (e.g. `"a.png 1x, b.png 2x"` or `"a.png 480w"`). Only the URL portion
|
|
394
|
+
* of each candidate is rewritten. This is a practical simplification, not
|
|
395
|
+
* the full srcset grammar (real URLs containing a literal comma would
|
|
396
|
+
* split incorrectly here) -- acceptable for the same "lightweight,
|
|
397
|
+
* honestly-documented" reason the rest of this module isn't a full
|
|
398
|
+
* parser.
|
|
399
|
+
*/
|
|
400
|
+
function rewriteSrcset(value, fromPath, resolveReference) {
|
|
401
|
+
return value
|
|
402
|
+
.split(",")
|
|
403
|
+
.map((candidate) => {
|
|
404
|
+
const trimmed = candidate.trim();
|
|
405
|
+
if (!trimmed) return trimmed;
|
|
406
|
+
const spaceIndex = trimmed.search(/\s/);
|
|
407
|
+
if (spaceIndex === -1) return resolveReference(trimmed, fromPath);
|
|
408
|
+
const url = trimmed.slice(0, spaceIndex);
|
|
409
|
+
const descriptor = trimmed.slice(spaceIndex).trim();
|
|
410
|
+
return `${resolveReference(url, fromPath)} ${descriptor}`;
|
|
411
|
+
})
|
|
412
|
+
.join(", ");
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
// ---------------------------------------------------------------------------
|
|
416
|
+
// CSS rewriting
|
|
417
|
+
//
|
|
418
|
+
// Two independent regex passes over the raw text: `url(...)` (covers both
|
|
419
|
+
// plain `url()` references and `@import url(...)`), then bare-string
|
|
420
|
+
// `@import "..."`/`@import '...'` (the form with no `url()` wrapper).
|
|
421
|
+
// Same honest caveat as the HTML side: not comment-aware (a url()/@import
|
|
422
|
+
// sitting inside a `/* ... */` comment will still be rewritten, harmlessly
|
|
423
|
+
// since it was inert either way, but not specially detected).
|
|
424
|
+
// ---------------------------------------------------------------------------
|
|
425
|
+
|
|
426
|
+
const CSS_URL_RE = /url\(\s*(?:"([^"]*)"|'([^']*)'|([^'")\s]*))\s*\)/g;
|
|
427
|
+
const CSS_IMPORT_STRING_RE = /(@import\s+)("([^"]*)"|'([^']*)')/g;
|
|
428
|
+
|
|
429
|
+
function rewriteCss(text, fromPath, resolveReference) {
|
|
430
|
+
let rewritten = text.replace(CSS_URL_RE, (full, dq, sq, bare) => {
|
|
431
|
+
const value = dq !== undefined ? dq : sq !== undefined ? sq : bare;
|
|
432
|
+
const newValue = resolveReference(value, fromPath);
|
|
433
|
+
if (newValue === value) return full;
|
|
434
|
+
const quoteChar = dq !== undefined ? '"' : sq !== undefined ? "'" : "";
|
|
435
|
+
return `url(${quoteChar}${newValue}${quoteChar})`;
|
|
436
|
+
});
|
|
437
|
+
|
|
438
|
+
rewritten = rewritten.replace(CSS_IMPORT_STRING_RE, (full, prefix, _quoted, dq, sq) => {
|
|
439
|
+
const value = dq !== undefined ? dq : sq;
|
|
440
|
+
const newValue = resolveReference(value, fromPath);
|
|
441
|
+
if (newValue === value) return full;
|
|
442
|
+
const quoteChar = dq !== undefined ? '"' : "'";
|
|
443
|
+
return `${prefix}${quoteChar}${newValue}${quoteChar}`;
|
|
444
|
+
});
|
|
445
|
+
|
|
446
|
+
return rewritten;
|
|
447
|
+
}
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
export const deCompressObject = (compressedData, format = "gzip") => {
|
|
2
|
+
const stream = new Blob([compressedData]).stream();
|
|
3
|
+
const decompressedStream = stream.pipeThrough(
|
|
4
|
+
new DecompressionStream(format)
|
|
5
|
+
);
|
|
6
|
+
return new Response(decompressedStream).arrayBuffer();
|
|
7
|
+
};
|
|
8
|
+
|
|
9
|
+
export const compressObject = (data, format = "gzip") => {
|
|
10
|
+
const stream = new Blob([data]).stream();
|
|
11
|
+
const compressedStream = stream.pipeThrough(new CompressionStream(format));
|
|
12
|
+
return new Response(compressedStream).arrayBuffer();
|
|
13
|
+
};
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { gzip, gunzip, constants } from "zlib";
|
|
2
|
+
import { promisify } from "util";
|
|
3
|
+
|
|
4
|
+
export const compressObject = (buffer, level = constants.Z_DEFAULT_COMPRESSION) => {
|
|
5
|
+
return new Promise((resolve, reject) => {
|
|
6
|
+
gzip(buffer, { level }, (error, result) => {
|
|
7
|
+
if (error) {
|
|
8
|
+
reject(error);
|
|
9
|
+
} else {
|
|
10
|
+
resolve(result);
|
|
11
|
+
}
|
|
12
|
+
});
|
|
13
|
+
});
|
|
14
|
+
};
|
|
15
|
+
|
|
16
|
+
export const deCompressObject = promisify(gunzip);
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
import { buildFileResponse } from "./response.mjs";
|
|
2
|
+
|
|
3
|
+
export const createRouter = (files, options = {}) => {
|
|
4
|
+
const {
|
|
5
|
+
alias = {},
|
|
6
|
+
cacheControl,
|
|
7
|
+
mimeTypes,
|
|
8
|
+
tryExtensions = [],
|
|
9
|
+
fallback,
|
|
10
|
+
} = options;
|
|
11
|
+
|
|
12
|
+
const responseOpts = { cacheControl, mimeTypes };
|
|
13
|
+
|
|
14
|
+
const resolve = (path) => {
|
|
15
|
+
if (files.has(path)) return path;
|
|
16
|
+
for (const ext of tryExtensions) {
|
|
17
|
+
const candidate = path + ext;
|
|
18
|
+
if (files.has(candidate)) return candidate;
|
|
19
|
+
}
|
|
20
|
+
return null;
|
|
21
|
+
};
|
|
22
|
+
|
|
23
|
+
const notFound = (input, ctx) => {
|
|
24
|
+
if (fallback) return fallback(input, ctx);
|
|
25
|
+
return new Response("Not Found", { status: 404 });
|
|
26
|
+
};
|
|
27
|
+
|
|
28
|
+
const handler = async (input, ctx) => {
|
|
29
|
+
let method, filePath, request;
|
|
30
|
+
|
|
31
|
+
if (typeof input === "string") {
|
|
32
|
+
method = "GET";
|
|
33
|
+
filePath = input;
|
|
34
|
+
request = null;
|
|
35
|
+
} else {
|
|
36
|
+
request = input;
|
|
37
|
+
method = request.method;
|
|
38
|
+
filePath = new URL(request.url).pathname;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
// Only GET and HEAD allowed
|
|
42
|
+
if (method !== "GET" && method !== "HEAD") {
|
|
43
|
+
if (fallback) return fallback(input, ctx);
|
|
44
|
+
return new Response("Method Not Allowed", {
|
|
45
|
+
status: 405,
|
|
46
|
+
headers: { Allow: "GET, HEAD" },
|
|
47
|
+
});
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
// Apply aliases
|
|
51
|
+
for (const [aliasPath, targetPath] of Object.entries(alias)) {
|
|
52
|
+
if (filePath === aliasPath) {
|
|
53
|
+
filePath = targetPath;
|
|
54
|
+
break;
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
// Remove leading slash
|
|
59
|
+
filePath = filePath.replace(/^\//, "");
|
|
60
|
+
|
|
61
|
+
// Look up file, trying extensions if needed
|
|
62
|
+
const resolved = resolve(filePath);
|
|
63
|
+
if (!resolved) return notFound(input, ctx);
|
|
64
|
+
filePath = resolved;
|
|
65
|
+
|
|
66
|
+
// Duck-type: if get() returns a Promise (LazyFileMap), await it
|
|
67
|
+
let entry = files.get(filePath);
|
|
68
|
+
if (entry && typeof entry.then === "function") {
|
|
69
|
+
entry = await entry;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
if (!entry) return notFound(input, ctx);
|
|
73
|
+
|
|
74
|
+
return buildFileResponse(request, filePath, entry, responseOpts);
|
|
75
|
+
};
|
|
76
|
+
|
|
77
|
+
handler.fetch = handler;
|
|
78
|
+
return handler;
|
|
79
|
+
};
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Deserializes this package's archive format (gzip(Web Bundle)) back into a
|
|
3
|
+
* `FilesMap`, via `fromWebBundle()` (`lib/web-bundle.mjs`). See
|
|
4
|
+
* `to-archive.mjs` for `ARCHIVE_BASE_URL` -- the same internal-only base URL
|
|
5
|
+
* used there is used here to strip each exchange's absolute URL back down
|
|
6
|
+
* to a relative path. Path-safety validation (rejecting an entry that
|
|
7
|
+
* normalizes outside the archive root) lives in `fromWebBundle()` itself
|
|
8
|
+
* now, applied whenever a `baseURL` is given -- not duplicated here.
|
|
9
|
+
*/
|
|
10
|
+
import { deCompressObject } from "./compression.mjs";
|
|
11
|
+
import { fromWebBundle } from "./web-bundle.mjs";
|
|
12
|
+
|
|
13
|
+
const ARCHIVE_BASE_URL = "https://packfile.invalid/";
|
|
14
|
+
|
|
15
|
+
export const fromArchive = async (buffer, opts = {}) => {
|
|
16
|
+
const { compressed = true } = opts;
|
|
17
|
+
|
|
18
|
+
if (compressed) {
|
|
19
|
+
buffer = await deCompressObject(buffer);
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
return fromWebBundle(buffer, { baseURL: ARCHIVE_BASE_URL });
|
|
23
|
+
};
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
import { readdir, stat, realpath } from "node:fs/promises";
|
|
2
|
+
import { join, relative, sep } from "node:path";
|
|
3
|
+
import { LazyFileMap } from "./lazy-file-map.mjs";
|
|
4
|
+
import { isWithinRoot } from "./safe-symlink.mjs";
|
|
5
|
+
|
|
6
|
+
// See the identical comment in from-directory.mjs: path.relative() returns
|
|
7
|
+
// backslash-separated paths on Windows, but every consumer (LazyFileMap's
|
|
8
|
+
// own get()/has(), createRouter(), ...) assumes POSIX-style "/" keys.
|
|
9
|
+
// path.join() on Windows accepts "/" transparently, so normalizing the key
|
|
10
|
+
// here doesn't break LazyFileMap#get()'s own join(basePath, key) call.
|
|
11
|
+
const toPosixPath = (path) => (sep === "/" ? path : path.split(sep).join("/"));
|
|
12
|
+
|
|
13
|
+
export const fromDirectoryLazy = async (directoryPath, options = {}) => {
|
|
14
|
+
const { ignorePatterns = [] } = options;
|
|
15
|
+
|
|
16
|
+
const paths = new Set();
|
|
17
|
+
const rootRealPath = await realpath(directoryPath);
|
|
18
|
+
|
|
19
|
+
// See fromDirectory() for why both an escape check and a cycle check are
|
|
20
|
+
// needed: a symlink inside directoryPath can point outside the tree
|
|
21
|
+
// (exposing arbitrary files via paths/get()), or back at an ancestor
|
|
22
|
+
// directory (e.g. `ln -s . loop`), which would otherwise recurse until
|
|
23
|
+
// the OS's ELOOP limit crashes the whole fromDirectoryLazy() call.
|
|
24
|
+
const walk = async (currentPath, currentRealPath, ancestors) => {
|
|
25
|
+
const entries = await readdir(currentPath, { withFileTypes: true });
|
|
26
|
+
|
|
27
|
+
for (const entry of entries) {
|
|
28
|
+
const fullPath = join(currentPath, entry.name);
|
|
29
|
+
const relativePath = toPosixPath(relative(directoryPath, fullPath));
|
|
30
|
+
|
|
31
|
+
if (ignorePatterns.some((pattern) => new RegExp(pattern).test(relativePath))) {
|
|
32
|
+
continue;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
let childRealPath = join(currentRealPath, entry.name);
|
|
36
|
+
|
|
37
|
+
if (entry.isSymbolicLink()) {
|
|
38
|
+
let realTarget;
|
|
39
|
+
try {
|
|
40
|
+
realTarget = await realpath(fullPath);
|
|
41
|
+
} catch {
|
|
42
|
+
continue; // broken symlink or too many levels of symlinks (ELOOP)
|
|
43
|
+
}
|
|
44
|
+
if (!isWithinRoot(rootRealPath, realTarget)) continue;
|
|
45
|
+
if (ancestors.has(realTarget)) continue; // symlink cycle
|
|
46
|
+
childRealPath = realTarget;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
const stats = await stat(fullPath);
|
|
50
|
+
|
|
51
|
+
if (stats.isDirectory()) {
|
|
52
|
+
await walk(fullPath, childRealPath, new Set(ancestors).add(childRealPath));
|
|
53
|
+
} else if (stats.isFile()) {
|
|
54
|
+
paths.add(relativePath);
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
};
|
|
58
|
+
|
|
59
|
+
await walk(directoryPath, rootRealPath, new Set([rootRealPath]));
|
|
60
|
+
return new LazyFileMap(directoryPath, paths);
|
|
61
|
+
};
|