@pylonsync/functions 0.4.7 → 0.4.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/ssr-runtime.d.ts +22 -0
- package/package.json +1 -1
- package/src/ssr-runtime.ts +27 -2
- package/src/ssr-utf8.test.ts +110 -0
package/dist/ssr-runtime.d.ts
CHANGED
|
@@ -339,6 +339,28 @@ export declare function isSafeRedirect(url: string, opts: {
|
|
|
339
339
|
* into a page's metadata. Explicit `metadata.icons.*` wins. */
|
|
340
340
|
export declare function applyAutoIcons(component: string, metadata: SsrMetadata | undefined): SsrMetadata | undefined;
|
|
341
341
|
export declare function applyAutoSocialImages(component: string, headers: Record<string, string> | undefined, metadata: SsrMetadata | undefined, requestUrl?: string): SsrMetadata | undefined;
|
|
342
|
+
/**
|
|
343
|
+
* Drain a `renderToReadableStream` reader, injecting `headBlob` immediately
|
|
344
|
+
* before the first `</head>` (or, if the document has none, the blob is
|
|
345
|
+
* never emitted — fragment renders have no head). `</head>` can straddle a
|
|
346
|
+
* chunk boundary, so a small carry buffer (len("</head>") − 1 bytes) is
|
|
347
|
+
* withheld at each chunk's tail until the next read confirms the match.
|
|
348
|
+
* Each emitted slice is handed to `sendChunk` as utf-8 text.
|
|
349
|
+
*
|
|
350
|
+
* Shared by the page render and the boundary render so head injection has
|
|
351
|
+
* exactly one implementation.
|
|
352
|
+
*
|
|
353
|
+
* Decoding uses ONE `TextDecoder` with `{stream: true}` for the whole
|
|
354
|
+
* reader, never a per-chunk decode. React splits the byte stream at
|
|
355
|
+
* arbitrary offsets, so a multi-byte character can land across two chunks;
|
|
356
|
+
* decoding each chunk independently turns the orphaned bytes into one
|
|
357
|
+
* U+FFFD apiece — `…` (e2 80 a6) arrives as three replacement characters.
|
|
358
|
+
* The failure is silent and position-dependent: markup added anywhere
|
|
359
|
+
* earlier shifts the boundary, so a page can render correctly for months
|
|
360
|
+
* and corrupt on an unrelated CSS change. A streaming decoder holds the
|
|
361
|
+
* partial sequence back until the bytes that complete it arrive.
|
|
362
|
+
*/
|
|
363
|
+
export declare function streamWithHeadInjection(reader: ReadableStreamDefaultReader<Uint8Array>, headBlob: string, sendChunk: (text: string) => void): Promise<void>;
|
|
342
364
|
/**
|
|
343
365
|
* Dev-only tail chunk: the `__PYLON_DEV__` info blob (cache verdict, render
|
|
344
366
|
* mode/timing, route) + the HUD bootstrap. Embedded after the page tail so the
|
package/package.json
CHANGED
package/src/ssr-runtime.ts
CHANGED
|
@@ -1310,8 +1310,18 @@ export function applyAutoSocialImages(
|
|
|
1310
1310
|
*
|
|
1311
1311
|
* Shared by the page render and the boundary render so head injection has
|
|
1312
1312
|
* exactly one implementation.
|
|
1313
|
+
*
|
|
1314
|
+
* Decoding uses ONE `TextDecoder` with `{stream: true}` for the whole
|
|
1315
|
+
* reader, never a per-chunk decode. React splits the byte stream at
|
|
1316
|
+
* arbitrary offsets, so a multi-byte character can land across two chunks;
|
|
1317
|
+
* decoding each chunk independently turns the orphaned bytes into one
|
|
1318
|
+
* U+FFFD apiece — `…` (e2 80 a6) arrives as three replacement characters.
|
|
1319
|
+
* The failure is silent and position-dependent: markup added anywhere
|
|
1320
|
+
* earlier shifts the boundary, so a page can render correctly for months
|
|
1321
|
+
* and corrupt on an unrelated CSS change. A streaming decoder holds the
|
|
1322
|
+
* partial sequence back until the bytes that complete it arrive.
|
|
1313
1323
|
*/
|
|
1314
|
-
async function streamWithHeadInjection(
|
|
1324
|
+
export async function streamWithHeadInjection(
|
|
1315
1325
|
reader: ReadableStreamDefaultReader<Uint8Array>,
|
|
1316
1326
|
headBlob: string,
|
|
1317
1327
|
sendChunk: (text: string) => void,
|
|
@@ -1319,11 +1329,15 @@ async function streamWithHeadInjection(
|
|
|
1319
1329
|
let headInjected = headBlob.length === 0;
|
|
1320
1330
|
let carry = "";
|
|
1321
1331
|
const HEAD_CLOSE = "</head>";
|
|
1332
|
+
const decoder = new TextDecoder("utf-8");
|
|
1322
1333
|
for (;;) {
|
|
1323
1334
|
const { value, done } = await reader.read();
|
|
1324
1335
|
if (done) break;
|
|
1325
1336
|
if (!value || value.byteLength === 0) continue;
|
|
1326
|
-
const text =
|
|
1337
|
+
const text = decoder.decode(value, { stream: true });
|
|
1338
|
+
// A chunk that ended mid-character decodes to "" — the bytes are held
|
|
1339
|
+
// in the decoder until the rest arrives. Nothing to emit yet.
|
|
1340
|
+
if (!text) continue;
|
|
1327
1341
|
if (!headInjected) {
|
|
1328
1342
|
const combined = carry + text;
|
|
1329
1343
|
const idx = combined.indexOf(HEAD_CLOSE);
|
|
@@ -1348,6 +1362,17 @@ async function streamWithHeadInjection(
|
|
|
1348
1362
|
sendChunk(text);
|
|
1349
1363
|
}
|
|
1350
1364
|
}
|
|
1365
|
+
// Flush the decoder. A stream that ends mid-character is genuinely
|
|
1366
|
+
// truncated input, and this is where it becomes a replacement character
|
|
1367
|
+
// rather than silently vanishing.
|
|
1368
|
+
const tail = decoder.decode();
|
|
1369
|
+
if (tail) {
|
|
1370
|
+
if (!headInjected) {
|
|
1371
|
+
carry += tail;
|
|
1372
|
+
} else {
|
|
1373
|
+
sendChunk(tail);
|
|
1374
|
+
}
|
|
1375
|
+
}
|
|
1351
1376
|
if (carry) sendChunk(carry);
|
|
1352
1377
|
}
|
|
1353
1378
|
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* SSR stream decoding — multi-byte UTF-8 across chunk boundaries.
|
|
3
|
+
*
|
|
4
|
+
* React's `renderToReadableStream` splits the byte stream at arbitrary
|
|
5
|
+
* offsets, so any non-ASCII character can land with its bytes divided
|
|
6
|
+
* between two chunks. Decoding each chunk independently turns the orphaned
|
|
7
|
+
* bytes into one U+FFFD apiece: `…` (e2 80 a6) arrives as three replacement
|
|
8
|
+
* characters.
|
|
9
|
+
*
|
|
10
|
+
* The failure is silent and position-dependent — markup added anywhere
|
|
11
|
+
* earlier in the document shifts the boundary — so a page renders correctly
|
|
12
|
+
* for months and corrupts on an unrelated CSS change. React only catches it
|
|
13
|
+
* when the string is hydrated and compared against the client render;
|
|
14
|
+
* server-only text corrupts with no warning at all.
|
|
15
|
+
*/
|
|
16
|
+
import { describe, expect, test } from "bun:test";
|
|
17
|
+
import { streamWithHeadInjection } from "./ssr-runtime";
|
|
18
|
+
|
|
19
|
+
/** A reader over pre-split byte chunks — the shape React hands us. */
|
|
20
|
+
function readerOf(chunks: Uint8Array[]): ReadableStreamDefaultReader<Uint8Array> {
|
|
21
|
+
return new ReadableStream<Uint8Array>({
|
|
22
|
+
start(controller) {
|
|
23
|
+
for (const c of chunks) controller.enqueue(c);
|
|
24
|
+
controller.close();
|
|
25
|
+
},
|
|
26
|
+
}).getReader();
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/** Split `text`'s utf-8 bytes at `at`, mid-character on purpose. */
|
|
30
|
+
function splitAt(text: string, at: number): Uint8Array[] {
|
|
31
|
+
const bytes = new TextEncoder().encode(text);
|
|
32
|
+
return [bytes.slice(0, at), bytes.slice(at)];
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
async function collect(chunks: Uint8Array[], headBlob = ""): Promise<string> {
|
|
36
|
+
const out: string[] = [];
|
|
37
|
+
await streamWithHeadInjection(readerOf(chunks), headBlob, (t) => out.push(t));
|
|
38
|
+
return out.join("");
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
describe("streamWithHeadInjection — UTF-8 across chunk boundaries", () => {
|
|
42
|
+
test("an ellipsis split across chunks survives", async () => {
|
|
43
|
+
const html = "<p>Search speakers by name…</p>";
|
|
44
|
+
const bytes = new TextEncoder().encode(html);
|
|
45
|
+
// "…" is e2 80 a6; cut between its first and second byte.
|
|
46
|
+
const cut = bytes.indexOf(0xe2) + 1;
|
|
47
|
+
expect(await collect(splitAt(html, cut))).toBe(html);
|
|
48
|
+
});
|
|
49
|
+
|
|
50
|
+
test("every split point of a 3-byte character round-trips", async () => {
|
|
51
|
+
const html = "<p>a…b</p>";
|
|
52
|
+
const bytes = new TextEncoder().encode(html);
|
|
53
|
+
for (let at = 1; at < bytes.length; at++) {
|
|
54
|
+
expect(await collect(splitAt(html, at))).toBe(html);
|
|
55
|
+
}
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
test("4-byte characters (emoji) survive every split", async () => {
|
|
59
|
+
const html = "<p>ship it 🚀 now</p>";
|
|
60
|
+
const bytes = new TextEncoder().encode(html);
|
|
61
|
+
for (let at = 1; at < bytes.length; at++) {
|
|
62
|
+
expect(await collect(splitAt(html, at))).toBe(html);
|
|
63
|
+
}
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
test("accented names and CJK survive every split", async () => {
|
|
67
|
+
// Ordinary content, not edge cases: a speaker called José, a Japanese
|
|
68
|
+
// session title, a middot in a byline.
|
|
69
|
+
const html = "<p>José · 日本語のセッション</p>";
|
|
70
|
+
const bytes = new TextEncoder().encode(html);
|
|
71
|
+
for (let at = 1; at < bytes.length; at++) {
|
|
72
|
+
expect(await collect(splitAt(html, at))).toBe(html);
|
|
73
|
+
}
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
test("a character split across THREE chunks survives", async () => {
|
|
77
|
+
// One byte per chunk — the decoder must hold state across two reads
|
|
78
|
+
// that each produce nothing.
|
|
79
|
+
const enc = new TextEncoder().encode("…");
|
|
80
|
+
const chunks = [enc.slice(0, 1), enc.slice(1, 2), enc.slice(2, 3)];
|
|
81
|
+
expect(await collect(chunks)).toBe("…");
|
|
82
|
+
});
|
|
83
|
+
|
|
84
|
+
test("head injection still lands, with a split character before it", async () => {
|
|
85
|
+
const html = "<html><head><title>é…</title></head><body>x</body></html>";
|
|
86
|
+
const bytes = new TextEncoder().encode(html);
|
|
87
|
+
const cut = bytes.indexOf(0xe2) + 1; // mid-ellipsis, before </head>
|
|
88
|
+
const out = await collect(splitAt(html, cut), "<meta name=x>");
|
|
89
|
+
expect(out).toBe(
|
|
90
|
+
"<html><head><title>é…</title><meta name=x></head><body>x</body></html>",
|
|
91
|
+
);
|
|
92
|
+
});
|
|
93
|
+
|
|
94
|
+
test("a stream truncated mid-character does not lose the tail silently", async () => {
|
|
95
|
+
// Genuinely broken input. One replacement character is the honest
|
|
96
|
+
// answer; dropping the bytes without a trace is not.
|
|
97
|
+
const enc = new TextEncoder().encode("ok…");
|
|
98
|
+
const out = await collect([enc.slice(0, enc.length - 1)]);
|
|
99
|
+
expect(out.startsWith("ok")).toBe(true);
|
|
100
|
+
expect(out).toContain("�");
|
|
101
|
+
});
|
|
102
|
+
|
|
103
|
+
test("pure ASCII is unaffected", async () => {
|
|
104
|
+
const html = "<p>plain ascii only</p>";
|
|
105
|
+
const bytes = new TextEncoder().encode(html);
|
|
106
|
+
for (let at = 1; at < bytes.length; at++) {
|
|
107
|
+
expect(await collect(splitAt(html, at))).toBe(html);
|
|
108
|
+
}
|
|
109
|
+
});
|
|
110
|
+
});
|