@statewalker/webrun-http-streams 0.1.1 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/README.md +288 -11
  2. package/dist/bytes.d.ts +43 -0
  3. package/dist/bytes.d.ts.map +1 -0
  4. package/dist/codec-default.d.ts +7 -0
  5. package/dist/codec-default.d.ts.map +1 -0
  6. package/dist/duplex-site-builder.d.ts +3 -0
  7. package/dist/duplex-site-builder.d.ts.map +1 -1
  8. package/dist/envelope.d.ts +10 -10
  9. package/dist/envelope.d.ts.map +1 -1
  10. package/dist/fetch.d.ts +3 -2
  11. package/dist/fetch.d.ts.map +1 -1
  12. package/dist/http-data.d.ts +11 -7
  13. package/dist/http-data.d.ts.map +1 -1
  14. package/dist/http-error.d.ts.map +1 -1
  15. package/dist/http-stubs.d.ts +12 -0
  16. package/dist/http-stubs.d.ts.map +1 -1
  17. package/dist/http1/chunked.d.ts +18 -0
  18. package/dist/http1/chunked.d.ts.map +1 -0
  19. package/dist/http1/decode.d.ts +5 -0
  20. package/dist/http1/decode.d.ts.map +1 -0
  21. package/dist/http1/encode.d.ts +20 -0
  22. package/dist/http1/encode.d.ts.map +1 -0
  23. package/dist/http1/errors.d.ts +9 -0
  24. package/dist/http1/errors.d.ts.map +1 -0
  25. package/dist/http1/headers.d.ts +78 -0
  26. package/dist/http1/headers.d.ts.map +1 -0
  27. package/dist/http1/index.d.ts +18 -0
  28. package/dist/http1/index.d.ts.map +1 -0
  29. package/dist/index.d.ts +6 -2
  30. package/dist/index.d.ts.map +1 -1
  31. package/dist/index.js +1039 -137
  32. package/dist/message.d.ts +50 -0
  33. package/dist/message.d.ts.map +1 -0
  34. package/dist/request-streams.d.ts +52 -0
  35. package/dist/request-streams.d.ts.map +1 -0
  36. package/dist/sniff.d.ts +18 -0
  37. package/dist/sniff.d.ts.map +1 -0
  38. package/package.json +8 -6
  39. package/src/bytes.ts +157 -0
  40. package/src/codec-default.ts +13 -0
  41. package/src/duplex-site-builder.ts +9 -1
  42. package/src/envelope.ts +43 -34
  43. package/src/fetch.ts +130 -12
  44. package/src/http-data.ts +160 -17
  45. package/src/http-stubs.ts +72 -18
  46. package/src/http1/chunked.ts +89 -0
  47. package/src/http1/decode.ts +208 -0
  48. package/src/http1/encode.ts +181 -0
  49. package/src/http1/errors.ts +8 -0
  50. package/src/http1/headers.ts +263 -0
  51. package/src/http1/index.ts +40 -0
  52. package/src/index.ts +18 -0
  53. package/src/message.ts +62 -0
  54. package/src/request-streams.ts +75 -0
  55. package/src/sniff.ts +68 -0
package/src/http-data.ts CHANGED
@@ -1,10 +1,23 @@
1
1
  import type { Duplex } from "@statewalker/webrun-streams";
2
- import {
3
- decodeMessage,
4
- encodeMessage,
5
- type RequestEnvelope,
6
- type ResponseEnvelope,
7
- } from "./envelope.js";
2
+ import { deserializeError, serializeError } from "@statewalker/webrun-streams";
3
+ import { discard } from "./bytes.js";
4
+ import { defaultCodec } from "./codec-default.js";
5
+ import { HttpParseError } from "./http1/errors.js";
6
+ import type {
7
+ ByteSource,
8
+ DecodedRequest,
9
+ MessageCodec,
10
+ RequestEnvelope,
11
+ ResponseEnvelope,
12
+ } from "./message.js";
13
+
14
+ /** Carries the serialized JS error between two webrun peers. */
15
+ export const PEER_ERROR_HEADER = "x-webrun-error";
16
+
17
+ export type HttpDataOptions = {
18
+ /** Wire format. Defaults to `defaultCodec`: writes HTTP/1.1, accepts either. */
19
+ codec?: MessageCodec;
20
+ };
8
21
 
9
22
  export interface HttpFetchResult {
10
23
  envelope: ResponseEnvelope;
@@ -13,7 +26,7 @@ export interface HttpFetchResult {
13
26
 
14
27
  export interface HttpDataHandlerResult {
15
28
  envelope: ResponseEnvelope;
16
- body?: AsyncIterable<Uint8Array> | Iterable<Uint8Array>;
29
+ body?: ByteSource;
17
30
  }
18
31
 
19
32
  /**
@@ -29,6 +42,100 @@ export type HttpDataHandler = (
29
42
  body: AsyncIterable<Uint8Array>,
30
43
  ) => Promise<HttpDataHandlerResult>;
31
44
 
45
+ /** Hard cap on the peer-error header (M3): see `encodeErrorResponse`. */
46
+ const MAX_DETAIL_CHARS = 4096;
47
+
48
+ /**
49
+ * JSON with every non-printable-ASCII character escaped, so the result is a
50
+ * legal latin-1 header value whatever the error message contained.
51
+ */
52
+ function asciiJson(value: unknown): string {
53
+ return JSON.stringify(value).replace(
54
+ /[^\x20-\x7E]/g,
55
+ (ch) => `\\u${ch.charCodeAt(0).toString(16).padStart(4, "0")}`,
56
+ );
57
+ }
58
+
59
+ /**
60
+ * Decision 13: a real HTTP peer cannot receive a JavaScript exception, only a
61
+ * response. So an uncaught handler error becomes a conforming 500 whose body
62
+ * carries the message, with the serialized error in a namespaced header that
63
+ * another webrun peer re-throws from.
64
+ */
65
+ function encodeErrorResponse(
66
+ codec: MessageCodec,
67
+ error: unknown,
68
+ method: string,
69
+ status = 500,
70
+ statusText = "Internal Server Error",
71
+ ): AsyncGenerator<Uint8Array> {
72
+ const serialized = serializeError(error);
73
+ let detail = asciiJson(serialized);
74
+ if (detail.length > MAX_DETAIL_CHARS) detail = asciiJson({ message: serialized.message });
75
+ // The fallback above is itself unbounded — a >64 KiB handler message would
76
+ // still blow the head past maxHeaderBytes, and the client would then
77
+ // report "head section exceeds 65536 bytes" instead of the peer's actual
78
+ // error (M3). Truncate unconditionally as the last resort.
79
+ if (detail.length > MAX_DETAIL_CHARS) detail = detail.slice(0, MAX_DETAIL_CHARS);
80
+ const message = serialized.message ?? statusText;
81
+ return codec.encodeResponse(
82
+ {
83
+ status,
84
+ statusText,
85
+ headers: [
86
+ ["Content-Type", "text/plain; charset=utf-8"],
87
+ [PEER_ERROR_HEADER, detail],
88
+ ],
89
+ },
90
+ [new TextEncoder().encode(message)],
91
+ { method },
92
+ );
93
+ }
94
+
95
+ /**
96
+ * `output` is the `Duplex` call's own generator — one logical call, per the
97
+ * `Duplex` contract in `@statewalker/webrun-streams`: "Consumer `.return()`
98
+ * on the output → producer's `finally` runs." On a mux transport
99
+ * (`emulateMux`), that `finally` is what frees the stream-table slot; skip it
100
+ * and every peer-error response leaks one slot, unboundedly, until the mux
101
+ * itself is exhausted (`maxStreams` reached, every further call rejected).
102
+ *
103
+ * Cancelling `output` here — not inside the codec's own decode logic — is
104
+ * deliberate. `codec.decodeResponse` already pulled from `output` to read the
105
+ * head before this function runs, so unlike the body case above there's no
106
+ * suspended-start no-op to worry about. And it's safe specifically *because*
107
+ * this is the client's own response-only generator: the request was already
108
+ * fully sent before we got here, so there is nothing left to write on it.
109
+ * The equivalent is NOT safe inside `src/http1/decode.ts`'s `ByteReader` —
110
+ * that code also runs when a caller wires `codec.decodeRequest`/
111
+ * `decodeResponse` directly onto a raw bidirectional socket (see
112
+ * `tests/http1-node-interop.test.ts`), where the *same* object is read from
113
+ * and then written back to (a server reads the request, then replies on the
114
+ * same socket); cancelling the read side there tears down the whole
115
+ * connection out from under the pending write. That was tried and reverted —
116
+ * see the Task 11 report for the "socket hang up" failure it caused.
117
+ */
118
+ async function throwIfPeerError(
119
+ result: HttpFetchResult,
120
+ output: AsyncGenerator<Uint8Array>,
121
+ ): Promise<void> {
122
+ const found = result.envelope.headers.find(([k]) => k.toLowerCase() === PEER_ERROR_HEADER);
123
+ if (!found) return;
124
+ await discard(result.body);
125
+ try {
126
+ await output.return?.(undefined);
127
+ } catch {
128
+ /* the call is being abandoned; a failing cancel must not mask the peer's error */
129
+ }
130
+ let payload: { message: string };
131
+ try {
132
+ payload = JSON.parse(found[1]) as { message: string };
133
+ } catch {
134
+ payload = { message: found[1] };
135
+ }
136
+ throw deserializeError(payload);
137
+ }
138
+
32
139
  /**
33
140
  * Initiate an HTTP call over a `Duplex`. The caller's `call: Duplex` is
34
141
  * obtained from any `webrun-streams-*` adapter's `connect`. Returns the
@@ -40,22 +147,58 @@ export type HttpDataHandler = (
40
147
  export async function httpFetch(
41
148
  call: Duplex,
42
149
  env: RequestEnvelope,
43
- body?: AsyncIterable<Uint8Array> | Iterable<Uint8Array>,
150
+ body?: ByteSource,
151
+ options: HttpDataOptions = {},
44
152
  ): Promise<HttpFetchResult> {
45
- const output = call(encodeMessage(env, body));
46
- return decodeMessage<ResponseEnvelope>(output);
153
+ const codec = options.codec ?? defaultCodec;
154
+ const output = call(codec.encodeRequest(env, body));
155
+ const result = await codec.decodeResponse(output, { method: env.method });
156
+ await throwIfPeerError(result, output);
157
+ return result;
47
158
  }
48
159
 
49
160
  /**
50
161
  * Wrap an HTTP handler as a `Duplex` so it can be registered with any
51
- * `webrun-streams-*` adapter's `serve`. The duplex `split`s the input to
52
- * recover envelope + body, dispatches to the handler, and emits the response
53
- * via `encodeMessage`.
162
+ * `webrun-streams-*` adapter's `serve`.
54
163
  */
55
- export function httpServe(handler: HttpDataHandler): Duplex {
164
+ export function httpServe(handler: HttpDataHandler, options: HttpDataOptions = {}): Duplex {
165
+ const codec = options.codec ?? defaultCodec;
56
166
  return async function* httpHandlerDuplex(input) {
57
- const { envelope: reqEnv, body: reqBody } = await decodeMessage<RequestEnvelope>(input);
58
- const result = await handler(reqEnv, reqBody);
59
- yield* encodeMessage(result.envelope, result.body);
167
+ // A refusal to parse the request must still produce a response. Left
168
+ // uncaught, the exception propagates out of this generator and NOTHING is
169
+ // written — a real peer waits, then sees the connection end with no status
170
+ // at all, which is the one place the "a real peer cannot receive a
171
+ // JavaScript exception" rule was not applied.
172
+ //
173
+ // Only HttpParseError is converted: that is the codec's refusal type, so
174
+ // it means the bytes were malformed. A transport failure is not our 400 to
175
+ // send and must keep propagating.
176
+ let decoded: DecodedRequest;
177
+ try {
178
+ decoded = await codec.decodeRequest(input);
179
+ } catch (error) {
180
+ if (!(error instanceof HttpParseError)) throw error;
181
+ // Answer in the format the peer was speaking, when sniffing got far
182
+ // enough to know it; otherwise fall back to our write codec.
183
+ const refusalCodec = (error as { codec?: MessageCodec }).codec ?? codec;
184
+ // The method is unknowable — the request line may be what failed — so
185
+ // encode as if for GET, which permits a body and so lets the peer read
186
+ // why it was refused.
187
+ yield* encodeErrorResponse(refusalCodec, error, "GET", 400, "Bad Request");
188
+ return;
189
+ }
190
+ // Reply in kind: answer with the codec that actually read the request, so a
191
+ // peer pinned to one format is never answered in the other.
192
+ const replyCodec = decoded.codec ?? codec;
193
+ let result: HttpDataHandlerResult;
194
+ try {
195
+ result = await handler(decoded.envelope, decoded.body);
196
+ } catch (error) {
197
+ yield* encodeErrorResponse(replyCodec, error, decoded.envelope.method);
198
+ return;
199
+ }
200
+ yield* replyCodec.encodeResponse(result.envelope, result.body, {
201
+ method: decoded.envelope.method,
202
+ });
60
203
  };
61
204
  }
package/src/http-stubs.ts CHANGED
@@ -1,4 +1,6 @@
1
1
  import { fromReadableStream, toReadableStream } from "@statewalker/webrun-streams";
2
+ import { discard } from "./bytes.js";
3
+ import { collectBytes, supportsRequestStreams } from "./request-streams.js";
2
4
 
3
5
  export interface SerializedHttpRequest {
4
6
  url: string;
@@ -23,14 +25,26 @@ export interface SerializedHttpResponse {
23
25
 
24
26
  export interface SerializedHttpEnvelope<Options> {
25
27
  options: Options;
28
+ /**
29
+ * The body bytes. Must be *productive*: it has to yield or finish on its own,
30
+ * because a body neither stub is allowed to read (a GET/HEAD/OPTIONS request,
31
+ * a null-body-status response) is released with `discard`, and `discard` must
32
+ * await one `.next()` before `.return()` — a bare `.return()` is a no-op on a
33
+ * generator in suspended start and would leak the producer. So a `content`
34
+ * that blocks forever without yielding makes the stub block with it, where an
35
+ * earlier version returned immediately and leaked instead. Every transport in
36
+ * this repo satisfies this; a `content` that waits on a peer that may never
37
+ * send should carry its own timeout.
38
+ */
26
39
  content: AsyncIterable<Uint8Array>;
27
40
  }
28
41
 
29
42
  export type HttpHandler = (request: Request) => Response | Promise<Response>;
30
43
 
31
44
  // Statuses that MUST NOT carry a body — `new Response(body, { status })` throws
32
- // for these (per the Fetch spec's "null body status" set).
33
- const NULL_BODY_STATUSES = new Set([101, 103, 204, 205, 304]);
45
+ // for these (per the Fetch spec's "null body status" set). Exported so
46
+ // `fetch.ts` mirrors this exact set rather than redefining it.
47
+ export const NULL_BODY_STATUSES = new Set([101, 103, 204, 205, 304]);
34
48
 
35
49
  const REQUEST_FIELDS = [
36
50
  "url",
@@ -45,6 +59,13 @@ const REQUEST_FIELDS = [
45
59
  "keepalive",
46
60
  ] as const;
47
61
 
62
+ /** `SerializedHttpEnvelope.content` is never absent, only empty. */
63
+ async function* noBytes(): AsyncGenerator<Uint8Array> {}
64
+
65
+ async function* oneChunk(chunk: Uint8Array): AsyncGenerator<Uint8Array> {
66
+ yield chunk;
67
+ }
68
+
48
69
  /**
49
70
  * Returns an HTTP handler that serializes a Request, hands the envelope to
50
71
  * `send` for transport, and deserializes the reply into a Response. Used on
@@ -63,9 +84,25 @@ export function newHttpClientStub(
63
84
  const val = (request as unknown as Record<string, unknown>)[field];
64
85
  if (val !== undefined && field !== "url") (options as Record<string, unknown>)[field] = val;
65
86
  }
66
- const content = request.body
67
- ? fromReadableStream(request.body as ReadableStream<Uint8Array>)
68
- : (async function* () {})();
87
+ let content: AsyncIterable<Uint8Array>;
88
+ if (request.body != null) {
89
+ // The streaming path stays first and unconditional: a runtime that has
90
+ // request streams never reaches the fallback, and never buffers an upload.
91
+ content = fromReadableStream(request.body as ReadableStream<Uint8Array>);
92
+ } else if (!supportsRequestStreams()) {
93
+ // Firefox. Without `request.body` there is nothing to stream from, so the
94
+ // only way to see the payload at all is to buffer the whole of it, and
95
+ // request streaming is lost outright on this path — a 1 GiB upload from a
96
+ // Firefox page is a 1 GiB allocation here.
97
+ //
98
+ // The empty-versus-absent body divergence that `fetch.ts` has to live
99
+ // with does not arise on this transport: `SerializedHttpEnvelope` always
100
+ // carries a `content` iterable, so a zero-length body and no body are
101
+ // already the same thing here, whichever runtime produced it.
102
+ content = oneChunk(new Uint8Array(await request.arrayBuffer()));
103
+ } else {
104
+ content = noBytes();
105
+ }
69
106
 
70
107
  const result = await send({ options, content });
71
108
 
@@ -77,14 +114,14 @@ export function newHttpClientStub(
77
114
  const method = options.method;
78
115
  // `new Response(body, ...)` throws for null-body statuses (204/205/304), and
79
116
  // HEAD/OPTIONS carry no body either — drain the (empty) content stream and
80
- // emit a bodyless response.
117
+ // emit a bodyless response. `discard` rather than a bare `.return()`: see
118
+ // the note at the matching site in `newHttpServerStub` below.
81
119
  if (
82
120
  method === "HEAD" ||
83
121
  method === "OPTIONS" ||
84
122
  NULL_BODY_STATUSES.has(responseOptions.status)
85
123
  ) {
86
- const returnable = result.content as AsyncIterable<Uint8Array> & { return?: () => unknown };
87
- await returnable.return?.();
124
+ await discard(result.content);
88
125
  return new Response(null, responseOptions);
89
126
  }
90
127
  return new Response(toReadableStream(result.content[Symbol.asyncIterator]()), responseOptions);
@@ -111,22 +148,39 @@ export function newHttpServerStub(
111
148
  for (const [key, value] of headers) requestHeaders.append(key, value);
112
149
 
113
150
  const hasBody = method !== "GET" && method !== "HEAD" && method !== "OPTIONS";
114
- let body: ReadableStream<Uint8Array> | undefined;
115
- if (hasBody) {
116
- body = toReadableStream(content[Symbol.asyncIterator]());
117
- } else {
118
- const returnable = content as AsyncIterable<Uint8Array> & { return?: () => unknown };
119
- await returnable.return?.();
120
- }
121
-
122
151
  const requestInit: RequestInit & { duplex?: "half" } = {
123
152
  ...forwardable,
124
153
  method,
125
154
  headers: requestHeaders,
126
155
  };
127
- if (hasBody) {
128
- requestInit.body = body;
156
+ if (!hasBody) {
157
+ // `discard`, not a bare `.return()`. GET/HEAD/OPTIONS arrive with a
158
+ // `content` iterable the transport built but nobody has pulled from yet,
159
+ // and `.return()` on an async generator still in *suspended start* is a
160
+ // no-op: the body never runs, its `try/finally` never unwinds, and
161
+ // whatever the producer holds open — a socket, a port subscription — is
162
+ // never released. `discard` calls `.next()` first for exactly this
163
+ // reason; see its comment in `bytes.ts`.
164
+ //
165
+ // `fetch.ts` has the same shape at its null-body-status branch and still
166
+ // uses the bare form — not because it is safe there (it is not; its
167
+ // producer's `finally` does not run either) but because `discard` cannot
168
+ // fix it: the release has to reach the `output` generator that
169
+ // `fetchOverDuplex` has no handle on. See the KNOWN GAP note there. Here
170
+ // the producer is reachable, so here it gets released.
171
+ await discard(content);
172
+ } else if (supportsRequestStreams()) {
173
+ requestInit.body = toReadableStream(content[Symbol.asyncIterator]());
129
174
  requestInit.duplex = "half";
175
+ } else {
176
+ // Firefox again, and this is the half that fails *silently*. The
177
+ // constructor does not reject a `ReadableStream` here, it stringifies it:
178
+ // the handler would read the literal text `[object ReadableStream]` and
179
+ // answer 200 with corrupt data, where the client direction above at least
180
+ // sends an empty body a JSON endpoint rejects loudly. Bytes it does
181
+ // accept, so drain first — at the same cost as above, the whole upload in
182
+ // memory and no streaming left for the handler.
183
+ requestInit.body = await collectBytes(content);
130
184
  }
131
185
 
132
186
  const request = new Request(url, requestInit);
@@ -0,0 +1,89 @@
1
+ import type { ByteReader } from "../bytes.js";
2
+ import { ByteStreamError } from "../bytes.js";
3
+ import type { ByteSource } from "../message.js";
4
+ import { HttpParseError } from "./errors.js";
5
+ import { decodeLatin1 } from "./headers.js";
6
+
7
+ const CRLF = new Uint8Array([0x0d, 0x0a]);
8
+ const LAST_CHUNK = new Uint8Array([0x30, 0x0d, 0x0a, 0x0d, 0x0a]); // "0\r\n\r\n"
9
+ const encoder = new TextEncoder();
10
+
11
+ /**
12
+ * Chunk sizes are written in HEXADECIMAL — a 21-byte chunk is `15`. Writing
13
+ * them in decimal is the defect note 15 found in @libp2p/http, and it is the
14
+ * one that cannot be caught downstream: decimal digits are also valid hex, so
15
+ * a conforming parser silently reads the wrong length.
16
+ *
17
+ * Zero-length source chunks are skipped; a zero-sized chunk on the wire is the
18
+ * body terminator.
19
+ */
20
+ export async function* encodeChunked(body: ByteSource): AsyncGenerator<Uint8Array> {
21
+ for await (const chunk of body) {
22
+ if (chunk.byteLength === 0) continue;
23
+ yield encoder.encode(`${chunk.byteLength.toString(16)}\r\n`);
24
+ yield chunk;
25
+ yield CRLF;
26
+ }
27
+ yield LAST_CHUNK;
28
+ }
29
+
30
+ /**
31
+ * Decode a chunked body, yielding data chunks as they arrive. Nothing is
32
+ * accumulated: a chunk larger than the transport's frame is yielded in pieces.
33
+ */
34
+ export async function* decodeChunked(
35
+ reader: ByteReader,
36
+ maxLineBytes: number,
37
+ ): AsyncGenerator<Uint8Array> {
38
+ while (true) {
39
+ const sizeLine = decodeLatin1(await reader.readLine(maxLineBytes));
40
+ const semicolon = sizeLine.indexOf(";"); // chunk extensions: accepted, ignored
41
+ const sizeText = semicolon === -1 ? sizeLine : sizeLine.slice(0, semicolon);
42
+ if (!/^[0-9a-fA-F]{1,32}$/.test(sizeText)) {
43
+ throw new HttpParseError(`invalid chunk size: ${JSON.stringify(sizeLine)}`);
44
+ }
45
+ const size = Number.parseInt(sizeText, 16);
46
+ if (!Number.isSafeInteger(size)) {
47
+ throw new HttpParseError(`chunk size is too large: ${JSON.stringify(sizeLine)}`);
48
+ }
49
+
50
+ if (size === 0) {
51
+ // Trailer section: legal input, read and discarded, never re-emitted.
52
+ let used = 0;
53
+ while (true) {
54
+ const remaining = maxLineBytes - used;
55
+ if (remaining <= 0) {
56
+ throw new HttpParseError(`trailer section exceeds ${maxLineBytes} bytes`);
57
+ }
58
+ const lineBytes = await reader.readLine(remaining);
59
+ used += lineBytes.byteLength + 2;
60
+ if (used > maxLineBytes) {
61
+ throw new HttpParseError(`trailer section exceeds ${maxLineBytes} bytes`);
62
+ }
63
+ if (lineBytes.byteLength === 0) break;
64
+ }
65
+ return;
66
+ }
67
+
68
+ let remaining = size;
69
+ while (remaining > 0) {
70
+ const part = await reader.readSome(remaining);
71
+ if (part === undefined) {
72
+ throw new HttpParseError(`chunk truncated: ${remaining} of ${size} bytes missing`);
73
+ }
74
+ remaining -= part.byteLength;
75
+ yield part;
76
+ }
77
+
78
+ try {
79
+ if ((await reader.readLine(2)).byteLength !== 0) {
80
+ throw new HttpParseError("chunk data not terminated by CRLF");
81
+ }
82
+ } catch (err) {
83
+ if (err instanceof ByteStreamError) {
84
+ throw new HttpParseError("chunk data not terminated by CRLF");
85
+ }
86
+ throw err;
87
+ }
88
+ }
89
+ }
@@ -0,0 +1,208 @@
1
+ import { ByteReader, ByteStreamError } from "../bytes.js";
2
+ import type { ByteSource, DecodedRequest, DecodedResponse } from "../message.js";
3
+ import { decodeChunked } from "./chunked.js";
4
+ import type { ResolvedHttpCodecOptions } from "./encode.js";
5
+ import { HttpParseError } from "./errors.js";
6
+ import {
7
+ assertValidHost,
8
+ assertValidTarget,
9
+ type BodyFraming,
10
+ decodeLatin1,
11
+ getAll,
12
+ isBodylessStatus,
13
+ isToken,
14
+ readHeaderSection,
15
+ resolveFraming,
16
+ } from "./headers.js";
17
+
18
+ const VERSIONS = new Set(["HTTP/1.1", "HTTP/1.0"]);
19
+ const ABSOLUTE_FORM = /^[a-zA-Z][a-zA-Z0-9+.-]*:\/\//;
20
+
21
+ /**
22
+ * One message per Duplex call (ADR-0006), so bytes after a complete message
23
+ * are an error. Checked against what is ALREADY BUFFERED rather than by
24
+ * awaiting end-of-stream: a live socket from a keep-alive peer never reaches
25
+ * EOF, so awaiting one would hang instead of failing.
26
+ */
27
+ function assertNoBufferedBytes(reader: ByteReader): void {
28
+ const extra = reader.bufferedLength();
29
+ if (extra > 0) {
30
+ throw new HttpParseError(`${extra} trailing bytes after a complete message`);
31
+ }
32
+ }
33
+
34
+ /**
35
+ * The public error contract (I2): everything leaving `decodeRequest` /
36
+ * `decodeResponse` is an `HttpParseError`, whether the refusal happened
37
+ * synchronously (a malformed start line or header) or lazily while the body
38
+ * is later drained. `ByteStreamError` — the `ByteReader`'s own class, never
39
+ * exported — is the one thing converted here, message and all preserved via
40
+ * `cause`. Anything else is a genuine failure of the underlying source (a
41
+ * dropped socket, say) and must reach the caller unchanged: blanket-catching
42
+ * would hide that distinction.
43
+ */
44
+ function toHttpParseError(err: unknown): never {
45
+ if (err instanceof ByteStreamError) {
46
+ throw new HttpParseError(err.message, { cause: err });
47
+ }
48
+ throw err;
49
+ }
50
+
51
+ /** Applies `toHttpParseError` across the whole lifetime of a body generator. */
52
+ async function* convertBodyErrors(source: AsyncGenerator<Uint8Array>): AsyncGenerator<Uint8Array> {
53
+ try {
54
+ yield* source;
55
+ } catch (err) {
56
+ toHttpParseError(err);
57
+ }
58
+ }
59
+
60
+ async function* readBody(
61
+ reader: ByteReader,
62
+ framing: BodyFraming,
63
+ opts: ResolvedHttpCodecOptions,
64
+ noneMeansEof: boolean,
65
+ ): AsyncGenerator<Uint8Array> {
66
+ if (framing.kind === "chunked") {
67
+ yield* decodeChunked(reader, opts.maxHeaderBytes);
68
+ assertNoBufferedBytes(reader);
69
+ return;
70
+ }
71
+ if (framing.kind === "length") {
72
+ let remaining = framing.length;
73
+ while (remaining > 0) {
74
+ const part = await reader.readSome(remaining);
75
+ if (part === undefined) {
76
+ throw new HttpParseError(`body truncated: ${remaining} of ${framing.length} bytes missing`);
77
+ }
78
+ remaining -= part.byteLength;
79
+ yield part;
80
+ }
81
+ assertNoBufferedBytes(reader);
82
+ return;
83
+ }
84
+ if (noneMeansEof) {
85
+ yield* reader.rest();
86
+ return;
87
+ }
88
+ assertNoBufferedBytes(reader);
89
+ }
90
+
91
+ /**
92
+ * `readLine`'s bound is best-effort — it is skipped when the line is already
93
+ * buffered — so a very long start line can reach these messages intact. Since
94
+ * a refusal is now echoed back to the sender in a 400 body, quote only enough
95
+ * to diagnose rather than reflecting the whole thing.
96
+ */
97
+ function quoteLine(line: string): string {
98
+ const MAX = 120;
99
+ return line.length <= MAX
100
+ ? JSON.stringify(line)
101
+ : `${JSON.stringify(line.slice(0, MAX))} (truncated from ${line.length} chars)`;
102
+ }
103
+
104
+ export async function decodeRequest(
105
+ input: ByteSource,
106
+ opts: ResolvedHttpCodecOptions,
107
+ ): Promise<DecodedRequest> {
108
+ const reader = new ByteReader(input);
109
+ try {
110
+ const startBytes = await reader.readLine(opts.maxHeaderBytes);
111
+ const startLine = decodeLatin1(startBytes);
112
+ const parts = startLine.split(" ");
113
+ if (parts.length !== 3) {
114
+ throw new HttpParseError(`malformed request line: ${quoteLine(startLine)}`);
115
+ }
116
+ // `parts.length === 3` is checked above, so all three elements are present.
117
+ const [method, target, version] = parts as [string, string, string];
118
+ if (!isToken(method)) throw new HttpParseError(`invalid method: ${JSON.stringify(method)}`);
119
+ if (!VERSIONS.has(version)) {
120
+ throw new HttpParseError(`unsupported HTTP version: ${JSON.stringify(version)}`);
121
+ }
122
+ if (target === "*") {
123
+ throw new HttpParseError("asterisk-form request target is not supported");
124
+ }
125
+ // A relay that decodes then re-encodes must never be able to put a
126
+ // control character (a raw CR chief among them) back onto a request
127
+ // line — see C1.
128
+ assertValidTarget(target);
129
+
130
+ const headers = await readHeaderSection(reader, opts.maxHeaderBytes, startBytes.byteLength + 2);
131
+
132
+ const hosts = getAll(headers, "host");
133
+ if (hosts.length > 1) throw new HttpParseError("multiple Host headers");
134
+ const host = hosts[0];
135
+
136
+ let url: string;
137
+ if (target.startsWith("/")) {
138
+ if (version === "HTTP/1.1" && host === undefined) {
139
+ throw new HttpParseError("HTTP/1.1 request has no Host header");
140
+ }
141
+ if (host !== undefined) assertValidHost(host);
142
+ // The scheme is not on the wire in origin-form; it comes from config.
143
+ url = `${opts.scheme}://${host ?? opts.host}${target}`;
144
+ } else if (ABSOLUTE_FORM.test(target)) {
145
+ url = target;
146
+ } else {
147
+ throw new HttpParseError(`unsupported request target: ${JSON.stringify(target)}`);
148
+ }
149
+
150
+ // `Host` grammar alone is not sufficient proof that the assembled url is
151
+ // sane — this is a fail-safe belt-and-braces check, not a
152
+ // re-serialisation: the parsed `URL` is discarded, and `url` itself is
153
+ // what reaches the envelope untouched.
154
+ try {
155
+ new URL(url);
156
+ } catch {
157
+ throw new HttpParseError(`decoded url is not a valid URL: ${JSON.stringify(url)}`);
158
+ }
159
+
160
+ return {
161
+ envelope: { url, method, headers },
162
+ body: convertBodyErrors(readBody(reader, resolveFraming(headers, version), opts, false)),
163
+ };
164
+ } catch (err) {
165
+ toHttpParseError(err);
166
+ }
167
+ }
168
+
169
+ export async function decodeResponse(
170
+ input: ByteSource,
171
+ opts: ResolvedHttpCodecOptions,
172
+ method: string,
173
+ ): Promise<DecodedResponse> {
174
+ const reader = new ByteReader(input);
175
+ try {
176
+ const startBytes = await reader.readLine(opts.maxHeaderBytes);
177
+ const startLine = decodeLatin1(startBytes);
178
+
179
+ const firstSp = startLine.indexOf(" ");
180
+ if (firstSp === -1) {
181
+ throw new HttpParseError(`malformed status line: ${quoteLine(startLine)}`);
182
+ }
183
+ const version = startLine.slice(0, firstSp);
184
+ if (!VERSIONS.has(version)) {
185
+ throw new HttpParseError(`unsupported HTTP version: ${JSON.stringify(version)}`);
186
+ }
187
+ const afterVersion = startLine.slice(firstSp + 1);
188
+ const secondSp = afterVersion.indexOf(" ");
189
+ const codeText = secondSp === -1 ? afterVersion : afterVersion.slice(0, secondSp);
190
+ if (!/^\d{3}$/.test(codeText)) {
191
+ throw new HttpParseError(`invalid status code: ${JSON.stringify(codeText)}`);
192
+ }
193
+ const status = Number(codeText);
194
+ const statusText = secondSp === -1 ? "" : afterVersion.slice(secondSp + 1);
195
+
196
+ const headers = await readHeaderSection(reader, opts.maxHeaderBytes, startBytes.byteLength + 2);
197
+
198
+ const bodyless = isBodylessStatus(status) || method.toUpperCase() === "HEAD";
199
+ const framing: BodyFraming = bodyless ? { kind: "none" } : resolveFraming(headers, version);
200
+
201
+ return {
202
+ envelope: { status, statusText, headers },
203
+ body: convertBodyErrors(readBody(reader, framing, opts, !bodyless)),
204
+ };
205
+ } catch (err) {
206
+ toHttpParseError(err);
207
+ }
208
+ }