@statewalker/webrun-http-streams 0.1.1 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/README.md +288 -11
  2. package/dist/bytes.d.ts +43 -0
  3. package/dist/bytes.d.ts.map +1 -0
  4. package/dist/codec-default.d.ts +7 -0
  5. package/dist/codec-default.d.ts.map +1 -0
  6. package/dist/duplex-site-builder.d.ts +3 -0
  7. package/dist/duplex-site-builder.d.ts.map +1 -1
  8. package/dist/envelope.d.ts +10 -10
  9. package/dist/envelope.d.ts.map +1 -1
  10. package/dist/fetch.d.ts +3 -2
  11. package/dist/fetch.d.ts.map +1 -1
  12. package/dist/http-data.d.ts +11 -7
  13. package/dist/http-data.d.ts.map +1 -1
  14. package/dist/http-error.d.ts.map +1 -1
  15. package/dist/http-stubs.d.ts +12 -0
  16. package/dist/http-stubs.d.ts.map +1 -1
  17. package/dist/http1/chunked.d.ts +18 -0
  18. package/dist/http1/chunked.d.ts.map +1 -0
  19. package/dist/http1/decode.d.ts +5 -0
  20. package/dist/http1/decode.d.ts.map +1 -0
  21. package/dist/http1/encode.d.ts +20 -0
  22. package/dist/http1/encode.d.ts.map +1 -0
  23. package/dist/http1/errors.d.ts +9 -0
  24. package/dist/http1/errors.d.ts.map +1 -0
  25. package/dist/http1/headers.d.ts +78 -0
  26. package/dist/http1/headers.d.ts.map +1 -0
  27. package/dist/http1/index.d.ts +18 -0
  28. package/dist/http1/index.d.ts.map +1 -0
  29. package/dist/index.d.ts +6 -2
  30. package/dist/index.d.ts.map +1 -1
  31. package/dist/index.js +1039 -137
  32. package/dist/message.d.ts +50 -0
  33. package/dist/message.d.ts.map +1 -0
  34. package/dist/request-streams.d.ts +52 -0
  35. package/dist/request-streams.d.ts.map +1 -0
  36. package/dist/sniff.d.ts +18 -0
  37. package/dist/sniff.d.ts.map +1 -0
  38. package/package.json +8 -6
  39. package/src/bytes.ts +157 -0
  40. package/src/codec-default.ts +13 -0
  41. package/src/duplex-site-builder.ts +9 -1
  42. package/src/envelope.ts +43 -34
  43. package/src/fetch.ts +130 -12
  44. package/src/http-data.ts +160 -17
  45. package/src/http-stubs.ts +72 -18
  46. package/src/http1/chunked.ts +89 -0
  47. package/src/http1/decode.ts +208 -0
  48. package/src/http1/encode.ts +181 -0
  49. package/src/http1/errors.ts +8 -0
  50. package/src/http1/headers.ts +263 -0
  51. package/src/http1/index.ts +40 -0
  52. package/src/index.ts +18 -0
  53. package/src/message.ts +62 -0
  54. package/src/request-streams.ts +75 -0
  55. package/src/sniff.ts +68 -0
@@ -0,0 +1,181 @@
1
+ import { discard } from "../bytes.js";
2
+ import type { ByteSource, RequestEnvelope, ResponseEnvelope } from "../message.js";
3
+ import { encodeChunked } from "./chunked.js";
4
+ import { HttpParseError } from "./errors.js";
5
+ import {
6
+ assertValidHost,
7
+ assertValidStatusText,
8
+ assertValidTarget,
9
+ encodeHeaderLines,
10
+ encodeLatin1,
11
+ getAll,
12
+ isBodylessStatus,
13
+ isToken,
14
+ parseContentLength,
15
+ withoutHeaders,
16
+ } from "./headers.js";
17
+
18
+ export type ResolvedHttpCodecOptions = {
19
+ scheme: "http" | "https";
20
+ host: string;
21
+ maxHeaderBytes: number;
22
+ };
23
+
24
+ /** Headers the codec owns: a caller-supplied copy is dropped and re-derived. */
25
+ const REQUEST_OWNED = ["host", "connection", "transfer-encoding"];
26
+ const RESPONSE_OWNED = ["connection", "transfer-encoding"];
27
+
28
+ /**
29
+ * Split a URL into an origin-form request target and an authority *without*
30
+ * going through `new URL()`. `URL` normalises percent-encoding and would
31
+ * re-serialise the target — which is the exact class of defect note 15 found
32
+ * in @libp2p/http, where rebuilding the request line silently dropped
33
+ * `url.search`. Here the target is a verbatim slice of the caller's string.
34
+ */
35
+ export function splitTarget(
36
+ url: string,
37
+ opts: ResolvedHttpCodecOptions,
38
+ ): { target: string; authority: string } {
39
+ if (url.startsWith("/")) {
40
+ // Neither field is trusted just because it came from a caller's own url:
41
+ // an unvalidated CRLF here would splice a second request line (or an
42
+ // extra header) onto the wire — see C1.
43
+ assertValidTarget(url);
44
+ return { target: url, authority: opts.host };
45
+ }
46
+ const match = /^[a-zA-Z][a-zA-Z0-9+.-]*:\/\/([^/?#]*)([^#]*)/.exec(url);
47
+ if (!match) {
48
+ throw new HttpParseError(`cannot derive a request target from url: ${JSON.stringify(url)}`);
49
+ }
50
+ // Both capture groups are mandatory in the regex above (neither is
51
+ // `?`-quantified), so a successful match always populates them — an empty
52
+ // string is possible, `undefined` is not.
53
+ const rawAuthority = match[1]!;
54
+ if (rawAuthority === "") {
55
+ throw new HttpParseError(`url has no authority: ${JSON.stringify(url)}`);
56
+ }
57
+ // Host is `uri-host [ ":" port ]` — no userinfo (RFC 9110 §7.2). Strip any
58
+ // `user:pass@` prefix so credentials never end up in a header proxies log.
59
+ const at = rawAuthority.lastIndexOf("@");
60
+ const authority = at === -1 ? rawAuthority : rawAuthority.slice(at + 1);
61
+ assertValidHost(authority);
62
+
63
+ const rawTarget = match[2]!;
64
+ const target = rawTarget === "" ? "/" : rawTarget.startsWith("/") ? rawTarget : `/${rawTarget}`;
65
+ assertValidTarget(target);
66
+ return { target, authority };
67
+ }
68
+
69
+ async function* emitBody(
70
+ body: ByteSource,
71
+ declared: number | undefined,
72
+ ): AsyncGenerator<Uint8Array> {
73
+ if (declared === undefined) {
74
+ yield* encodeChunked(body);
75
+ return;
76
+ }
77
+ let sent = 0;
78
+ for await (const chunk of body) {
79
+ if (chunk.byteLength === 0) continue;
80
+ sent += chunk.byteLength;
81
+ if (sent > declared) {
82
+ throw new HttpParseError(
83
+ `body exceeds declared Content-Length ${declared} (${sent} bytes so far)`,
84
+ );
85
+ }
86
+ yield chunk;
87
+ }
88
+ if (sent !== declared) {
89
+ throw new HttpParseError(`body is ${sent} bytes but Content-Length declares ${declared}`);
90
+ }
91
+ }
92
+
93
+ export async function* encodeRequest(
94
+ env: RequestEnvelope,
95
+ body: ByteSource | undefined,
96
+ opts: ResolvedHttpCodecOptions,
97
+ ): AsyncGenerator<Uint8Array> {
98
+ if (!isToken(env.method)) {
99
+ throw new HttpParseError(`invalid method: ${JSON.stringify(env.method)}`);
100
+ }
101
+ const { target, authority } = splitTarget(env.url, opts);
102
+ if (authority === "") {
103
+ throw new HttpParseError("no Host available: url has no authority and no host is configured");
104
+ }
105
+
106
+ const carried = withoutHeaders(env.headers, REQUEST_OWNED);
107
+ const declared = parseContentLength(getAll(carried, "content-length"));
108
+
109
+ let head = `${env.method} ${target} HTTP/1.1\r\n`;
110
+ head += `Host: ${authority}\r\n`;
111
+ head += encodeHeaderLines(carried);
112
+ head += "Connection: close\r\n";
113
+ if (body !== undefined && declared === undefined) head += "Transfer-Encoding: chunked\r\n";
114
+ head += "\r\n";
115
+ yield encodeLatin1(head);
116
+
117
+ if (body === undefined) {
118
+ // A head promising `declared` bytes with no body at all is a knowingly
119
+ // truncated message — the failure must land here, not on the peer that
120
+ // trusted the header. Content-Length: 0 is the one declared length an
121
+ // absent body actually satisfies. See I1.
122
+ if (declared !== undefined && declared !== 0) {
123
+ throw new HttpParseError(`body is 0 bytes but Content-Length declares ${declared}`);
124
+ }
125
+ return;
126
+ }
127
+ yield* emitBody(body, declared);
128
+ }
129
+
130
+ export async function* encodeResponse(
131
+ env: ResponseEnvelope,
132
+ body: ByteSource | undefined,
133
+ // Unused: response encoding needs no scheme/host/maxHeaderBytes. Kept
134
+ // positionally so this signature mirrors `encodeRequest`'s, which the
135
+ // `newHttpCodec` wrapper in ./index.ts relies on when partially applying
136
+ // `opts` to both.
137
+ _opts: ResolvedHttpCodecOptions,
138
+ requestMethod?: string,
139
+ ): AsyncGenerator<Uint8Array> {
140
+ if (!Number.isInteger(env.status) || env.status < 100 || env.status > 599) {
141
+ throw new HttpParseError(`invalid status: ${env.status}`);
142
+ }
143
+ const reason = env.statusText ?? "";
144
+ assertValidStatusText(reason);
145
+
146
+ // RFC 9110 §8.6: a 1xx or 204 MUST NOT carry Content-Length — there is no
147
+ // body to measure, ever, regardless of what the caller supplied. A
148
+ // response to HEAD is different: no body is sent on *this* response, but
149
+ // the Content-Length describes the body a GET would have sent, so it stays
150
+ // (M4).
151
+ const bodylessStatus = isBodylessStatus(env.status);
152
+ const bodyless = bodylessStatus || requestMethod?.toUpperCase() === "HEAD";
153
+
154
+ const owned = bodylessStatus ? [...RESPONSE_OWNED, "content-length"] : RESPONSE_OWNED;
155
+ const carried = withoutHeaders(env.headers, owned);
156
+ const declared = parseContentLength(getAll(carried, "content-length"));
157
+
158
+ let head = `HTTP/1.1 ${env.status} ${reason}\r\n`;
159
+ head += encodeHeaderLines(carried);
160
+ head += "Connection: close\r\n";
161
+ if (!bodyless && body !== undefined && declared === undefined) {
162
+ head += "Transfer-Encoding: chunked\r\n";
163
+ }
164
+ head += "\r\n";
165
+ yield encodeLatin1(head);
166
+
167
+ if (bodyless) {
168
+ await discard(body);
169
+ return;
170
+ }
171
+ if (body === undefined) {
172
+ // Same truncation guard as encodeRequest — see I1. Not reached when
173
+ // bodyless: a response to HEAD legitimately keeps a declared
174
+ // Content-Length with no body (M4), and that is not a truncation.
175
+ if (declared !== undefined && declared !== 0) {
176
+ throw new HttpParseError(`body is 0 bytes but Content-Length declares ${declared}`);
177
+ }
178
+ return;
179
+ }
180
+ yield* emitBody(body, declared);
181
+ }
@@ -0,0 +1,8 @@
1
+ /**
2
+ * Raised for any byte sequence this codec refuses to interpret. Every case is
3
+ * a refusal to guess: HTTP/1.1 parsers that guess are how request smuggling
4
+ * works.
5
+ */
6
+ export class HttpParseError extends Error {
7
+ override readonly name = "HttpParseError";
8
+ }
@@ -0,0 +1,263 @@
1
+ import type { ByteReader } from "../bytes.js";
2
+ import { HttpParseError } from "./errors.js";
3
+
4
+ export type HeaderList = [string, string][];
5
+
6
+ const TCHAR = new Set(
7
+ "!#$%&'*+-.^_`|~0123456789abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ",
8
+ );
9
+
10
+ const SP = 0x20;
11
+ const HTAB = 0x09;
12
+
13
+ export function isTokenChar(byte: number): boolean {
14
+ return byte > SP && byte < 0x7f && TCHAR.has(String.fromCharCode(byte));
15
+ }
16
+
17
+ export function isToken(value: string): boolean {
18
+ if (value.length === 0) return false;
19
+ for (const ch of value) {
20
+ if (!TCHAR.has(ch)) return false;
21
+ }
22
+ return true;
23
+ }
24
+
25
+ /**
26
+ * Header bytes are latin-1: byte-preserving, matching Node, and lossless
27
+ * across a decode/encode round-trip. Control characters are rejected
28
+ * separately by `assertValidHeaderValue`.
29
+ */
30
+ export function decodeLatin1(bytes: Uint8Array): string {
31
+ let out = "";
32
+ for (const byte of bytes) out += String.fromCharCode(byte);
33
+ return out;
34
+ }
35
+
36
+ export function encodeLatin1(text: string): Uint8Array {
37
+ const out = new Uint8Array(text.length);
38
+ for (let i = 0; i < text.length; i++) {
39
+ const code = text.charCodeAt(i);
40
+ if (code > 0xff) {
41
+ throw new HttpParseError(`not latin-1 encodable: ${JSON.stringify(text)}`);
42
+ }
43
+ out[i] = code;
44
+ }
45
+ return out;
46
+ }
47
+
48
+ // RFC 3986 `authority` grammar, restricted to `host [ ":" port ]` (no
49
+ // userinfo — RFC 9110 §7.2 excludes it from Host). Two shapes: an IP-literal
50
+ // in brackets, or a reg-name. Shared by the decoder (validating a Host header
51
+ // taken off the wire) and the encoder (validating the authority `splitTarget`
52
+ // derives from a caller-supplied url) so the grammar is defined exactly once.
53
+ const HOST_IP_LITERAL = /^\[[0-9A-Fa-f:.]+\](:\d{1,5})?$/;
54
+ const HOST_REG_NAME = /^[A-Za-z0-9._~-]+(:\d{1,5})?$/;
55
+
56
+ /**
57
+ * Validates a value that is about to become (or was read as) a `Host`
58
+ * header: untrusted input either way, so an unrecognised shape is a refusal,
59
+ * not a guess. Rejects a present-but-empty value, one carrying userinfo
60
+ * (`evil.com@good.com`), and — the reason this also runs on encode — one
61
+ * smuggling a CRLF-terminated line into the authority of a caller-supplied
62
+ * url. Never applied to trusted configuration (a codec's `opts.host`).
63
+ */
64
+ export function assertValidHost(host: string): void {
65
+ if (!HOST_IP_LITERAL.test(host) && !HOST_REG_NAME.test(host)) {
66
+ throw new HttpParseError(`invalid Host header: ${JSON.stringify(host)}`);
67
+ }
68
+ }
69
+
70
+ /**
71
+ * A request-target must be entirely visible ASCII (RFC 9110 VCHAR) — nothing
72
+ * `<= 0x20` (space and every control character, CR/LF included) and nothing
73
+ * `>= 0x7F` (DEL and beyond). Applied on encode (`splitTarget`, so a
74
+ * caller-supplied url can never inject a second request line) and on decode
75
+ * (so a relay that decodes then re-encodes can never put an embedded control
76
+ * character back on the wire).
77
+ */
78
+ export function assertValidTarget(target: string): void {
79
+ for (let i = 0; i < target.length; i++) {
80
+ const code = target.charCodeAt(i);
81
+ if (code <= SP || code >= 0x7f) {
82
+ throw new HttpParseError(`invalid request-target: ${JSON.stringify(target)}`);
83
+ }
84
+ }
85
+ }
86
+
87
+ /**
88
+ * Shared control-character check behind both `assertValidHeaderValue` and
89
+ * `assertValidStatusText` — the two were previously checked by different,
90
+ * looser rules (statusText only rejected CR/LF), which is exactly the
91
+ * asymmetry class C1 and M5 are both instances of. `label` is prepended
92
+ * verbatim to each message, so callers keep their own wording.
93
+ */
94
+ function assertNoControlChars(label: string, value: string): void {
95
+ for (let i = 0; i < value.length; i++) {
96
+ const code = value.charCodeAt(i);
97
+ if (code === 0x0d || code === 0x0a) {
98
+ throw new HttpParseError(`${label} contains CR or LF`);
99
+ }
100
+ if ((code < SP && code !== HTAB) || code === 0x7f) {
101
+ throw new HttpParseError(`${label} contains a control character`);
102
+ }
103
+ if (code > 0xff) {
104
+ throw new HttpParseError(`${label} is not latin-1 encodable`);
105
+ }
106
+ }
107
+ }
108
+
109
+ export function assertValidHeaderValue(name: string, value: string): void {
110
+ assertNoControlChars(`header "${name}" value`, value);
111
+ }
112
+
113
+ /** Same character class as a header value (M5) — CR/LF, every C0 control, and DEL. */
114
+ export function assertValidStatusText(value: string): void {
115
+ assertNoControlChars("statusText", value);
116
+ }
117
+
118
+ export function encodeHeaderLines(headers: HeaderList): string {
119
+ let out = "";
120
+ for (const [name, value] of headers) {
121
+ if (!isToken(name)) throw new HttpParseError(`invalid header name: ${JSON.stringify(name)}`);
122
+ assertValidHeaderValue(name, value);
123
+ out += `${name}: ${value}\r\n`;
124
+ }
125
+ return out;
126
+ }
127
+
128
+ /**
129
+ * Read to the blank line. `alreadyUsed` is the byte count of the start line,
130
+ * so the bound covers the whole head section rather than the headers alone.
131
+ */
132
+ export async function readHeaderSection(
133
+ reader: ByteReader,
134
+ maxHeaderBytes: number,
135
+ alreadyUsed: number,
136
+ ): Promise<HeaderList> {
137
+ const headers: HeaderList = [];
138
+ let used = alreadyUsed;
139
+ while (true) {
140
+ const remaining = maxHeaderBytes - used;
141
+ if (remaining <= 0) {
142
+ throw new HttpParseError(`head section exceeds ${maxHeaderBytes} bytes`);
143
+ }
144
+ const lineBytes = await reader.readLine(remaining);
145
+ used += lineBytes.byteLength + 2;
146
+ if (used > maxHeaderBytes) {
147
+ throw new HttpParseError(`head section exceeds ${maxHeaderBytes} bytes`);
148
+ }
149
+ if (lineBytes.byteLength === 0) return headers;
150
+ if (lineBytes[0] === SP || lineBytes[0] === HTAB) {
151
+ throw new HttpParseError("obs-fold header continuation is not accepted");
152
+ }
153
+ const line = decodeLatin1(lineBytes);
154
+ const colon = line.indexOf(":");
155
+ if (colon <= 0) throw new HttpParseError(`malformed header line: ${JSON.stringify(line)}`);
156
+ const name = line.slice(0, colon);
157
+ if (!isToken(name)) throw new HttpParseError(`invalid header name: ${JSON.stringify(name)}`);
158
+ const value = line
159
+ .slice(colon + 1)
160
+ .replace(/^[ \t]+/, "")
161
+ .replace(/[ \t]+$/, "");
162
+ assertValidHeaderValue(name, value);
163
+ headers.push([name, value]);
164
+ }
165
+ }
166
+
167
+ export function getAll(headers: HeaderList, name: string): string[] {
168
+ const lower = name.toLowerCase();
169
+ return headers.filter(([k]) => k.toLowerCase() === lower).map(([, v]) => v);
170
+ }
171
+
172
+ export function withoutHeaders(headers: HeaderList, names: string[]): HeaderList {
173
+ const drop = new Set(names.map((n) => n.toLowerCase()));
174
+ return headers.filter(([k]) => !drop.has(k.toLowerCase()));
175
+ }
176
+
177
+ /**
178
+ * Parse a `Content-Length` field into a single validated value.
179
+ *
180
+ * RFC 9110 §8.6 permits the value to be a comma-separated list of identical
181
+ * numbers (a relay may have appended one), so the list is folded; differing
182
+ * values are a refusal, not a choice. Shared by `resolveFraming` on decode and
183
+ * by the encoder's declared-length check, which previously carried its own
184
+ * copy that did NOT split on commas — so decode accepted `5, 5` while encode
185
+ * rejected it, and a relay that decoded then re-encoded threw.
186
+ *
187
+ * Returns undefined when the field is absent.
188
+ */
189
+ export function parseContentLength(values: string[]): number | undefined {
190
+ if (values.length === 0) return undefined;
191
+ const unique = new Set(values.flatMap((v) => v.split(",").map((s) => s.trim())));
192
+ if (unique.size !== 1) {
193
+ throw new HttpParseError(`conflicting Content-Length values: ${[...unique].join(", ")}`);
194
+ }
195
+ // `unique.size === 1` is checked directly above, so this spread has exactly
196
+ // one element. The `!` is what lets a consumer compiling with
197
+ // `noUncheckedIndexedAccess` build against this source: without it `raw` is
198
+ // `string | undefined` and the `.test(raw)` below is a type error in THEIR
199
+ // build, not in this package's own (which does not enable the flag, so it
200
+ // cannot catch the regression). It was lost once already, in a rebase that
201
+ // resolved toward a refactor of this function.
202
+ const raw = [...unique][0]!;
203
+ if (!/^\d{1,15}$/.test(raw)) {
204
+ throw new HttpParseError(`invalid Content-Length: ${JSON.stringify(raw)}`);
205
+ }
206
+ return Number(raw);
207
+ }
208
+
209
+ /**
210
+ * Statuses that carry no body whatever the headers say (RFC 9110 §8.6): 1xx,
211
+ * 204 and 304. Defined once because the encoder must not frame a body for
212
+ * them and the decoder must not try to read one — two lists that agreed today
213
+ * and were free to drift tomorrow.
214
+ *
215
+ * A response to HEAD is also bodyless, but that depends on the request rather
216
+ * than the status, so callers test it separately.
217
+ */
218
+ export function isBodylessStatus(status: number): boolean {
219
+ return status < 200 || status === 204 || status === 304;
220
+ }
221
+
222
+ export type BodyFraming =
223
+ | { kind: "chunked" }
224
+ | { kind: "length"; length: number }
225
+ | { kind: "none" };
226
+
227
+ /**
228
+ * RFC 9112 §6.3, with every ambiguity turned into a refusal. In particular a
229
+ * message declaring both Content-Length and Transfer-Encoding is rejected
230
+ * rather than resolved — disagreeing on which one wins is request smuggling.
231
+ */
232
+ export function resolveFraming(headers: HeaderList, version: string): BodyFraming {
233
+ const te = getAll(headers, "transfer-encoding");
234
+ const cl = getAll(headers, "content-length");
235
+
236
+ if (te.length > 0 && cl.length > 0) {
237
+ throw new HttpParseError(
238
+ "message declares both Content-Length and Transfer-Encoding; refusing (request smuggling)",
239
+ );
240
+ }
241
+
242
+ if (te.length > 0) {
243
+ const encodings = te
244
+ .join(",")
245
+ .split(",")
246
+ .map((s) => s.trim().toLowerCase())
247
+ .filter((s) => s !== "");
248
+ if (encodings.length !== 1 || encodings[0] !== "chunked") {
249
+ throw new HttpParseError(`unsupported Transfer-Encoding: ${JSON.stringify(te.join(", "))}`);
250
+ }
251
+ if (version === "HTTP/1.0") {
252
+ throw new HttpParseError("Transfer-Encoding is not valid in HTTP/1.0");
253
+ }
254
+ return { kind: "chunked" };
255
+ }
256
+
257
+ const length = parseContentLength(cl);
258
+ if (length !== undefined) {
259
+ return { kind: "length", length };
260
+ }
261
+
262
+ return { kind: "none" };
263
+ }
@@ -0,0 +1,40 @@
1
+ import type { MessageCodec } from "../message.js";
2
+ import { decodeRequest, decodeResponse } from "./decode.js";
3
+ import { encodeRequest, encodeResponse, type ResolvedHttpCodecOptions } from "./encode.js";
4
+ import { isTokenChar } from "./headers.js";
5
+
6
+ export type HttpCodecOptions = {
7
+ /**
8
+ * Scheme used to rebuild an absolute url on decode. HTTP/1.1 origin-form
9
+ * carries no scheme, so this is configuration, not wire data — a peer
10
+ * configured `http` rebuilds an `https` url as `http`. Accepted cost of
11
+ * decision 6; see ADR-0006.
12
+ */
13
+ scheme?: "http" | "https";
14
+ /** Authority used when a url carries none. Also fills the mandatory Host header. */
15
+ host?: string;
16
+ /** Bound on the whole head section, start line included. Default 65536. */
17
+ maxHeaderBytes?: number;
18
+ };
19
+
20
+ export function newHttpCodec(options: HttpCodecOptions = {}): MessageCodec {
21
+ const opts: ResolvedHttpCodecOptions = {
22
+ scheme: options.scheme ?? "http",
23
+ host: options.host ?? "localhost",
24
+ maxHeaderBytes: options.maxHeaderBytes ?? 65536,
25
+ };
26
+ return {
27
+ name: "http/1.1",
28
+ // A request starts with a method token, a response with "HTTP/1.1" — both
29
+ // begin with a token character. "{" is not a token character, so this is
30
+ // disjoint from jsonEnvelopeCodec by construction, not by convention.
31
+ sniff: (byte: number): boolean => isTokenChar(byte),
32
+ encodeRequest: (env, body) => encodeRequest(env, body, opts),
33
+ encodeResponse: (env, body, o) => encodeResponse(env, body, opts, o?.method),
34
+ decodeRequest: (input) => decodeRequest(input, opts),
35
+ decodeResponse: (input, o) => decodeResponse(input, opts, o.method),
36
+ };
37
+ }
38
+
39
+ export const httpCodec: MessageCodec = newHttpCodec();
40
+ export { HttpParseError } from "./errors.js";
package/src/index.ts CHANGED
@@ -1,7 +1,9 @@
1
+ export { defaultCodec } from "./codec-default.js";
1
2
  export { DuplexSiteBuilder, type SiteHandler } from "./duplex-site-builder.js";
2
3
  export {
3
4
  decodeMessage,
4
5
  encodeMessage,
6
+ jsonEnvelopeCodec,
5
7
  type RequestEnvelope,
6
8
  type ResponseEnvelope,
7
9
  } from "./envelope.js";
@@ -9,9 +11,11 @@ export { fetchOverDuplex, serveFetchOverDuplex } from "./fetch.js";
9
11
  export {
10
12
  type HttpDataHandler,
11
13
  type HttpDataHandlerResult,
14
+ type HttpDataOptions,
12
15
  type HttpFetchResult,
13
16
  httpFetch,
14
17
  httpServe,
18
+ PEER_ERROR_HEADER,
15
19
  } from "./http-data.js";
16
20
  export { HttpError, type HttpErrorOptions } from "./http-error.js";
17
21
  export {
@@ -22,3 +26,17 @@ export {
22
26
  type SerializedHttpRequest,
23
27
  type SerializedHttpResponse,
24
28
  } from "./http-stubs.js";
29
+ export {
30
+ type HttpCodecOptions,
31
+ HttpParseError,
32
+ httpCodec,
33
+ newHttpCodec,
34
+ } from "./http1/index.js";
35
+ export type {
36
+ ByteSource,
37
+ DecodedRequest,
38
+ DecodedResponse,
39
+ MessageCodec,
40
+ ResponseCodecOptions,
41
+ } from "./message.js";
42
+ export { newSniffingCodec, type SniffingCodecOptions } from "./sniff.js";
package/src/message.ts ADDED
@@ -0,0 +1,62 @@
1
+ /** Anything the codecs will read bytes from. */
2
+ export type ByteSource = AsyncIterable<Uint8Array> | Iterable<Uint8Array>;
3
+
4
+ export type RequestEnvelope = {
5
+ url: string;
6
+ method: string;
7
+ headers: [string, string][];
8
+ };
9
+
10
+ export type ResponseEnvelope = {
11
+ status: number;
12
+ statusText: string;
13
+ headers: [string, string][];
14
+ };
15
+
16
+ export type DecodedRequest = {
17
+ envelope: RequestEnvelope;
18
+ body: AsyncIterable<Uint8Array>;
19
+ /** The codec that actually read this message; set by the sniffing codec. */
20
+ codec?: MessageCodec;
21
+ };
22
+
23
+ export type DecodedResponse = {
24
+ envelope: ResponseEnvelope;
25
+ body: AsyncIterable<Uint8Array>;
26
+ };
27
+
28
+ /** Extra context a codec may need that the envelope does not carry. */
29
+ export type ResponseCodecOptions = {
30
+ /**
31
+ * Method of the request this response answers. Required by HTTP/1.1: a
32
+ * response to HEAD carries framing headers but no body, and no parser can
33
+ * know that from the response bytes alone.
34
+ */
35
+ method: string;
36
+ };
37
+
38
+ /**
39
+ * One wire format. Requests and responses are separate operations because
40
+ * HTTP/1.1 serialises them differently — a codec cannot be generic over the
41
+ * envelope the way the JSON format was.
42
+ */
43
+ export interface MessageCodec {
44
+ readonly name: string;
45
+
46
+ /**
47
+ * True if a message in this format may begin with `byte`. Used by the
48
+ * sniffing codec to dispatch without a negotiation handshake. Implementations
49
+ * must be mutually exclusive with every other codec they are paired with.
50
+ */
51
+ sniff(byte: number): boolean;
52
+
53
+ encodeRequest(env: RequestEnvelope, body?: ByteSource): AsyncGenerator<Uint8Array>;
54
+ encodeResponse(
55
+ env: ResponseEnvelope,
56
+ body?: ByteSource,
57
+ options?: ResponseCodecOptions,
58
+ ): AsyncGenerator<Uint8Array>;
59
+
60
+ decodeRequest(input: ByteSource): Promise<DecodedRequest>;
61
+ decodeResponse(input: ByteSource, options: ResponseCodecOptions): Promise<DecodedResponse>;
62
+ }
@@ -0,0 +1,75 @@
1
+ /**
2
+ * The one place that knows a runtime may not implement request body streams,
3
+ * and the buffering both fallbacks need. `fetch.ts` and `http-stubs.ts` each
4
+ * carry a client and a server direction that must agree on the answer, so the
5
+ * predicate lives here rather than being written out four times.
6
+ */
7
+
8
+ /**
9
+ * Whether `Request.prototype` exposes a `body` accessor. Two call sites read
10
+ * that property, and two more depend on the constructor accepting a
11
+ * `ReadableStream` as `init.body`; this predicate answers for all four.
12
+ *
13
+ * What was verified, and all that is claimed here: Firefox (146 at the time of
14
+ * writing) has *neither*, and Chromium and Node have *both*. On Firefox it is
15
+ * not that `body` is `undefined` on the instance —
16
+ * `Object.getOwnPropertyDescriptor(Request.prototype, "body")` is `null`, the
17
+ * accessor is genuinely absent — and because a `ReadableStream` is then not a
18
+ * recognised `BodyInit`, the constructor falls through to the string branch and
19
+ * stores the literal text `[object ReadableStream]`.
20
+ *
21
+ * The two halves are NOT guaranteed to ship together, so do not read this as a
22
+ * test for "request streams" in general. Safari is the counterexample: it has
23
+ * had `Request.body` since 11.1 but only accepts a stream as `init.body` from
24
+ * Technology Preview 250, so a shipping Safari has the reader half without the
25
+ * upload half and this returns `true` there. That looks benign — WebKit appears
26
+ * to store the stream on the `Request` rather than stringify it, and its error
27
+ * comes from `fetch()`, which this package never calls on the objects it builds
28
+ * — but it is untested, and it is the case to look at first if a Safari report
29
+ * arrives. The sharper probe, if one is ever needed, is whether
30
+ * `new Request(url, {method:"POST", body:new ReadableStream(), duplex:"half"})`
31
+ * has a `content-type` of `text/plain;charset=UTF-8` (stringified) or `null`
32
+ * (stored); it is not used here because it costs a `Request` and a
33
+ * `ReadableStream` per call and agrees with the descriptor check on every
34
+ * runtime measured.
35
+ *
36
+ * A capability check, never a user-agent test: the question is what this
37
+ * runtime does, and the answer flips on its own the day Firefox ships request
38
+ * streams. Evaluated per call rather than cached at module load so that a test
39
+ * can install a `Request` without the capability and exercise the real branch
40
+ * under Node.
41
+ */
42
+ export function supportsRequestStreams(): boolean {
43
+ return (
44
+ typeof Request === "function" &&
45
+ Object.getOwnPropertyDescriptor(Request.prototype, "body") != null
46
+ );
47
+ }
48
+
49
+ /**
50
+ * Drain a body iterable into one contiguous buffer. Only the fallbacks need it.
51
+ * Allocates its own buffer rather than reusing `bytes.ts`'s `concatChunks`,
52
+ * which passes a single chunk straight through: that chunk is a view onto the
53
+ * decoder's own read buffer, and the result here is handed to `new Request` and
54
+ * outlives the decode. The allocation is also what makes it a
55
+ * `Uint8Array<ArrayBuffer>`, which `BodyInit` accepts and the looser
56
+ * `Uint8Array<ArrayBufferLike>` does not.
57
+ */
58
+ export async function collectBytes(
59
+ source: AsyncIterable<Uint8Array>,
60
+ ): Promise<Uint8Array<ArrayBuffer>> {
61
+ const parts: Uint8Array[] = [];
62
+ let total = 0;
63
+ for await (const chunk of source) {
64
+ if (chunk.byteLength === 0) continue;
65
+ parts.push(chunk);
66
+ total += chunk.byteLength;
67
+ }
68
+ const out = new Uint8Array(total);
69
+ let offset = 0;
70
+ for (const part of parts) {
71
+ out.set(part, offset);
72
+ offset += part.byteLength;
73
+ }
74
+ return out;
75
+ }