@statewalker/webrun-http-streams 0.1.1 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +288 -11
- package/dist/bytes.d.ts +43 -0
- package/dist/bytes.d.ts.map +1 -0
- package/dist/codec-default.d.ts +7 -0
- package/dist/codec-default.d.ts.map +1 -0
- package/dist/duplex-site-builder.d.ts +3 -0
- package/dist/duplex-site-builder.d.ts.map +1 -1
- package/dist/envelope.d.ts +10 -10
- package/dist/envelope.d.ts.map +1 -1
- package/dist/fetch.d.ts +3 -2
- package/dist/fetch.d.ts.map +1 -1
- package/dist/http-data.d.ts +11 -7
- package/dist/http-data.d.ts.map +1 -1
- package/dist/http-error.d.ts.map +1 -1
- package/dist/http-stubs.d.ts +12 -0
- package/dist/http-stubs.d.ts.map +1 -1
- package/dist/http1/chunked.d.ts +18 -0
- package/dist/http1/chunked.d.ts.map +1 -0
- package/dist/http1/decode.d.ts +5 -0
- package/dist/http1/decode.d.ts.map +1 -0
- package/dist/http1/encode.d.ts +20 -0
- package/dist/http1/encode.d.ts.map +1 -0
- package/dist/http1/errors.d.ts +9 -0
- package/dist/http1/errors.d.ts.map +1 -0
- package/dist/http1/headers.d.ts +78 -0
- package/dist/http1/headers.d.ts.map +1 -0
- package/dist/http1/index.d.ts +18 -0
- package/dist/http1/index.d.ts.map +1 -0
- package/dist/index.d.ts +6 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1039 -137
- package/dist/message.d.ts +50 -0
- package/dist/message.d.ts.map +1 -0
- package/dist/request-streams.d.ts +52 -0
- package/dist/request-streams.d.ts.map +1 -0
- package/dist/sniff.d.ts +18 -0
- package/dist/sniff.d.ts.map +1 -0
- package/package.json +8 -6
- package/src/bytes.ts +157 -0
- package/src/codec-default.ts +13 -0
- package/src/duplex-site-builder.ts +9 -1
- package/src/envelope.ts +43 -34
- package/src/fetch.ts +130 -12
- package/src/http-data.ts +160 -17
- package/src/http-stubs.ts +72 -18
- package/src/http1/chunked.ts +89 -0
- package/src/http1/decode.ts +208 -0
- package/src/http1/encode.ts +181 -0
- package/src/http1/errors.ts +8 -0
- package/src/http1/headers.ts +263 -0
- package/src/http1/index.ts +40 -0
- package/src/index.ts +18 -0
- package/src/message.ts +62 -0
- package/src/request-streams.ts +75 -0
- package/src/sniff.ts +68 -0
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
import { discard } from "../bytes.js";
|
|
2
|
+
import type { ByteSource, RequestEnvelope, ResponseEnvelope } from "../message.js";
|
|
3
|
+
import { encodeChunked } from "./chunked.js";
|
|
4
|
+
import { HttpParseError } from "./errors.js";
|
|
5
|
+
import {
|
|
6
|
+
assertValidHost,
|
|
7
|
+
assertValidStatusText,
|
|
8
|
+
assertValidTarget,
|
|
9
|
+
encodeHeaderLines,
|
|
10
|
+
encodeLatin1,
|
|
11
|
+
getAll,
|
|
12
|
+
isBodylessStatus,
|
|
13
|
+
isToken,
|
|
14
|
+
parseContentLength,
|
|
15
|
+
withoutHeaders,
|
|
16
|
+
} from "./headers.js";
|
|
17
|
+
|
|
18
|
+
export type ResolvedHttpCodecOptions = {
|
|
19
|
+
scheme: "http" | "https";
|
|
20
|
+
host: string;
|
|
21
|
+
maxHeaderBytes: number;
|
|
22
|
+
};
|
|
23
|
+
|
|
24
|
+
/** Headers the codec owns: a caller-supplied copy is dropped and re-derived. */
|
|
25
|
+
const REQUEST_OWNED = ["host", "connection", "transfer-encoding"];
|
|
26
|
+
const RESPONSE_OWNED = ["connection", "transfer-encoding"];
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* Split a URL into an origin-form request target and an authority *without*
|
|
30
|
+
* going through `new URL()`. `URL` normalises percent-encoding and would
|
|
31
|
+
* re-serialise the target — which is the exact class of defect note 15 found
|
|
32
|
+
* in @libp2p/http, where rebuilding the request line silently dropped
|
|
33
|
+
* `url.search`. Here the target is a verbatim slice of the caller's string.
|
|
34
|
+
*/
|
|
35
|
+
export function splitTarget(
|
|
36
|
+
url: string,
|
|
37
|
+
opts: ResolvedHttpCodecOptions,
|
|
38
|
+
): { target: string; authority: string } {
|
|
39
|
+
if (url.startsWith("/")) {
|
|
40
|
+
// Neither field is trusted just because it came from a caller's own url:
|
|
41
|
+
// an unvalidated CRLF here would splice a second request line (or an
|
|
42
|
+
// extra header) onto the wire — see C1.
|
|
43
|
+
assertValidTarget(url);
|
|
44
|
+
return { target: url, authority: opts.host };
|
|
45
|
+
}
|
|
46
|
+
const match = /^[a-zA-Z][a-zA-Z0-9+.-]*:\/\/([^/?#]*)([^#]*)/.exec(url);
|
|
47
|
+
if (!match) {
|
|
48
|
+
throw new HttpParseError(`cannot derive a request target from url: ${JSON.stringify(url)}`);
|
|
49
|
+
}
|
|
50
|
+
// Both capture groups are mandatory in the regex above (neither is
|
|
51
|
+
// `?`-quantified), so a successful match always populates them — an empty
|
|
52
|
+
// string is possible, `undefined` is not.
|
|
53
|
+
const rawAuthority = match[1]!;
|
|
54
|
+
if (rawAuthority === "") {
|
|
55
|
+
throw new HttpParseError(`url has no authority: ${JSON.stringify(url)}`);
|
|
56
|
+
}
|
|
57
|
+
// Host is `uri-host [ ":" port ]` — no userinfo (RFC 9110 §7.2). Strip any
|
|
58
|
+
// `user:pass@` prefix so credentials never end up in a header proxies log.
|
|
59
|
+
const at = rawAuthority.lastIndexOf("@");
|
|
60
|
+
const authority = at === -1 ? rawAuthority : rawAuthority.slice(at + 1);
|
|
61
|
+
assertValidHost(authority);
|
|
62
|
+
|
|
63
|
+
const rawTarget = match[2]!;
|
|
64
|
+
const target = rawTarget === "" ? "/" : rawTarget.startsWith("/") ? rawTarget : `/${rawTarget}`;
|
|
65
|
+
assertValidTarget(target);
|
|
66
|
+
return { target, authority };
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
async function* emitBody(
|
|
70
|
+
body: ByteSource,
|
|
71
|
+
declared: number | undefined,
|
|
72
|
+
): AsyncGenerator<Uint8Array> {
|
|
73
|
+
if (declared === undefined) {
|
|
74
|
+
yield* encodeChunked(body);
|
|
75
|
+
return;
|
|
76
|
+
}
|
|
77
|
+
let sent = 0;
|
|
78
|
+
for await (const chunk of body) {
|
|
79
|
+
if (chunk.byteLength === 0) continue;
|
|
80
|
+
sent += chunk.byteLength;
|
|
81
|
+
if (sent > declared) {
|
|
82
|
+
throw new HttpParseError(
|
|
83
|
+
`body exceeds declared Content-Length ${declared} (${sent} bytes so far)`,
|
|
84
|
+
);
|
|
85
|
+
}
|
|
86
|
+
yield chunk;
|
|
87
|
+
}
|
|
88
|
+
if (sent !== declared) {
|
|
89
|
+
throw new HttpParseError(`body is ${sent} bytes but Content-Length declares ${declared}`);
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
export async function* encodeRequest(
|
|
94
|
+
env: RequestEnvelope,
|
|
95
|
+
body: ByteSource | undefined,
|
|
96
|
+
opts: ResolvedHttpCodecOptions,
|
|
97
|
+
): AsyncGenerator<Uint8Array> {
|
|
98
|
+
if (!isToken(env.method)) {
|
|
99
|
+
throw new HttpParseError(`invalid method: ${JSON.stringify(env.method)}`);
|
|
100
|
+
}
|
|
101
|
+
const { target, authority } = splitTarget(env.url, opts);
|
|
102
|
+
if (authority === "") {
|
|
103
|
+
throw new HttpParseError("no Host available: url has no authority and no host is configured");
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
const carried = withoutHeaders(env.headers, REQUEST_OWNED);
|
|
107
|
+
const declared = parseContentLength(getAll(carried, "content-length"));
|
|
108
|
+
|
|
109
|
+
let head = `${env.method} ${target} HTTP/1.1\r\n`;
|
|
110
|
+
head += `Host: ${authority}\r\n`;
|
|
111
|
+
head += encodeHeaderLines(carried);
|
|
112
|
+
head += "Connection: close\r\n";
|
|
113
|
+
if (body !== undefined && declared === undefined) head += "Transfer-Encoding: chunked\r\n";
|
|
114
|
+
head += "\r\n";
|
|
115
|
+
yield encodeLatin1(head);
|
|
116
|
+
|
|
117
|
+
if (body === undefined) {
|
|
118
|
+
// A head promising `declared` bytes with no body at all is a knowingly
|
|
119
|
+
// truncated message — the failure must land here, not on the peer that
|
|
120
|
+
// trusted the header. Content-Length: 0 is the one declared length an
|
|
121
|
+
// absent body actually satisfies. See I1.
|
|
122
|
+
if (declared !== undefined && declared !== 0) {
|
|
123
|
+
throw new HttpParseError(`body is 0 bytes but Content-Length declares ${declared}`);
|
|
124
|
+
}
|
|
125
|
+
return;
|
|
126
|
+
}
|
|
127
|
+
yield* emitBody(body, declared);
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
export async function* encodeResponse(
|
|
131
|
+
env: ResponseEnvelope,
|
|
132
|
+
body: ByteSource | undefined,
|
|
133
|
+
// Unused: response encoding needs no scheme/host/maxHeaderBytes. Kept
|
|
134
|
+
// positionally so this signature mirrors `encodeRequest`'s, which the
|
|
135
|
+
// `newHttpCodec` wrapper in ./index.ts relies on when partially applying
|
|
136
|
+
// `opts` to both.
|
|
137
|
+
_opts: ResolvedHttpCodecOptions,
|
|
138
|
+
requestMethod?: string,
|
|
139
|
+
): AsyncGenerator<Uint8Array> {
|
|
140
|
+
if (!Number.isInteger(env.status) || env.status < 100 || env.status > 599) {
|
|
141
|
+
throw new HttpParseError(`invalid status: ${env.status}`);
|
|
142
|
+
}
|
|
143
|
+
const reason = env.statusText ?? "";
|
|
144
|
+
assertValidStatusText(reason);
|
|
145
|
+
|
|
146
|
+
// RFC 9110 §8.6: a 1xx or 204 MUST NOT carry Content-Length — there is no
|
|
147
|
+
// body to measure, ever, regardless of what the caller supplied. A
|
|
148
|
+
// response to HEAD is different: no body is sent on *this* response, but
|
|
149
|
+
// the Content-Length describes the body a GET would have sent, so it stays
|
|
150
|
+
// (M4).
|
|
151
|
+
const bodylessStatus = isBodylessStatus(env.status);
|
|
152
|
+
const bodyless = bodylessStatus || requestMethod?.toUpperCase() === "HEAD";
|
|
153
|
+
|
|
154
|
+
const owned = bodylessStatus ? [...RESPONSE_OWNED, "content-length"] : RESPONSE_OWNED;
|
|
155
|
+
const carried = withoutHeaders(env.headers, owned);
|
|
156
|
+
const declared = parseContentLength(getAll(carried, "content-length"));
|
|
157
|
+
|
|
158
|
+
let head = `HTTP/1.1 ${env.status} ${reason}\r\n`;
|
|
159
|
+
head += encodeHeaderLines(carried);
|
|
160
|
+
head += "Connection: close\r\n";
|
|
161
|
+
if (!bodyless && body !== undefined && declared === undefined) {
|
|
162
|
+
head += "Transfer-Encoding: chunked\r\n";
|
|
163
|
+
}
|
|
164
|
+
head += "\r\n";
|
|
165
|
+
yield encodeLatin1(head);
|
|
166
|
+
|
|
167
|
+
if (bodyless) {
|
|
168
|
+
await discard(body);
|
|
169
|
+
return;
|
|
170
|
+
}
|
|
171
|
+
if (body === undefined) {
|
|
172
|
+
// Same truncation guard as encodeRequest — see I1. Not reached when
|
|
173
|
+
// bodyless: a response to HEAD legitimately keeps a declared
|
|
174
|
+
// Content-Length with no body (M4), and that is not a truncation.
|
|
175
|
+
if (declared !== undefined && declared !== 0) {
|
|
176
|
+
throw new HttpParseError(`body is 0 bytes but Content-Length declares ${declared}`);
|
|
177
|
+
}
|
|
178
|
+
return;
|
|
179
|
+
}
|
|
180
|
+
yield* emitBody(body, declared);
|
|
181
|
+
}
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Raised for any byte sequence this codec refuses to interpret. Every case is
|
|
3
|
+
* a refusal to guess: HTTP/1.1 parsers that guess are how request smuggling
|
|
4
|
+
* works.
|
|
5
|
+
*/
|
|
6
|
+
export class HttpParseError extends Error {
|
|
7
|
+
override readonly name = "HttpParseError";
|
|
8
|
+
}
|
|
@@ -0,0 +1,263 @@
|
|
|
1
|
+
import type { ByteReader } from "../bytes.js";
|
|
2
|
+
import { HttpParseError } from "./errors.js";
|
|
3
|
+
|
|
4
|
+
export type HeaderList = [string, string][];
|
|
5
|
+
|
|
6
|
+
const TCHAR = new Set(
|
|
7
|
+
"!#$%&'*+-.^_`|~0123456789abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ",
|
|
8
|
+
);
|
|
9
|
+
|
|
10
|
+
const SP = 0x20;
|
|
11
|
+
const HTAB = 0x09;
|
|
12
|
+
|
|
13
|
+
export function isTokenChar(byte: number): boolean {
|
|
14
|
+
return byte > SP && byte < 0x7f && TCHAR.has(String.fromCharCode(byte));
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
export function isToken(value: string): boolean {
|
|
18
|
+
if (value.length === 0) return false;
|
|
19
|
+
for (const ch of value) {
|
|
20
|
+
if (!TCHAR.has(ch)) return false;
|
|
21
|
+
}
|
|
22
|
+
return true;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* Header bytes are latin-1: byte-preserving, matching Node, and lossless
|
|
27
|
+
* across a decode/encode round-trip. Control characters are rejected
|
|
28
|
+
* separately by `assertValidHeaderValue`.
|
|
29
|
+
*/
|
|
30
|
+
export function decodeLatin1(bytes: Uint8Array): string {
|
|
31
|
+
let out = "";
|
|
32
|
+
for (const byte of bytes) out += String.fromCharCode(byte);
|
|
33
|
+
return out;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export function encodeLatin1(text: string): Uint8Array {
|
|
37
|
+
const out = new Uint8Array(text.length);
|
|
38
|
+
for (let i = 0; i < text.length; i++) {
|
|
39
|
+
const code = text.charCodeAt(i);
|
|
40
|
+
if (code > 0xff) {
|
|
41
|
+
throw new HttpParseError(`not latin-1 encodable: ${JSON.stringify(text)}`);
|
|
42
|
+
}
|
|
43
|
+
out[i] = code;
|
|
44
|
+
}
|
|
45
|
+
return out;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
// RFC 3986 `authority` grammar, restricted to `host [ ":" port ]` (no
|
|
49
|
+
// userinfo — RFC 9110 §7.2 excludes it from Host). Two shapes: an IP-literal
|
|
50
|
+
// in brackets, or a reg-name. Shared by the decoder (validating a Host header
|
|
51
|
+
// taken off the wire) and the encoder (validating the authority `splitTarget`
|
|
52
|
+
// derives from a caller-supplied url) so the grammar is defined exactly once.
|
|
53
|
+
const HOST_IP_LITERAL = /^\[[0-9A-Fa-f:.]+\](:\d{1,5})?$/;
|
|
54
|
+
const HOST_REG_NAME = /^[A-Za-z0-9._~-]+(:\d{1,5})?$/;
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Validates a value that is about to become (or was read as) a `Host`
|
|
58
|
+
* header: untrusted input either way, so an unrecognised shape is a refusal,
|
|
59
|
+
* not a guess. Rejects a present-but-empty value, one carrying userinfo
|
|
60
|
+
* (`evil.com@good.com`), and — the reason this also runs on encode — one
|
|
61
|
+
* smuggling a CRLF-terminated line into the authority of a caller-supplied
|
|
62
|
+
* url. Never applied to trusted configuration (a codec's `opts.host`).
|
|
63
|
+
*/
|
|
64
|
+
export function assertValidHost(host: string): void {
|
|
65
|
+
if (!HOST_IP_LITERAL.test(host) && !HOST_REG_NAME.test(host)) {
|
|
66
|
+
throw new HttpParseError(`invalid Host header: ${JSON.stringify(host)}`);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* A request-target must be entirely visible ASCII (RFC 9110 VCHAR) — nothing
|
|
72
|
+
* `<= 0x20` (space and every control character, CR/LF included) and nothing
|
|
73
|
+
* `>= 0x7F` (DEL and beyond). Applied on encode (`splitTarget`, so a
|
|
74
|
+
* caller-supplied url can never inject a second request line) and on decode
|
|
75
|
+
* (so a relay that decodes then re-encodes can never put an embedded control
|
|
76
|
+
* character back on the wire).
|
|
77
|
+
*/
|
|
78
|
+
export function assertValidTarget(target: string): void {
|
|
79
|
+
for (let i = 0; i < target.length; i++) {
|
|
80
|
+
const code = target.charCodeAt(i);
|
|
81
|
+
if (code <= SP || code >= 0x7f) {
|
|
82
|
+
throw new HttpParseError(`invalid request-target: ${JSON.stringify(target)}`);
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* Shared control-character check behind both `assertValidHeaderValue` and
|
|
89
|
+
* `assertValidStatusText` — the two were previously checked by different,
|
|
90
|
+
* looser rules (statusText only rejected CR/LF), which is exactly the
|
|
91
|
+
* asymmetry class C1 and M5 are both instances of. `label` is prepended
|
|
92
|
+
* verbatim to each message, so callers keep their own wording.
|
|
93
|
+
*/
|
|
94
|
+
function assertNoControlChars(label: string, value: string): void {
|
|
95
|
+
for (let i = 0; i < value.length; i++) {
|
|
96
|
+
const code = value.charCodeAt(i);
|
|
97
|
+
if (code === 0x0d || code === 0x0a) {
|
|
98
|
+
throw new HttpParseError(`${label} contains CR or LF`);
|
|
99
|
+
}
|
|
100
|
+
if ((code < SP && code !== HTAB) || code === 0x7f) {
|
|
101
|
+
throw new HttpParseError(`${label} contains a control character`);
|
|
102
|
+
}
|
|
103
|
+
if (code > 0xff) {
|
|
104
|
+
throw new HttpParseError(`${label} is not latin-1 encodable`);
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
export function assertValidHeaderValue(name: string, value: string): void {
|
|
110
|
+
assertNoControlChars(`header "${name}" value`, value);
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/** Same character class as a header value (M5) — CR/LF, every C0 control, and DEL. */
|
|
114
|
+
export function assertValidStatusText(value: string): void {
|
|
115
|
+
assertNoControlChars("statusText", value);
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
export function encodeHeaderLines(headers: HeaderList): string {
|
|
119
|
+
let out = "";
|
|
120
|
+
for (const [name, value] of headers) {
|
|
121
|
+
if (!isToken(name)) throw new HttpParseError(`invalid header name: ${JSON.stringify(name)}`);
|
|
122
|
+
assertValidHeaderValue(name, value);
|
|
123
|
+
out += `${name}: ${value}\r\n`;
|
|
124
|
+
}
|
|
125
|
+
return out;
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
/**
|
|
129
|
+
* Read to the blank line. `alreadyUsed` is the byte count of the start line,
|
|
130
|
+
* so the bound covers the whole head section rather than the headers alone.
|
|
131
|
+
*/
|
|
132
|
+
export async function readHeaderSection(
|
|
133
|
+
reader: ByteReader,
|
|
134
|
+
maxHeaderBytes: number,
|
|
135
|
+
alreadyUsed: number,
|
|
136
|
+
): Promise<HeaderList> {
|
|
137
|
+
const headers: HeaderList = [];
|
|
138
|
+
let used = alreadyUsed;
|
|
139
|
+
while (true) {
|
|
140
|
+
const remaining = maxHeaderBytes - used;
|
|
141
|
+
if (remaining <= 0) {
|
|
142
|
+
throw new HttpParseError(`head section exceeds ${maxHeaderBytes} bytes`);
|
|
143
|
+
}
|
|
144
|
+
const lineBytes = await reader.readLine(remaining);
|
|
145
|
+
used += lineBytes.byteLength + 2;
|
|
146
|
+
if (used > maxHeaderBytes) {
|
|
147
|
+
throw new HttpParseError(`head section exceeds ${maxHeaderBytes} bytes`);
|
|
148
|
+
}
|
|
149
|
+
if (lineBytes.byteLength === 0) return headers;
|
|
150
|
+
if (lineBytes[0] === SP || lineBytes[0] === HTAB) {
|
|
151
|
+
throw new HttpParseError("obs-fold header continuation is not accepted");
|
|
152
|
+
}
|
|
153
|
+
const line = decodeLatin1(lineBytes);
|
|
154
|
+
const colon = line.indexOf(":");
|
|
155
|
+
if (colon <= 0) throw new HttpParseError(`malformed header line: ${JSON.stringify(line)}`);
|
|
156
|
+
const name = line.slice(0, colon);
|
|
157
|
+
if (!isToken(name)) throw new HttpParseError(`invalid header name: ${JSON.stringify(name)}`);
|
|
158
|
+
const value = line
|
|
159
|
+
.slice(colon + 1)
|
|
160
|
+
.replace(/^[ \t]+/, "")
|
|
161
|
+
.replace(/[ \t]+$/, "");
|
|
162
|
+
assertValidHeaderValue(name, value);
|
|
163
|
+
headers.push([name, value]);
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
export function getAll(headers: HeaderList, name: string): string[] {
|
|
168
|
+
const lower = name.toLowerCase();
|
|
169
|
+
return headers.filter(([k]) => k.toLowerCase() === lower).map(([, v]) => v);
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
export function withoutHeaders(headers: HeaderList, names: string[]): HeaderList {
|
|
173
|
+
const drop = new Set(names.map((n) => n.toLowerCase()));
|
|
174
|
+
return headers.filter(([k]) => !drop.has(k.toLowerCase()));
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* Parse a `Content-Length` field into a single validated value.
|
|
179
|
+
*
|
|
180
|
+
* RFC 9110 §8.6 permits the value to be a comma-separated list of identical
|
|
181
|
+
* numbers (a relay may have appended one), so the list is folded; differing
|
|
182
|
+
* values are a refusal, not a choice. Shared by `resolveFraming` on decode and
|
|
183
|
+
* by the encoder's declared-length check, which previously carried its own
|
|
184
|
+
* copy that did NOT split on commas — so decode accepted `5, 5` while encode
|
|
185
|
+
* rejected it, and a relay that decoded then re-encoded threw.
|
|
186
|
+
*
|
|
187
|
+
* Returns undefined when the field is absent.
|
|
188
|
+
*/
|
|
189
|
+
export function parseContentLength(values: string[]): number | undefined {
|
|
190
|
+
if (values.length === 0) return undefined;
|
|
191
|
+
const unique = new Set(values.flatMap((v) => v.split(",").map((s) => s.trim())));
|
|
192
|
+
if (unique.size !== 1) {
|
|
193
|
+
throw new HttpParseError(`conflicting Content-Length values: ${[...unique].join(", ")}`);
|
|
194
|
+
}
|
|
195
|
+
// `unique.size === 1` is checked directly above, so this spread has exactly
|
|
196
|
+
// one element. The `!` is what lets a consumer compiling with
|
|
197
|
+
// `noUncheckedIndexedAccess` build against this source: without it `raw` is
|
|
198
|
+
// `string | undefined` and the `.test(raw)` below is a type error in THEIR
|
|
199
|
+
// build, not in this package's own (which does not enable the flag, so it
|
|
200
|
+
// cannot catch the regression). It was lost once already, in a rebase that
|
|
201
|
+
// resolved toward a refactor of this function.
|
|
202
|
+
const raw = [...unique][0]!;
|
|
203
|
+
if (!/^\d{1,15}$/.test(raw)) {
|
|
204
|
+
throw new HttpParseError(`invalid Content-Length: ${JSON.stringify(raw)}`);
|
|
205
|
+
}
|
|
206
|
+
return Number(raw);
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
/**
|
|
210
|
+
* Statuses that carry no body whatever the headers say (RFC 9110 §8.6): 1xx,
|
|
211
|
+
* 204 and 304. Defined once because the encoder must not frame a body for
|
|
212
|
+
* them and the decoder must not try to read one — two lists that agreed today
|
|
213
|
+
* and were free to drift tomorrow.
|
|
214
|
+
*
|
|
215
|
+
* A response to HEAD is also bodyless, but that depends on the request rather
|
|
216
|
+
* than the status, so callers test it separately.
|
|
217
|
+
*/
|
|
218
|
+
export function isBodylessStatus(status: number): boolean {
|
|
219
|
+
return status < 200 || status === 204 || status === 304;
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
export type BodyFraming =
|
|
223
|
+
| { kind: "chunked" }
|
|
224
|
+
| { kind: "length"; length: number }
|
|
225
|
+
| { kind: "none" };
|
|
226
|
+
|
|
227
|
+
/**
|
|
228
|
+
* RFC 9112 §6.3, with every ambiguity turned into a refusal. In particular a
|
|
229
|
+
* message declaring both Content-Length and Transfer-Encoding is rejected
|
|
230
|
+
* rather than resolved — disagreeing on which one wins is request smuggling.
|
|
231
|
+
*/
|
|
232
|
+
export function resolveFraming(headers: HeaderList, version: string): BodyFraming {
|
|
233
|
+
const te = getAll(headers, "transfer-encoding");
|
|
234
|
+
const cl = getAll(headers, "content-length");
|
|
235
|
+
|
|
236
|
+
if (te.length > 0 && cl.length > 0) {
|
|
237
|
+
throw new HttpParseError(
|
|
238
|
+
"message declares both Content-Length and Transfer-Encoding; refusing (request smuggling)",
|
|
239
|
+
);
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
if (te.length > 0) {
|
|
243
|
+
const encodings = te
|
|
244
|
+
.join(",")
|
|
245
|
+
.split(",")
|
|
246
|
+
.map((s) => s.trim().toLowerCase())
|
|
247
|
+
.filter((s) => s !== "");
|
|
248
|
+
if (encodings.length !== 1 || encodings[0] !== "chunked") {
|
|
249
|
+
throw new HttpParseError(`unsupported Transfer-Encoding: ${JSON.stringify(te.join(", "))}`);
|
|
250
|
+
}
|
|
251
|
+
if (version === "HTTP/1.0") {
|
|
252
|
+
throw new HttpParseError("Transfer-Encoding is not valid in HTTP/1.0");
|
|
253
|
+
}
|
|
254
|
+
return { kind: "chunked" };
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
const length = parseContentLength(cl);
|
|
258
|
+
if (length !== undefined) {
|
|
259
|
+
return { kind: "length", length };
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
return { kind: "none" };
|
|
263
|
+
}
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import type { MessageCodec } from "../message.js";
|
|
2
|
+
import { decodeRequest, decodeResponse } from "./decode.js";
|
|
3
|
+
import { encodeRequest, encodeResponse, type ResolvedHttpCodecOptions } from "./encode.js";
|
|
4
|
+
import { isTokenChar } from "./headers.js";
|
|
5
|
+
|
|
6
|
+
export type HttpCodecOptions = {
|
|
7
|
+
/**
|
|
8
|
+
* Scheme used to rebuild an absolute url on decode. HTTP/1.1 origin-form
|
|
9
|
+
* carries no scheme, so this is configuration, not wire data — a peer
|
|
10
|
+
* configured `http` rebuilds an `https` url as `http`. Accepted cost of
|
|
11
|
+
* decision 6; see ADR-0006.
|
|
12
|
+
*/
|
|
13
|
+
scheme?: "http" | "https";
|
|
14
|
+
/** Authority used when a url carries none. Also fills the mandatory Host header. */
|
|
15
|
+
host?: string;
|
|
16
|
+
/** Bound on the whole head section, start line included. Default 65536. */
|
|
17
|
+
maxHeaderBytes?: number;
|
|
18
|
+
};
|
|
19
|
+
|
|
20
|
+
export function newHttpCodec(options: HttpCodecOptions = {}): MessageCodec {
|
|
21
|
+
const opts: ResolvedHttpCodecOptions = {
|
|
22
|
+
scheme: options.scheme ?? "http",
|
|
23
|
+
host: options.host ?? "localhost",
|
|
24
|
+
maxHeaderBytes: options.maxHeaderBytes ?? 65536,
|
|
25
|
+
};
|
|
26
|
+
return {
|
|
27
|
+
name: "http/1.1",
|
|
28
|
+
// A request starts with a method token, a response with "HTTP/1.1" — both
|
|
29
|
+
// begin with a token character. "{" is not a token character, so this is
|
|
30
|
+
// disjoint from jsonEnvelopeCodec by construction, not by convention.
|
|
31
|
+
sniff: (byte: number): boolean => isTokenChar(byte),
|
|
32
|
+
encodeRequest: (env, body) => encodeRequest(env, body, opts),
|
|
33
|
+
encodeResponse: (env, body, o) => encodeResponse(env, body, opts, o?.method),
|
|
34
|
+
decodeRequest: (input) => decodeRequest(input, opts),
|
|
35
|
+
decodeResponse: (input, o) => decodeResponse(input, opts, o.method),
|
|
36
|
+
};
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export const httpCodec: MessageCodec = newHttpCodec();
|
|
40
|
+
export { HttpParseError } from "./errors.js";
|
package/src/index.ts
CHANGED
|
@@ -1,7 +1,9 @@
|
|
|
1
|
+
export { defaultCodec } from "./codec-default.js";
|
|
1
2
|
export { DuplexSiteBuilder, type SiteHandler } from "./duplex-site-builder.js";
|
|
2
3
|
export {
|
|
3
4
|
decodeMessage,
|
|
4
5
|
encodeMessage,
|
|
6
|
+
jsonEnvelopeCodec,
|
|
5
7
|
type RequestEnvelope,
|
|
6
8
|
type ResponseEnvelope,
|
|
7
9
|
} from "./envelope.js";
|
|
@@ -9,9 +11,11 @@ export { fetchOverDuplex, serveFetchOverDuplex } from "./fetch.js";
|
|
|
9
11
|
export {
|
|
10
12
|
type HttpDataHandler,
|
|
11
13
|
type HttpDataHandlerResult,
|
|
14
|
+
type HttpDataOptions,
|
|
12
15
|
type HttpFetchResult,
|
|
13
16
|
httpFetch,
|
|
14
17
|
httpServe,
|
|
18
|
+
PEER_ERROR_HEADER,
|
|
15
19
|
} from "./http-data.js";
|
|
16
20
|
export { HttpError, type HttpErrorOptions } from "./http-error.js";
|
|
17
21
|
export {
|
|
@@ -22,3 +26,17 @@ export {
|
|
|
22
26
|
type SerializedHttpRequest,
|
|
23
27
|
type SerializedHttpResponse,
|
|
24
28
|
} from "./http-stubs.js";
|
|
29
|
+
export {
|
|
30
|
+
type HttpCodecOptions,
|
|
31
|
+
HttpParseError,
|
|
32
|
+
httpCodec,
|
|
33
|
+
newHttpCodec,
|
|
34
|
+
} from "./http1/index.js";
|
|
35
|
+
export type {
|
|
36
|
+
ByteSource,
|
|
37
|
+
DecodedRequest,
|
|
38
|
+
DecodedResponse,
|
|
39
|
+
MessageCodec,
|
|
40
|
+
ResponseCodecOptions,
|
|
41
|
+
} from "./message.js";
|
|
42
|
+
export { newSniffingCodec, type SniffingCodecOptions } from "./sniff.js";
|
package/src/message.ts
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
/** Anything the codecs will read bytes from. */
|
|
2
|
+
export type ByteSource = AsyncIterable<Uint8Array> | Iterable<Uint8Array>;
|
|
3
|
+
|
|
4
|
+
export type RequestEnvelope = {
|
|
5
|
+
url: string;
|
|
6
|
+
method: string;
|
|
7
|
+
headers: [string, string][];
|
|
8
|
+
};
|
|
9
|
+
|
|
10
|
+
export type ResponseEnvelope = {
|
|
11
|
+
status: number;
|
|
12
|
+
statusText: string;
|
|
13
|
+
headers: [string, string][];
|
|
14
|
+
};
|
|
15
|
+
|
|
16
|
+
export type DecodedRequest = {
|
|
17
|
+
envelope: RequestEnvelope;
|
|
18
|
+
body: AsyncIterable<Uint8Array>;
|
|
19
|
+
/** The codec that actually read this message; set by the sniffing codec. */
|
|
20
|
+
codec?: MessageCodec;
|
|
21
|
+
};
|
|
22
|
+
|
|
23
|
+
export type DecodedResponse = {
|
|
24
|
+
envelope: ResponseEnvelope;
|
|
25
|
+
body: AsyncIterable<Uint8Array>;
|
|
26
|
+
};
|
|
27
|
+
|
|
28
|
+
/** Extra context a codec may need that the envelope does not carry. */
|
|
29
|
+
export type ResponseCodecOptions = {
|
|
30
|
+
/**
|
|
31
|
+
* Method of the request this response answers. Required by HTTP/1.1: a
|
|
32
|
+
* response to HEAD carries framing headers but no body, and no parser can
|
|
33
|
+
* know that from the response bytes alone.
|
|
34
|
+
*/
|
|
35
|
+
method: string;
|
|
36
|
+
};
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* One wire format. Requests and responses are separate operations because
|
|
40
|
+
* HTTP/1.1 serialises them differently — a codec cannot be generic over the
|
|
41
|
+
* envelope the way the JSON format was.
|
|
42
|
+
*/
|
|
43
|
+
export interface MessageCodec {
|
|
44
|
+
readonly name: string;
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* True if a message in this format may begin with `byte`. Used by the
|
|
48
|
+
* sniffing codec to dispatch without a negotiation handshake. Implementations
|
|
49
|
+
* must be mutually exclusive with every other codec they are paired with.
|
|
50
|
+
*/
|
|
51
|
+
sniff(byte: number): boolean;
|
|
52
|
+
|
|
53
|
+
encodeRequest(env: RequestEnvelope, body?: ByteSource): AsyncGenerator<Uint8Array>;
|
|
54
|
+
encodeResponse(
|
|
55
|
+
env: ResponseEnvelope,
|
|
56
|
+
body?: ByteSource,
|
|
57
|
+
options?: ResponseCodecOptions,
|
|
58
|
+
): AsyncGenerator<Uint8Array>;
|
|
59
|
+
|
|
60
|
+
decodeRequest(input: ByteSource): Promise<DecodedRequest>;
|
|
61
|
+
decodeResponse(input: ByteSource, options: ResponseCodecOptions): Promise<DecodedResponse>;
|
|
62
|
+
}
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The one place that knows a runtime may not implement request body streams,
|
|
3
|
+
* and the buffering both fallbacks need. `fetch.ts` and `http-stubs.ts` each
|
|
4
|
+
* carry a client and a server direction that must agree on the answer, so the
|
|
5
|
+
* predicate lives here rather than being written out four times.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Whether `Request.prototype` exposes a `body` accessor. Two call sites read
|
|
10
|
+
* that property, and two more depend on the constructor accepting a
|
|
11
|
+
* `ReadableStream` as `init.body`; this predicate answers for all four.
|
|
12
|
+
*
|
|
13
|
+
* What was verified, and all that is claimed here: Firefox (146 at the time of
|
|
14
|
+
* writing) has *neither*, and Chromium and Node have *both*. On Firefox it is
|
|
15
|
+
* not that `body` is `undefined` on the instance —
|
|
16
|
+
* `Object.getOwnPropertyDescriptor(Request.prototype, "body")` is `null`, the
|
|
17
|
+
* accessor is genuinely absent — and because a `ReadableStream` is then not a
|
|
18
|
+
* recognised `BodyInit`, the constructor falls through to the string branch and
|
|
19
|
+
* stores the literal text `[object ReadableStream]`.
|
|
20
|
+
*
|
|
21
|
+
* The two halves are NOT guaranteed to ship together, so do not read this as a
|
|
22
|
+
* test for "request streams" in general. Safari is the counterexample: it has
|
|
23
|
+
* had `Request.body` since 11.1 but only accepts a stream as `init.body` from
|
|
24
|
+
* Technology Preview 250, so a shipping Safari has the reader half without the
|
|
25
|
+
* upload half and this returns `true` there. That looks benign — WebKit appears
|
|
26
|
+
* to store the stream on the `Request` rather than stringify it, and its error
|
|
27
|
+
* comes from `fetch()`, which this package never calls on the objects it builds
|
|
28
|
+
* — but it is untested, and it is the case to look at first if a Safari report
|
|
29
|
+
* arrives. The sharper probe, if one is ever needed, is whether
|
|
30
|
+
* `new Request(url, {method:"POST", body:new ReadableStream(), duplex:"half"})`
|
|
31
|
+
* has a `content-type` of `text/plain;charset=UTF-8` (stringified) or `null`
|
|
32
|
+
* (stored); it is not used here because it costs a `Request` and a
|
|
33
|
+
* `ReadableStream` per call and agrees with the descriptor check on every
|
|
34
|
+
* runtime measured.
|
|
35
|
+
*
|
|
36
|
+
* A capability check, never a user-agent test: the question is what this
|
|
37
|
+
* runtime does, and the answer flips on its own the day Firefox ships request
|
|
38
|
+
* streams. Evaluated per call rather than cached at module load so that a test
|
|
39
|
+
* can install a `Request` without the capability and exercise the real branch
|
|
40
|
+
* under Node.
|
|
41
|
+
*/
|
|
42
|
+
export function supportsRequestStreams(): boolean {
|
|
43
|
+
return (
|
|
44
|
+
typeof Request === "function" &&
|
|
45
|
+
Object.getOwnPropertyDescriptor(Request.prototype, "body") != null
|
|
46
|
+
);
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Drain a body iterable into one contiguous buffer. Only the fallbacks need it.
|
|
51
|
+
* Allocates its own buffer rather than reusing `bytes.ts`'s `concatChunks`,
|
|
52
|
+
* which passes a single chunk straight through: that chunk is a view onto the
|
|
53
|
+
* decoder's own read buffer, and the result here is handed to `new Request` and
|
|
54
|
+
* outlives the decode. The allocation is also what makes it a
|
|
55
|
+
* `Uint8Array<ArrayBuffer>`, which `BodyInit` accepts and the looser
|
|
56
|
+
* `Uint8Array<ArrayBufferLike>` does not.
|
|
57
|
+
*/
|
|
58
|
+
export async function collectBytes(
|
|
59
|
+
source: AsyncIterable<Uint8Array>,
|
|
60
|
+
): Promise<Uint8Array<ArrayBuffer>> {
|
|
61
|
+
const parts: Uint8Array[] = [];
|
|
62
|
+
let total = 0;
|
|
63
|
+
for await (const chunk of source) {
|
|
64
|
+
if (chunk.byteLength === 0) continue;
|
|
65
|
+
parts.push(chunk);
|
|
66
|
+
total += chunk.byteLength;
|
|
67
|
+
}
|
|
68
|
+
const out = new Uint8Array(total);
|
|
69
|
+
let offset = 0;
|
|
70
|
+
for (const part of parts) {
|
|
71
|
+
out.set(part, offset);
|
|
72
|
+
offset += part.byteLength;
|
|
73
|
+
}
|
|
74
|
+
return out;
|
|
75
|
+
}
|