@statewalker/webrun-http-streams 0.1.1 → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +361 -13
- package/dist/bytes.d.ts +43 -0
- package/dist/bytes.d.ts.map +1 -0
- package/dist/codec-default.d.ts +7 -0
- package/dist/codec-default.d.ts.map +1 -0
- package/dist/duplex-site-builder.d.ts +4 -1
- package/dist/duplex-site-builder.d.ts.map +1 -1
- package/dist/envelope.d.ts +10 -10
- package/dist/envelope.d.ts.map +1 -1
- package/dist/fetch.d.ts +3 -2
- package/dist/fetch.d.ts.map +1 -1
- package/dist/http-data.d.ts +11 -7
- package/dist/http-data.d.ts.map +1 -1
- package/dist/http-error.d.ts.map +1 -1
- package/dist/http-stubs.d.ts +12 -0
- package/dist/http-stubs.d.ts.map +1 -1
- package/dist/http1/chunked.d.ts +18 -0
- package/dist/http1/chunked.d.ts.map +1 -0
- package/dist/http1/decode.d.ts +5 -0
- package/dist/http1/decode.d.ts.map +1 -0
- package/dist/http1/encode.d.ts +20 -0
- package/dist/http1/encode.d.ts.map +1 -0
- package/dist/http1/errors.d.ts +9 -0
- package/dist/http1/errors.d.ts.map +1 -0
- package/dist/http1/headers.d.ts +78 -0
- package/dist/http1/headers.d.ts.map +1 -0
- package/dist/http1/index.d.ts +18 -0
- package/dist/http1/index.d.ts.map +1 -0
- package/dist/index.d.ts +6 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1040 -138
- package/dist/message.d.ts +50 -0
- package/dist/message.d.ts.map +1 -0
- package/dist/request-streams.d.ts +52 -0
- package/dist/request-streams.d.ts.map +1 -0
- package/dist/sniff.d.ts +18 -0
- package/dist/sniff.d.ts.map +1 -0
- package/package.json +13 -7
- package/src/bytes.ts +157 -0
- package/src/codec-default.ts +13 -0
- package/src/duplex-site-builder.ts +10 -2
- package/src/envelope.ts +43 -34
- package/src/fetch.ts +130 -12
- package/src/http-data.ts +160 -17
- package/src/http-stubs.ts +72 -18
- package/src/http1/chunked.ts +89 -0
- package/src/http1/decode.ts +208 -0
- package/src/http1/encode.ts +181 -0
- package/src/http1/errors.ts +8 -0
- package/src/http1/headers.ts +263 -0
- package/src/http1/index.ts +40 -0
- package/src/index.ts +18 -0
- package/src/message.ts +62 -0
- package/src/request-streams.ts +75 -0
- package/src/sniff.ts +68 -0
- package/LICENSE +0 -21
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
import { ByteReader, ByteStreamError } from "../bytes.js";
|
|
2
|
+
import type { ByteSource, DecodedRequest, DecodedResponse } from "../message.js";
|
|
3
|
+
import { decodeChunked } from "./chunked.js";
|
|
4
|
+
import type { ResolvedHttpCodecOptions } from "./encode.js";
|
|
5
|
+
import { HttpParseError } from "./errors.js";
|
|
6
|
+
import {
|
|
7
|
+
assertValidHost,
|
|
8
|
+
assertValidTarget,
|
|
9
|
+
type BodyFraming,
|
|
10
|
+
decodeLatin1,
|
|
11
|
+
getAll,
|
|
12
|
+
isBodylessStatus,
|
|
13
|
+
isToken,
|
|
14
|
+
readHeaderSection,
|
|
15
|
+
resolveFraming,
|
|
16
|
+
} from "./headers.js";
|
|
17
|
+
|
|
18
|
+
const VERSIONS = new Set(["HTTP/1.1", "HTTP/1.0"]);
|
|
19
|
+
const ABSOLUTE_FORM = /^[a-zA-Z][a-zA-Z0-9+.-]*:\/\//;
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* One message per Duplex call (ADR-0006), so bytes after a complete message
|
|
23
|
+
* are an error. Checked against what is ALREADY BUFFERED rather than by
|
|
24
|
+
* awaiting end-of-stream: a live socket from a keep-alive peer never reaches
|
|
25
|
+
* EOF, so awaiting one would hang instead of failing.
|
|
26
|
+
*/
|
|
27
|
+
function assertNoBufferedBytes(reader: ByteReader): void {
|
|
28
|
+
const extra = reader.bufferedLength();
|
|
29
|
+
if (extra > 0) {
|
|
30
|
+
throw new HttpParseError(`${extra} trailing bytes after a complete message`);
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* The public error contract (I2): everything leaving `decodeRequest` /
|
|
36
|
+
* `decodeResponse` is an `HttpParseError`, whether the refusal happened
|
|
37
|
+
* synchronously (a malformed start line or header) or lazily while the body
|
|
38
|
+
* is later drained. `ByteStreamError` — the `ByteReader`'s own class, never
|
|
39
|
+
* exported — is the one thing converted here, message and all preserved via
|
|
40
|
+
* `cause`. Anything else is a genuine failure of the underlying source (a
|
|
41
|
+
* dropped socket, say) and must reach the caller unchanged: blanket-catching
|
|
42
|
+
* would hide that distinction.
|
|
43
|
+
*/
|
|
44
|
+
function toHttpParseError(err: unknown): never {
|
|
45
|
+
if (err instanceof ByteStreamError) {
|
|
46
|
+
throw new HttpParseError(err.message, { cause: err });
|
|
47
|
+
}
|
|
48
|
+
throw err;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** Applies `toHttpParseError` across the whole lifetime of a body generator. */
|
|
52
|
+
async function* convertBodyErrors(source: AsyncGenerator<Uint8Array>): AsyncGenerator<Uint8Array> {
|
|
53
|
+
try {
|
|
54
|
+
yield* source;
|
|
55
|
+
} catch (err) {
|
|
56
|
+
toHttpParseError(err);
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
async function* readBody(
|
|
61
|
+
reader: ByteReader,
|
|
62
|
+
framing: BodyFraming,
|
|
63
|
+
opts: ResolvedHttpCodecOptions,
|
|
64
|
+
noneMeansEof: boolean,
|
|
65
|
+
): AsyncGenerator<Uint8Array> {
|
|
66
|
+
if (framing.kind === "chunked") {
|
|
67
|
+
yield* decodeChunked(reader, opts.maxHeaderBytes);
|
|
68
|
+
assertNoBufferedBytes(reader);
|
|
69
|
+
return;
|
|
70
|
+
}
|
|
71
|
+
if (framing.kind === "length") {
|
|
72
|
+
let remaining = framing.length;
|
|
73
|
+
while (remaining > 0) {
|
|
74
|
+
const part = await reader.readSome(remaining);
|
|
75
|
+
if (part === undefined) {
|
|
76
|
+
throw new HttpParseError(`body truncated: ${remaining} of ${framing.length} bytes missing`);
|
|
77
|
+
}
|
|
78
|
+
remaining -= part.byteLength;
|
|
79
|
+
yield part;
|
|
80
|
+
}
|
|
81
|
+
assertNoBufferedBytes(reader);
|
|
82
|
+
return;
|
|
83
|
+
}
|
|
84
|
+
if (noneMeansEof) {
|
|
85
|
+
yield* reader.rest();
|
|
86
|
+
return;
|
|
87
|
+
}
|
|
88
|
+
assertNoBufferedBytes(reader);
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* `readLine`'s bound is best-effort — it is skipped when the line is already
|
|
93
|
+
* buffered — so a very long start line can reach these messages intact. Since
|
|
94
|
+
* a refusal is now echoed back to the sender in a 400 body, quote only enough
|
|
95
|
+
* to diagnose rather than reflecting the whole thing.
|
|
96
|
+
*/
|
|
97
|
+
function quoteLine(line: string): string {
|
|
98
|
+
const MAX = 120;
|
|
99
|
+
return line.length <= MAX
|
|
100
|
+
? JSON.stringify(line)
|
|
101
|
+
: `${JSON.stringify(line.slice(0, MAX))} (truncated from ${line.length} chars)`;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
export async function decodeRequest(
|
|
105
|
+
input: ByteSource,
|
|
106
|
+
opts: ResolvedHttpCodecOptions,
|
|
107
|
+
): Promise<DecodedRequest> {
|
|
108
|
+
const reader = new ByteReader(input);
|
|
109
|
+
try {
|
|
110
|
+
const startBytes = await reader.readLine(opts.maxHeaderBytes);
|
|
111
|
+
const startLine = decodeLatin1(startBytes);
|
|
112
|
+
const parts = startLine.split(" ");
|
|
113
|
+
if (parts.length !== 3) {
|
|
114
|
+
throw new HttpParseError(`malformed request line: ${quoteLine(startLine)}`);
|
|
115
|
+
}
|
|
116
|
+
// `parts.length === 3` is checked above, so all three elements are present.
|
|
117
|
+
const [method, target, version] = parts as [string, string, string];
|
|
118
|
+
if (!isToken(method)) throw new HttpParseError(`invalid method: ${JSON.stringify(method)}`);
|
|
119
|
+
if (!VERSIONS.has(version)) {
|
|
120
|
+
throw new HttpParseError(`unsupported HTTP version: ${JSON.stringify(version)}`);
|
|
121
|
+
}
|
|
122
|
+
if (target === "*") {
|
|
123
|
+
throw new HttpParseError("asterisk-form request target is not supported");
|
|
124
|
+
}
|
|
125
|
+
// A relay that decodes then re-encodes must never be able to put a
|
|
126
|
+
// control character (a raw CR chief among them) back onto a request
|
|
127
|
+
// line — see C1.
|
|
128
|
+
assertValidTarget(target);
|
|
129
|
+
|
|
130
|
+
const headers = await readHeaderSection(reader, opts.maxHeaderBytes, startBytes.byteLength + 2);
|
|
131
|
+
|
|
132
|
+
const hosts = getAll(headers, "host");
|
|
133
|
+
if (hosts.length > 1) throw new HttpParseError("multiple Host headers");
|
|
134
|
+
const host = hosts[0];
|
|
135
|
+
|
|
136
|
+
let url: string;
|
|
137
|
+
if (target.startsWith("/")) {
|
|
138
|
+
if (version === "HTTP/1.1" && host === undefined) {
|
|
139
|
+
throw new HttpParseError("HTTP/1.1 request has no Host header");
|
|
140
|
+
}
|
|
141
|
+
if (host !== undefined) assertValidHost(host);
|
|
142
|
+
// The scheme is not on the wire in origin-form; it comes from config.
|
|
143
|
+
url = `${opts.scheme}://${host ?? opts.host}${target}`;
|
|
144
|
+
} else if (ABSOLUTE_FORM.test(target)) {
|
|
145
|
+
url = target;
|
|
146
|
+
} else {
|
|
147
|
+
throw new HttpParseError(`unsupported request target: ${JSON.stringify(target)}`);
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
// `Host` grammar alone is not sufficient proof that the assembled url is
|
|
151
|
+
// sane — this is a fail-safe belt-and-braces check, not a
|
|
152
|
+
// re-serialisation: the parsed `URL` is discarded, and `url` itself is
|
|
153
|
+
// what reaches the envelope untouched.
|
|
154
|
+
try {
|
|
155
|
+
new URL(url);
|
|
156
|
+
} catch {
|
|
157
|
+
throw new HttpParseError(`decoded url is not a valid URL: ${JSON.stringify(url)}`);
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
return {
|
|
161
|
+
envelope: { url, method, headers },
|
|
162
|
+
body: convertBodyErrors(readBody(reader, resolveFraming(headers, version), opts, false)),
|
|
163
|
+
};
|
|
164
|
+
} catch (err) {
|
|
165
|
+
toHttpParseError(err);
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
export async function decodeResponse(
|
|
170
|
+
input: ByteSource,
|
|
171
|
+
opts: ResolvedHttpCodecOptions,
|
|
172
|
+
method: string,
|
|
173
|
+
): Promise<DecodedResponse> {
|
|
174
|
+
const reader = new ByteReader(input);
|
|
175
|
+
try {
|
|
176
|
+
const startBytes = await reader.readLine(opts.maxHeaderBytes);
|
|
177
|
+
const startLine = decodeLatin1(startBytes);
|
|
178
|
+
|
|
179
|
+
const firstSp = startLine.indexOf(" ");
|
|
180
|
+
if (firstSp === -1) {
|
|
181
|
+
throw new HttpParseError(`malformed status line: ${quoteLine(startLine)}`);
|
|
182
|
+
}
|
|
183
|
+
const version = startLine.slice(0, firstSp);
|
|
184
|
+
if (!VERSIONS.has(version)) {
|
|
185
|
+
throw new HttpParseError(`unsupported HTTP version: ${JSON.stringify(version)}`);
|
|
186
|
+
}
|
|
187
|
+
const afterVersion = startLine.slice(firstSp + 1);
|
|
188
|
+
const secondSp = afterVersion.indexOf(" ");
|
|
189
|
+
const codeText = secondSp === -1 ? afterVersion : afterVersion.slice(0, secondSp);
|
|
190
|
+
if (!/^\d{3}$/.test(codeText)) {
|
|
191
|
+
throw new HttpParseError(`invalid status code: ${JSON.stringify(codeText)}`);
|
|
192
|
+
}
|
|
193
|
+
const status = Number(codeText);
|
|
194
|
+
const statusText = secondSp === -1 ? "" : afterVersion.slice(secondSp + 1);
|
|
195
|
+
|
|
196
|
+
const headers = await readHeaderSection(reader, opts.maxHeaderBytes, startBytes.byteLength + 2);
|
|
197
|
+
|
|
198
|
+
const bodyless = isBodylessStatus(status) || method.toUpperCase() === "HEAD";
|
|
199
|
+
const framing: BodyFraming = bodyless ? { kind: "none" } : resolveFraming(headers, version);
|
|
200
|
+
|
|
201
|
+
return {
|
|
202
|
+
envelope: { status, statusText, headers },
|
|
203
|
+
body: convertBodyErrors(readBody(reader, framing, opts, !bodyless)),
|
|
204
|
+
};
|
|
205
|
+
} catch (err) {
|
|
206
|
+
toHttpParseError(err);
|
|
207
|
+
}
|
|
208
|
+
}
|
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
import { discard } from "../bytes.js";
|
|
2
|
+
import type { ByteSource, RequestEnvelope, ResponseEnvelope } from "../message.js";
|
|
3
|
+
import { encodeChunked } from "./chunked.js";
|
|
4
|
+
import { HttpParseError } from "./errors.js";
|
|
5
|
+
import {
|
|
6
|
+
assertValidHost,
|
|
7
|
+
assertValidStatusText,
|
|
8
|
+
assertValidTarget,
|
|
9
|
+
encodeHeaderLines,
|
|
10
|
+
encodeLatin1,
|
|
11
|
+
getAll,
|
|
12
|
+
isBodylessStatus,
|
|
13
|
+
isToken,
|
|
14
|
+
parseContentLength,
|
|
15
|
+
withoutHeaders,
|
|
16
|
+
} from "./headers.js";
|
|
17
|
+
|
|
18
|
+
export type ResolvedHttpCodecOptions = {
|
|
19
|
+
scheme: "http" | "https";
|
|
20
|
+
host: string;
|
|
21
|
+
maxHeaderBytes: number;
|
|
22
|
+
};
|
|
23
|
+
|
|
24
|
+
/** Headers the codec owns: a caller-supplied copy is dropped and re-derived. */
|
|
25
|
+
const REQUEST_OWNED = ["host", "connection", "transfer-encoding"];
|
|
26
|
+
const RESPONSE_OWNED = ["connection", "transfer-encoding"];
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* Split a URL into an origin-form request target and an authority *without*
|
|
30
|
+
* going through `new URL()`. `URL` normalises percent-encoding and would
|
|
31
|
+
* re-serialise the target — which is the exact class of defect note 15 found
|
|
32
|
+
* in @libp2p/http, where rebuilding the request line silently dropped
|
|
33
|
+
* `url.search`. Here the target is a verbatim slice of the caller's string.
|
|
34
|
+
*/
|
|
35
|
+
export function splitTarget(
|
|
36
|
+
url: string,
|
|
37
|
+
opts: ResolvedHttpCodecOptions,
|
|
38
|
+
): { target: string; authority: string } {
|
|
39
|
+
if (url.startsWith("/")) {
|
|
40
|
+
// Neither field is trusted just because it came from a caller's own url:
|
|
41
|
+
// an unvalidated CRLF here would splice a second request line (or an
|
|
42
|
+
// extra header) onto the wire — see C1.
|
|
43
|
+
assertValidTarget(url);
|
|
44
|
+
return { target: url, authority: opts.host };
|
|
45
|
+
}
|
|
46
|
+
const match = /^[a-zA-Z][a-zA-Z0-9+.-]*:\/\/([^/?#]*)([^#]*)/.exec(url);
|
|
47
|
+
if (!match) {
|
|
48
|
+
throw new HttpParseError(`cannot derive a request target from url: ${JSON.stringify(url)}`);
|
|
49
|
+
}
|
|
50
|
+
// Both capture groups are mandatory in the regex above (neither is
|
|
51
|
+
// `?`-quantified), so a successful match always populates them — an empty
|
|
52
|
+
// string is possible, `undefined` is not.
|
|
53
|
+
const rawAuthority = match[1]!;
|
|
54
|
+
if (rawAuthority === "") {
|
|
55
|
+
throw new HttpParseError(`url has no authority: ${JSON.stringify(url)}`);
|
|
56
|
+
}
|
|
57
|
+
// Host is `uri-host [ ":" port ]` — no userinfo (RFC 9110 §7.2). Strip any
|
|
58
|
+
// `user:pass@` prefix so credentials never end up in a header proxies log.
|
|
59
|
+
const at = rawAuthority.lastIndexOf("@");
|
|
60
|
+
const authority = at === -1 ? rawAuthority : rawAuthority.slice(at + 1);
|
|
61
|
+
assertValidHost(authority);
|
|
62
|
+
|
|
63
|
+
const rawTarget = match[2]!;
|
|
64
|
+
const target = rawTarget === "" ? "/" : rawTarget.startsWith("/") ? rawTarget : `/${rawTarget}`;
|
|
65
|
+
assertValidTarget(target);
|
|
66
|
+
return { target, authority };
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
async function* emitBody(
|
|
70
|
+
body: ByteSource,
|
|
71
|
+
declared: number | undefined,
|
|
72
|
+
): AsyncGenerator<Uint8Array> {
|
|
73
|
+
if (declared === undefined) {
|
|
74
|
+
yield* encodeChunked(body);
|
|
75
|
+
return;
|
|
76
|
+
}
|
|
77
|
+
let sent = 0;
|
|
78
|
+
for await (const chunk of body) {
|
|
79
|
+
if (chunk.byteLength === 0) continue;
|
|
80
|
+
sent += chunk.byteLength;
|
|
81
|
+
if (sent > declared) {
|
|
82
|
+
throw new HttpParseError(
|
|
83
|
+
`body exceeds declared Content-Length ${declared} (${sent} bytes so far)`,
|
|
84
|
+
);
|
|
85
|
+
}
|
|
86
|
+
yield chunk;
|
|
87
|
+
}
|
|
88
|
+
if (sent !== declared) {
|
|
89
|
+
throw new HttpParseError(`body is ${sent} bytes but Content-Length declares ${declared}`);
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
export async function* encodeRequest(
|
|
94
|
+
env: RequestEnvelope,
|
|
95
|
+
body: ByteSource | undefined,
|
|
96
|
+
opts: ResolvedHttpCodecOptions,
|
|
97
|
+
): AsyncGenerator<Uint8Array> {
|
|
98
|
+
if (!isToken(env.method)) {
|
|
99
|
+
throw new HttpParseError(`invalid method: ${JSON.stringify(env.method)}`);
|
|
100
|
+
}
|
|
101
|
+
const { target, authority } = splitTarget(env.url, opts);
|
|
102
|
+
if (authority === "") {
|
|
103
|
+
throw new HttpParseError("no Host available: url has no authority and no host is configured");
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
const carried = withoutHeaders(env.headers, REQUEST_OWNED);
|
|
107
|
+
const declared = parseContentLength(getAll(carried, "content-length"));
|
|
108
|
+
|
|
109
|
+
let head = `${env.method} ${target} HTTP/1.1\r\n`;
|
|
110
|
+
head += `Host: ${authority}\r\n`;
|
|
111
|
+
head += encodeHeaderLines(carried);
|
|
112
|
+
head += "Connection: close\r\n";
|
|
113
|
+
if (body !== undefined && declared === undefined) head += "Transfer-Encoding: chunked\r\n";
|
|
114
|
+
head += "\r\n";
|
|
115
|
+
yield encodeLatin1(head);
|
|
116
|
+
|
|
117
|
+
if (body === undefined) {
|
|
118
|
+
// A head promising `declared` bytes with no body at all is a knowingly
|
|
119
|
+
// truncated message — the failure must land here, not on the peer that
|
|
120
|
+
// trusted the header. Content-Length: 0 is the one declared length an
|
|
121
|
+
// absent body actually satisfies. See I1.
|
|
122
|
+
if (declared !== undefined && declared !== 0) {
|
|
123
|
+
throw new HttpParseError(`body is 0 bytes but Content-Length declares ${declared}`);
|
|
124
|
+
}
|
|
125
|
+
return;
|
|
126
|
+
}
|
|
127
|
+
yield* emitBody(body, declared);
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
export async function* encodeResponse(
|
|
131
|
+
env: ResponseEnvelope,
|
|
132
|
+
body: ByteSource | undefined,
|
|
133
|
+
// Unused: response encoding needs no scheme/host/maxHeaderBytes. Kept
|
|
134
|
+
// positionally so this signature mirrors `encodeRequest`'s, which the
|
|
135
|
+
// `newHttpCodec` wrapper in ./index.ts relies on when partially applying
|
|
136
|
+
// `opts` to both.
|
|
137
|
+
_opts: ResolvedHttpCodecOptions,
|
|
138
|
+
requestMethod?: string,
|
|
139
|
+
): AsyncGenerator<Uint8Array> {
|
|
140
|
+
if (!Number.isInteger(env.status) || env.status < 100 || env.status > 599) {
|
|
141
|
+
throw new HttpParseError(`invalid status: ${env.status}`);
|
|
142
|
+
}
|
|
143
|
+
const reason = env.statusText ?? "";
|
|
144
|
+
assertValidStatusText(reason);
|
|
145
|
+
|
|
146
|
+
// RFC 9110 §8.6: a 1xx or 204 MUST NOT carry Content-Length — there is no
|
|
147
|
+
// body to measure, ever, regardless of what the caller supplied. A
|
|
148
|
+
// response to HEAD is different: no body is sent on *this* response, but
|
|
149
|
+
// the Content-Length describes the body a GET would have sent, so it stays
|
|
150
|
+
// (M4).
|
|
151
|
+
const bodylessStatus = isBodylessStatus(env.status);
|
|
152
|
+
const bodyless = bodylessStatus || requestMethod?.toUpperCase() === "HEAD";
|
|
153
|
+
|
|
154
|
+
const owned = bodylessStatus ? [...RESPONSE_OWNED, "content-length"] : RESPONSE_OWNED;
|
|
155
|
+
const carried = withoutHeaders(env.headers, owned);
|
|
156
|
+
const declared = parseContentLength(getAll(carried, "content-length"));
|
|
157
|
+
|
|
158
|
+
let head = `HTTP/1.1 ${env.status} ${reason}\r\n`;
|
|
159
|
+
head += encodeHeaderLines(carried);
|
|
160
|
+
head += "Connection: close\r\n";
|
|
161
|
+
if (!bodyless && body !== undefined && declared === undefined) {
|
|
162
|
+
head += "Transfer-Encoding: chunked\r\n";
|
|
163
|
+
}
|
|
164
|
+
head += "\r\n";
|
|
165
|
+
yield encodeLatin1(head);
|
|
166
|
+
|
|
167
|
+
if (bodyless) {
|
|
168
|
+
await discard(body);
|
|
169
|
+
return;
|
|
170
|
+
}
|
|
171
|
+
if (body === undefined) {
|
|
172
|
+
// Same truncation guard as encodeRequest — see I1. Not reached when
|
|
173
|
+
// bodyless: a response to HEAD legitimately keeps a declared
|
|
174
|
+
// Content-Length with no body (M4), and that is not a truncation.
|
|
175
|
+
if (declared !== undefined && declared !== 0) {
|
|
176
|
+
throw new HttpParseError(`body is 0 bytes but Content-Length declares ${declared}`);
|
|
177
|
+
}
|
|
178
|
+
return;
|
|
179
|
+
}
|
|
180
|
+
yield* emitBody(body, declared);
|
|
181
|
+
}
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Raised for any byte sequence this codec refuses to interpret. Every case is
|
|
3
|
+
* a refusal to guess: HTTP/1.1 parsers that guess are how request smuggling
|
|
4
|
+
* works.
|
|
5
|
+
*/
|
|
6
|
+
export class HttpParseError extends Error {
|
|
7
|
+
override readonly name = "HttpParseError";
|
|
8
|
+
}
|
|
@@ -0,0 +1,263 @@
|
|
|
1
|
+
import type { ByteReader } from "../bytes.js";
|
|
2
|
+
import { HttpParseError } from "./errors.js";
|
|
3
|
+
|
|
4
|
+
export type HeaderList = [string, string][];
|
|
5
|
+
|
|
6
|
+
const TCHAR = new Set(
|
|
7
|
+
"!#$%&'*+-.^_`|~0123456789abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ",
|
|
8
|
+
);
|
|
9
|
+
|
|
10
|
+
const SP = 0x20;
|
|
11
|
+
const HTAB = 0x09;
|
|
12
|
+
|
|
13
|
+
export function isTokenChar(byte: number): boolean {
|
|
14
|
+
return byte > SP && byte < 0x7f && TCHAR.has(String.fromCharCode(byte));
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
export function isToken(value: string): boolean {
|
|
18
|
+
if (value.length === 0) return false;
|
|
19
|
+
for (const ch of value) {
|
|
20
|
+
if (!TCHAR.has(ch)) return false;
|
|
21
|
+
}
|
|
22
|
+
return true;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* Header bytes are latin-1: byte-preserving, matching Node, and lossless
|
|
27
|
+
* across a decode/encode round-trip. Control characters are rejected
|
|
28
|
+
* separately by `assertValidHeaderValue`.
|
|
29
|
+
*/
|
|
30
|
+
export function decodeLatin1(bytes: Uint8Array): string {
|
|
31
|
+
let out = "";
|
|
32
|
+
for (const byte of bytes) out += String.fromCharCode(byte);
|
|
33
|
+
return out;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export function encodeLatin1(text: string): Uint8Array {
|
|
37
|
+
const out = new Uint8Array(text.length);
|
|
38
|
+
for (let i = 0; i < text.length; i++) {
|
|
39
|
+
const code = text.charCodeAt(i);
|
|
40
|
+
if (code > 0xff) {
|
|
41
|
+
throw new HttpParseError(`not latin-1 encodable: ${JSON.stringify(text)}`);
|
|
42
|
+
}
|
|
43
|
+
out[i] = code;
|
|
44
|
+
}
|
|
45
|
+
return out;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
// RFC 3986 `authority` grammar, restricted to `host [ ":" port ]` (no
|
|
49
|
+
// userinfo — RFC 9110 §7.2 excludes it from Host). Two shapes: an IP-literal
|
|
50
|
+
// in brackets, or a reg-name. Shared by the decoder (validating a Host header
|
|
51
|
+
// taken off the wire) and the encoder (validating the authority `splitTarget`
|
|
52
|
+
// derives from a caller-supplied url) so the grammar is defined exactly once.
|
|
53
|
+
const HOST_IP_LITERAL = /^\[[0-9A-Fa-f:.]+\](:\d{1,5})?$/;
|
|
54
|
+
const HOST_REG_NAME = /^[A-Za-z0-9._~-]+(:\d{1,5})?$/;
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Validates a value that is about to become (or was read as) a `Host`
|
|
58
|
+
* header: untrusted input either way, so an unrecognised shape is a refusal,
|
|
59
|
+
* not a guess. Rejects a present-but-empty value, one carrying userinfo
|
|
60
|
+
* (`evil.com@good.com`), and — the reason this also runs on encode — one
|
|
61
|
+
* smuggling a CRLF-terminated line into the authority of a caller-supplied
|
|
62
|
+
* url. Never applied to trusted configuration (a codec's `opts.host`).
|
|
63
|
+
*/
|
|
64
|
+
export function assertValidHost(host: string): void {
|
|
65
|
+
if (!HOST_IP_LITERAL.test(host) && !HOST_REG_NAME.test(host)) {
|
|
66
|
+
throw new HttpParseError(`invalid Host header: ${JSON.stringify(host)}`);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* A request-target must be entirely visible ASCII (RFC 9110 VCHAR) — nothing
|
|
72
|
+
* `<= 0x20` (space and every control character, CR/LF included) and nothing
|
|
73
|
+
* `>= 0x7F` (DEL and beyond). Applied on encode (`splitTarget`, so a
|
|
74
|
+
* caller-supplied url can never inject a second request line) and on decode
|
|
75
|
+
* (so a relay that decodes then re-encodes can never put an embedded control
|
|
76
|
+
* character back on the wire).
|
|
77
|
+
*/
|
|
78
|
+
export function assertValidTarget(target: string): void {
|
|
79
|
+
for (let i = 0; i < target.length; i++) {
|
|
80
|
+
const code = target.charCodeAt(i);
|
|
81
|
+
if (code <= SP || code >= 0x7f) {
|
|
82
|
+
throw new HttpParseError(`invalid request-target: ${JSON.stringify(target)}`);
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* Shared control-character check behind both `assertValidHeaderValue` and
|
|
89
|
+
* `assertValidStatusText` — the two were previously checked by different,
|
|
90
|
+
* looser rules (statusText only rejected CR/LF), which is exactly the
|
|
91
|
+
* asymmetry class C1 and M5 are both instances of. `label` is prepended
|
|
92
|
+
* verbatim to each message, so callers keep their own wording.
|
|
93
|
+
*/
|
|
94
|
+
function assertNoControlChars(label: string, value: string): void {
|
|
95
|
+
for (let i = 0; i < value.length; i++) {
|
|
96
|
+
const code = value.charCodeAt(i);
|
|
97
|
+
if (code === 0x0d || code === 0x0a) {
|
|
98
|
+
throw new HttpParseError(`${label} contains CR or LF`);
|
|
99
|
+
}
|
|
100
|
+
if ((code < SP && code !== HTAB) || code === 0x7f) {
|
|
101
|
+
throw new HttpParseError(`${label} contains a control character`);
|
|
102
|
+
}
|
|
103
|
+
if (code > 0xff) {
|
|
104
|
+
throw new HttpParseError(`${label} is not latin-1 encodable`);
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
export function assertValidHeaderValue(name: string, value: string): void {
|
|
110
|
+
assertNoControlChars(`header "${name}" value`, value);
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/** Same character class as a header value (M5) — CR/LF, every C0 control, and DEL. */
|
|
114
|
+
export function assertValidStatusText(value: string): void {
|
|
115
|
+
assertNoControlChars("statusText", value);
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
export function encodeHeaderLines(headers: HeaderList): string {
|
|
119
|
+
let out = "";
|
|
120
|
+
for (const [name, value] of headers) {
|
|
121
|
+
if (!isToken(name)) throw new HttpParseError(`invalid header name: ${JSON.stringify(name)}`);
|
|
122
|
+
assertValidHeaderValue(name, value);
|
|
123
|
+
out += `${name}: ${value}\r\n`;
|
|
124
|
+
}
|
|
125
|
+
return out;
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
/**
|
|
129
|
+
* Read to the blank line. `alreadyUsed` is the byte count of the start line,
|
|
130
|
+
* so the bound covers the whole head section rather than the headers alone.
|
|
131
|
+
*/
|
|
132
|
+
export async function readHeaderSection(
|
|
133
|
+
reader: ByteReader,
|
|
134
|
+
maxHeaderBytes: number,
|
|
135
|
+
alreadyUsed: number,
|
|
136
|
+
): Promise<HeaderList> {
|
|
137
|
+
const headers: HeaderList = [];
|
|
138
|
+
let used = alreadyUsed;
|
|
139
|
+
while (true) {
|
|
140
|
+
const remaining = maxHeaderBytes - used;
|
|
141
|
+
if (remaining <= 0) {
|
|
142
|
+
throw new HttpParseError(`head section exceeds ${maxHeaderBytes} bytes`);
|
|
143
|
+
}
|
|
144
|
+
const lineBytes = await reader.readLine(remaining);
|
|
145
|
+
used += lineBytes.byteLength + 2;
|
|
146
|
+
if (used > maxHeaderBytes) {
|
|
147
|
+
throw new HttpParseError(`head section exceeds ${maxHeaderBytes} bytes`);
|
|
148
|
+
}
|
|
149
|
+
if (lineBytes.byteLength === 0) return headers;
|
|
150
|
+
if (lineBytes[0] === SP || lineBytes[0] === HTAB) {
|
|
151
|
+
throw new HttpParseError("obs-fold header continuation is not accepted");
|
|
152
|
+
}
|
|
153
|
+
const line = decodeLatin1(lineBytes);
|
|
154
|
+
const colon = line.indexOf(":");
|
|
155
|
+
if (colon <= 0) throw new HttpParseError(`malformed header line: ${JSON.stringify(line)}`);
|
|
156
|
+
const name = line.slice(0, colon);
|
|
157
|
+
if (!isToken(name)) throw new HttpParseError(`invalid header name: ${JSON.stringify(name)}`);
|
|
158
|
+
const value = line
|
|
159
|
+
.slice(colon + 1)
|
|
160
|
+
.replace(/^[ \t]+/, "")
|
|
161
|
+
.replace(/[ \t]+$/, "");
|
|
162
|
+
assertValidHeaderValue(name, value);
|
|
163
|
+
headers.push([name, value]);
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
export function getAll(headers: HeaderList, name: string): string[] {
|
|
168
|
+
const lower = name.toLowerCase();
|
|
169
|
+
return headers.filter(([k]) => k.toLowerCase() === lower).map(([, v]) => v);
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
export function withoutHeaders(headers: HeaderList, names: string[]): HeaderList {
|
|
173
|
+
const drop = new Set(names.map((n) => n.toLowerCase()));
|
|
174
|
+
return headers.filter(([k]) => !drop.has(k.toLowerCase()));
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* Parse a `Content-Length` field into a single validated value.
|
|
179
|
+
*
|
|
180
|
+
* RFC 9110 §8.6 permits the value to be a comma-separated list of identical
|
|
181
|
+
* numbers (a relay may have appended one), so the list is folded; differing
|
|
182
|
+
* values are a refusal, not a choice. Shared by `resolveFraming` on decode and
|
|
183
|
+
* by the encoder's declared-length check, which previously carried its own
|
|
184
|
+
* copy that did NOT split on commas — so decode accepted `5, 5` while encode
|
|
185
|
+
* rejected it, and a relay that decoded then re-encoded threw.
|
|
186
|
+
*
|
|
187
|
+
* Returns undefined when the field is absent.
|
|
188
|
+
*/
|
|
189
|
+
export function parseContentLength(values: string[]): number | undefined {
|
|
190
|
+
if (values.length === 0) return undefined;
|
|
191
|
+
const unique = new Set(values.flatMap((v) => v.split(",").map((s) => s.trim())));
|
|
192
|
+
if (unique.size !== 1) {
|
|
193
|
+
throw new HttpParseError(`conflicting Content-Length values: ${[...unique].join(", ")}`);
|
|
194
|
+
}
|
|
195
|
+
// `unique.size === 1` is checked directly above, so this spread has exactly
|
|
196
|
+
// one element. The `!` is what lets a consumer compiling with
|
|
197
|
+
// `noUncheckedIndexedAccess` build against this source: without it `raw` is
|
|
198
|
+
// `string | undefined` and the `.test(raw)` below is a type error in THEIR
|
|
199
|
+
// build, not in this package's own (which does not enable the flag, so it
|
|
200
|
+
// cannot catch the regression). It was lost once already, in a rebase that
|
|
201
|
+
// resolved toward a refactor of this function.
|
|
202
|
+
const raw = [...unique][0]!;
|
|
203
|
+
if (!/^\d{1,15}$/.test(raw)) {
|
|
204
|
+
throw new HttpParseError(`invalid Content-Length: ${JSON.stringify(raw)}`);
|
|
205
|
+
}
|
|
206
|
+
return Number(raw);
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
/**
|
|
210
|
+
* Statuses that carry no body whatever the headers say (RFC 9110 §8.6): 1xx,
|
|
211
|
+
* 204 and 304. Defined once because the encoder must not frame a body for
|
|
212
|
+
* them and the decoder must not try to read one — two lists that agreed today
|
|
213
|
+
* and were free to drift tomorrow.
|
|
214
|
+
*
|
|
215
|
+
* A response to HEAD is also bodyless, but that depends on the request rather
|
|
216
|
+
* than the status, so callers test it separately.
|
|
217
|
+
*/
|
|
218
|
+
export function isBodylessStatus(status: number): boolean {
|
|
219
|
+
return status < 200 || status === 204 || status === 304;
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
export type BodyFraming =
|
|
223
|
+
| { kind: "chunked" }
|
|
224
|
+
| { kind: "length"; length: number }
|
|
225
|
+
| { kind: "none" };
|
|
226
|
+
|
|
227
|
+
/**
|
|
228
|
+
* RFC 9112 §6.3, with every ambiguity turned into a refusal. In particular a
|
|
229
|
+
* message declaring both Content-Length and Transfer-Encoding is rejected
|
|
230
|
+
* rather than resolved — disagreeing on which one wins is request smuggling.
|
|
231
|
+
*/
|
|
232
|
+
export function resolveFraming(headers: HeaderList, version: string): BodyFraming {
|
|
233
|
+
const te = getAll(headers, "transfer-encoding");
|
|
234
|
+
const cl = getAll(headers, "content-length");
|
|
235
|
+
|
|
236
|
+
if (te.length > 0 && cl.length > 0) {
|
|
237
|
+
throw new HttpParseError(
|
|
238
|
+
"message declares both Content-Length and Transfer-Encoding; refusing (request smuggling)",
|
|
239
|
+
);
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
if (te.length > 0) {
|
|
243
|
+
const encodings = te
|
|
244
|
+
.join(",")
|
|
245
|
+
.split(",")
|
|
246
|
+
.map((s) => s.trim().toLowerCase())
|
|
247
|
+
.filter((s) => s !== "");
|
|
248
|
+
if (encodings.length !== 1 || encodings[0] !== "chunked") {
|
|
249
|
+
throw new HttpParseError(`unsupported Transfer-Encoding: ${JSON.stringify(te.join(", "))}`);
|
|
250
|
+
}
|
|
251
|
+
if (version === "HTTP/1.0") {
|
|
252
|
+
throw new HttpParseError("Transfer-Encoding is not valid in HTTP/1.0");
|
|
253
|
+
}
|
|
254
|
+
return { kind: "chunked" };
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
const length = parseContentLength(cl);
|
|
258
|
+
if (length !== undefined) {
|
|
259
|
+
return { kind: "length", length };
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
return { kind: "none" };
|
|
263
|
+
}
|