@statewalker/webrun-http-streams 0.1.1 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/README.md +361 -13
  2. package/dist/bytes.d.ts +43 -0
  3. package/dist/bytes.d.ts.map +1 -0
  4. package/dist/codec-default.d.ts +7 -0
  5. package/dist/codec-default.d.ts.map +1 -0
  6. package/dist/duplex-site-builder.d.ts +4 -1
  7. package/dist/duplex-site-builder.d.ts.map +1 -1
  8. package/dist/envelope.d.ts +10 -10
  9. package/dist/envelope.d.ts.map +1 -1
  10. package/dist/fetch.d.ts +3 -2
  11. package/dist/fetch.d.ts.map +1 -1
  12. package/dist/http-data.d.ts +11 -7
  13. package/dist/http-data.d.ts.map +1 -1
  14. package/dist/http-error.d.ts.map +1 -1
  15. package/dist/http-stubs.d.ts +12 -0
  16. package/dist/http-stubs.d.ts.map +1 -1
  17. package/dist/http1/chunked.d.ts +18 -0
  18. package/dist/http1/chunked.d.ts.map +1 -0
  19. package/dist/http1/decode.d.ts +5 -0
  20. package/dist/http1/decode.d.ts.map +1 -0
  21. package/dist/http1/encode.d.ts +20 -0
  22. package/dist/http1/encode.d.ts.map +1 -0
  23. package/dist/http1/errors.d.ts +9 -0
  24. package/dist/http1/errors.d.ts.map +1 -0
  25. package/dist/http1/headers.d.ts +78 -0
  26. package/dist/http1/headers.d.ts.map +1 -0
  27. package/dist/http1/index.d.ts +18 -0
  28. package/dist/http1/index.d.ts.map +1 -0
  29. package/dist/index.d.ts +6 -2
  30. package/dist/index.d.ts.map +1 -1
  31. package/dist/index.js +1040 -138
  32. package/dist/message.d.ts +50 -0
  33. package/dist/message.d.ts.map +1 -0
  34. package/dist/request-streams.d.ts +52 -0
  35. package/dist/request-streams.d.ts.map +1 -0
  36. package/dist/sniff.d.ts +18 -0
  37. package/dist/sniff.d.ts.map +1 -0
  38. package/package.json +13 -7
  39. package/src/bytes.ts +157 -0
  40. package/src/codec-default.ts +13 -0
  41. package/src/duplex-site-builder.ts +10 -2
  42. package/src/envelope.ts +43 -34
  43. package/src/fetch.ts +130 -12
  44. package/src/http-data.ts +160 -17
  45. package/src/http-stubs.ts +72 -18
  46. package/src/http1/chunked.ts +89 -0
  47. package/src/http1/decode.ts +208 -0
  48. package/src/http1/encode.ts +181 -0
  49. package/src/http1/errors.ts +8 -0
  50. package/src/http1/headers.ts +263 -0
  51. package/src/http1/index.ts +40 -0
  52. package/src/index.ts +18 -0
  53. package/src/message.ts +62 -0
  54. package/src/request-streams.ts +75 -0
  55. package/src/sniff.ts +68 -0
  56. package/LICENSE +0 -21
@@ -0,0 +1,208 @@
1
+ import { ByteReader, ByteStreamError } from "../bytes.js";
2
+ import type { ByteSource, DecodedRequest, DecodedResponse } from "../message.js";
3
+ import { decodeChunked } from "./chunked.js";
4
+ import type { ResolvedHttpCodecOptions } from "./encode.js";
5
+ import { HttpParseError } from "./errors.js";
6
+ import {
7
+ assertValidHost,
8
+ assertValidTarget,
9
+ type BodyFraming,
10
+ decodeLatin1,
11
+ getAll,
12
+ isBodylessStatus,
13
+ isToken,
14
+ readHeaderSection,
15
+ resolveFraming,
16
+ } from "./headers.js";
17
+
18
+ const VERSIONS = new Set(["HTTP/1.1", "HTTP/1.0"]);
19
+ const ABSOLUTE_FORM = /^[a-zA-Z][a-zA-Z0-9+.-]*:\/\//;
20
+
21
+ /**
22
+ * One message per Duplex call (ADR-0006), so bytes after a complete message
23
+ * are an error. Checked against what is ALREADY BUFFERED rather than by
24
+ * awaiting end-of-stream: a live socket from a keep-alive peer never reaches
25
+ * EOF, so awaiting one would hang instead of failing.
26
+ */
27
+ function assertNoBufferedBytes(reader: ByteReader): void {
28
+ const extra = reader.bufferedLength();
29
+ if (extra > 0) {
30
+ throw new HttpParseError(`${extra} trailing bytes after a complete message`);
31
+ }
32
+ }
33
+
34
+ /**
35
+ * The public error contract (I2): everything leaving `decodeRequest` /
36
+ * `decodeResponse` is an `HttpParseError`, whether the refusal happened
37
+ * synchronously (a malformed start line or header) or lazily while the body
38
+ * is later drained. `ByteStreamError` — the `ByteReader`'s own class, never
39
+ * exported — is the one thing converted here, message and all preserved via
40
+ * `cause`. Anything else is a genuine failure of the underlying source (a
41
+ * dropped socket, say) and must reach the caller unchanged: blanket-catching
42
+ * would hide that distinction.
43
+ */
44
+ function toHttpParseError(err: unknown): never {
45
+ if (err instanceof ByteStreamError) {
46
+ throw new HttpParseError(err.message, { cause: err });
47
+ }
48
+ throw err;
49
+ }
50
+
51
+ /** Applies `toHttpParseError` across the whole lifetime of a body generator. */
52
+ async function* convertBodyErrors(source: AsyncGenerator<Uint8Array>): AsyncGenerator<Uint8Array> {
53
+ try {
54
+ yield* source;
55
+ } catch (err) {
56
+ toHttpParseError(err);
57
+ }
58
+ }
59
+
60
+ async function* readBody(
61
+ reader: ByteReader,
62
+ framing: BodyFraming,
63
+ opts: ResolvedHttpCodecOptions,
64
+ noneMeansEof: boolean,
65
+ ): AsyncGenerator<Uint8Array> {
66
+ if (framing.kind === "chunked") {
67
+ yield* decodeChunked(reader, opts.maxHeaderBytes);
68
+ assertNoBufferedBytes(reader);
69
+ return;
70
+ }
71
+ if (framing.kind === "length") {
72
+ let remaining = framing.length;
73
+ while (remaining > 0) {
74
+ const part = await reader.readSome(remaining);
75
+ if (part === undefined) {
76
+ throw new HttpParseError(`body truncated: ${remaining} of ${framing.length} bytes missing`);
77
+ }
78
+ remaining -= part.byteLength;
79
+ yield part;
80
+ }
81
+ assertNoBufferedBytes(reader);
82
+ return;
83
+ }
84
+ if (noneMeansEof) {
85
+ yield* reader.rest();
86
+ return;
87
+ }
88
+ assertNoBufferedBytes(reader);
89
+ }
90
+
91
+ /**
92
+ * `readLine`'s bound is best-effort — it is skipped when the line is already
93
+ * buffered — so a very long start line can reach these messages intact. Since
94
+ * a refusal is now echoed back to the sender in a 400 body, quote only enough
95
+ * to diagnose rather than reflecting the whole thing.
96
+ */
97
+ function quoteLine(line: string): string {
98
+ const MAX = 120;
99
+ return line.length <= MAX
100
+ ? JSON.stringify(line)
101
+ : `${JSON.stringify(line.slice(0, MAX))} (truncated from ${line.length} chars)`;
102
+ }
103
+
104
+ export async function decodeRequest(
105
+ input: ByteSource,
106
+ opts: ResolvedHttpCodecOptions,
107
+ ): Promise<DecodedRequest> {
108
+ const reader = new ByteReader(input);
109
+ try {
110
+ const startBytes = await reader.readLine(opts.maxHeaderBytes);
111
+ const startLine = decodeLatin1(startBytes);
112
+ const parts = startLine.split(" ");
113
+ if (parts.length !== 3) {
114
+ throw new HttpParseError(`malformed request line: ${quoteLine(startLine)}`);
115
+ }
116
+ // `parts.length === 3` is checked above, so all three elements are present.
117
+ const [method, target, version] = parts as [string, string, string];
118
+ if (!isToken(method)) throw new HttpParseError(`invalid method: ${JSON.stringify(method)}`);
119
+ if (!VERSIONS.has(version)) {
120
+ throw new HttpParseError(`unsupported HTTP version: ${JSON.stringify(version)}`);
121
+ }
122
+ if (target === "*") {
123
+ throw new HttpParseError("asterisk-form request target is not supported");
124
+ }
125
+ // A relay that decodes then re-encodes must never be able to put a
126
+ // control character (a raw CR chief among them) back onto a request
127
+ // line — see C1.
128
+ assertValidTarget(target);
129
+
130
+ const headers = await readHeaderSection(reader, opts.maxHeaderBytes, startBytes.byteLength + 2);
131
+
132
+ const hosts = getAll(headers, "host");
133
+ if (hosts.length > 1) throw new HttpParseError("multiple Host headers");
134
+ const host = hosts[0];
135
+
136
+ let url: string;
137
+ if (target.startsWith("/")) {
138
+ if (version === "HTTP/1.1" && host === undefined) {
139
+ throw new HttpParseError("HTTP/1.1 request has no Host header");
140
+ }
141
+ if (host !== undefined) assertValidHost(host);
142
+ // The scheme is not on the wire in origin-form; it comes from config.
143
+ url = `${opts.scheme}://${host ?? opts.host}${target}`;
144
+ } else if (ABSOLUTE_FORM.test(target)) {
145
+ url = target;
146
+ } else {
147
+ throw new HttpParseError(`unsupported request target: ${JSON.stringify(target)}`);
148
+ }
149
+
150
+ // `Host` grammar alone is not sufficient proof that the assembled url is
151
+ // sane — this is a fail-safe belt-and-braces check, not a
152
+ // re-serialisation: the parsed `URL` is discarded, and `url` itself is
153
+ // what reaches the envelope untouched.
154
+ try {
155
+ new URL(url);
156
+ } catch {
157
+ throw new HttpParseError(`decoded url is not a valid URL: ${JSON.stringify(url)}`);
158
+ }
159
+
160
+ return {
161
+ envelope: { url, method, headers },
162
+ body: convertBodyErrors(readBody(reader, resolveFraming(headers, version), opts, false)),
163
+ };
164
+ } catch (err) {
165
+ toHttpParseError(err);
166
+ }
167
+ }
168
+
169
+ export async function decodeResponse(
170
+ input: ByteSource,
171
+ opts: ResolvedHttpCodecOptions,
172
+ method: string,
173
+ ): Promise<DecodedResponse> {
174
+ const reader = new ByteReader(input);
175
+ try {
176
+ const startBytes = await reader.readLine(opts.maxHeaderBytes);
177
+ const startLine = decodeLatin1(startBytes);
178
+
179
+ const firstSp = startLine.indexOf(" ");
180
+ if (firstSp === -1) {
181
+ throw new HttpParseError(`malformed status line: ${quoteLine(startLine)}`);
182
+ }
183
+ const version = startLine.slice(0, firstSp);
184
+ if (!VERSIONS.has(version)) {
185
+ throw new HttpParseError(`unsupported HTTP version: ${JSON.stringify(version)}`);
186
+ }
187
+ const afterVersion = startLine.slice(firstSp + 1);
188
+ const secondSp = afterVersion.indexOf(" ");
189
+ const codeText = secondSp === -1 ? afterVersion : afterVersion.slice(0, secondSp);
190
+ if (!/^\d{3}$/.test(codeText)) {
191
+ throw new HttpParseError(`invalid status code: ${JSON.stringify(codeText)}`);
192
+ }
193
+ const status = Number(codeText);
194
+ const statusText = secondSp === -1 ? "" : afterVersion.slice(secondSp + 1);
195
+
196
+ const headers = await readHeaderSection(reader, opts.maxHeaderBytes, startBytes.byteLength + 2);
197
+
198
+ const bodyless = isBodylessStatus(status) || method.toUpperCase() === "HEAD";
199
+ const framing: BodyFraming = bodyless ? { kind: "none" } : resolveFraming(headers, version);
200
+
201
+ return {
202
+ envelope: { status, statusText, headers },
203
+ body: convertBodyErrors(readBody(reader, framing, opts, !bodyless)),
204
+ };
205
+ } catch (err) {
206
+ toHttpParseError(err);
207
+ }
208
+ }
@@ -0,0 +1,181 @@
1
+ import { discard } from "../bytes.js";
2
+ import type { ByteSource, RequestEnvelope, ResponseEnvelope } from "../message.js";
3
+ import { encodeChunked } from "./chunked.js";
4
+ import { HttpParseError } from "./errors.js";
5
+ import {
6
+ assertValidHost,
7
+ assertValidStatusText,
8
+ assertValidTarget,
9
+ encodeHeaderLines,
10
+ encodeLatin1,
11
+ getAll,
12
+ isBodylessStatus,
13
+ isToken,
14
+ parseContentLength,
15
+ withoutHeaders,
16
+ } from "./headers.js";
17
+
18
+ export type ResolvedHttpCodecOptions = {
19
+ scheme: "http" | "https";
20
+ host: string;
21
+ maxHeaderBytes: number;
22
+ };
23
+
24
+ /** Headers the codec owns: a caller-supplied copy is dropped and re-derived. */
25
+ const REQUEST_OWNED = ["host", "connection", "transfer-encoding"];
26
+ const RESPONSE_OWNED = ["connection", "transfer-encoding"];
27
+
28
+ /**
29
+ * Split a URL into an origin-form request target and an authority *without*
30
+ * going through `new URL()`. `URL` normalises percent-encoding and would
31
+ * re-serialise the target — which is the exact class of defect note 15 found
32
+ * in @libp2p/http, where rebuilding the request line silently dropped
33
+ * `url.search`. Here the target is a verbatim slice of the caller's string.
34
+ */
35
+ export function splitTarget(
36
+ url: string,
37
+ opts: ResolvedHttpCodecOptions,
38
+ ): { target: string; authority: string } {
39
+ if (url.startsWith("/")) {
40
+ // Neither field is trusted just because it came from a caller's own url:
41
+ // an unvalidated CRLF here would splice a second request line (or an
42
+ // extra header) onto the wire — see C1.
43
+ assertValidTarget(url);
44
+ return { target: url, authority: opts.host };
45
+ }
46
+ const match = /^[a-zA-Z][a-zA-Z0-9+.-]*:\/\/([^/?#]*)([^#]*)/.exec(url);
47
+ if (!match) {
48
+ throw new HttpParseError(`cannot derive a request target from url: ${JSON.stringify(url)}`);
49
+ }
50
+ // Both capture groups are mandatory in the regex above (neither is
51
+ // `?`-quantified), so a successful match always populates them — an empty
52
+ // string is possible, `undefined` is not.
53
+ const rawAuthority = match[1]!;
54
+ if (rawAuthority === "") {
55
+ throw new HttpParseError(`url has no authority: ${JSON.stringify(url)}`);
56
+ }
57
+ // Host is `uri-host [ ":" port ]` — no userinfo (RFC 9110 §7.2). Strip any
58
+ // `user:pass@` prefix so credentials never end up in a header proxies log.
59
+ const at = rawAuthority.lastIndexOf("@");
60
+ const authority = at === -1 ? rawAuthority : rawAuthority.slice(at + 1);
61
+ assertValidHost(authority);
62
+
63
+ const rawTarget = match[2]!;
64
+ const target = rawTarget === "" ? "/" : rawTarget.startsWith("/") ? rawTarget : `/${rawTarget}`;
65
+ assertValidTarget(target);
66
+ return { target, authority };
67
+ }
68
+
69
+ async function* emitBody(
70
+ body: ByteSource,
71
+ declared: number | undefined,
72
+ ): AsyncGenerator<Uint8Array> {
73
+ if (declared === undefined) {
74
+ yield* encodeChunked(body);
75
+ return;
76
+ }
77
+ let sent = 0;
78
+ for await (const chunk of body) {
79
+ if (chunk.byteLength === 0) continue;
80
+ sent += chunk.byteLength;
81
+ if (sent > declared) {
82
+ throw new HttpParseError(
83
+ `body exceeds declared Content-Length ${declared} (${sent} bytes so far)`,
84
+ );
85
+ }
86
+ yield chunk;
87
+ }
88
+ if (sent !== declared) {
89
+ throw new HttpParseError(`body is ${sent} bytes but Content-Length declares ${declared}`);
90
+ }
91
+ }
92
+
93
+ export async function* encodeRequest(
94
+ env: RequestEnvelope,
95
+ body: ByteSource | undefined,
96
+ opts: ResolvedHttpCodecOptions,
97
+ ): AsyncGenerator<Uint8Array> {
98
+ if (!isToken(env.method)) {
99
+ throw new HttpParseError(`invalid method: ${JSON.stringify(env.method)}`);
100
+ }
101
+ const { target, authority } = splitTarget(env.url, opts);
102
+ if (authority === "") {
103
+ throw new HttpParseError("no Host available: url has no authority and no host is configured");
104
+ }
105
+
106
+ const carried = withoutHeaders(env.headers, REQUEST_OWNED);
107
+ const declared = parseContentLength(getAll(carried, "content-length"));
108
+
109
+ let head = `${env.method} ${target} HTTP/1.1\r\n`;
110
+ head += `Host: ${authority}\r\n`;
111
+ head += encodeHeaderLines(carried);
112
+ head += "Connection: close\r\n";
113
+ if (body !== undefined && declared === undefined) head += "Transfer-Encoding: chunked\r\n";
114
+ head += "\r\n";
115
+ yield encodeLatin1(head);
116
+
117
+ if (body === undefined) {
118
+ // A head promising `declared` bytes with no body at all is a knowingly
119
+ // truncated message — the failure must land here, not on the peer that
120
+ // trusted the header. Content-Length: 0 is the one declared length an
121
+ // absent body actually satisfies. See I1.
122
+ if (declared !== undefined && declared !== 0) {
123
+ throw new HttpParseError(`body is 0 bytes but Content-Length declares ${declared}`);
124
+ }
125
+ return;
126
+ }
127
+ yield* emitBody(body, declared);
128
+ }
129
+
130
+ export async function* encodeResponse(
131
+ env: ResponseEnvelope,
132
+ body: ByteSource | undefined,
133
+ // Unused: response encoding needs no scheme/host/maxHeaderBytes. Kept
134
+ // positionally so this signature mirrors `encodeRequest`'s, which the
135
+ // `newHttpCodec` wrapper in ./index.ts relies on when partially applying
136
+ // `opts` to both.
137
+ _opts: ResolvedHttpCodecOptions,
138
+ requestMethod?: string,
139
+ ): AsyncGenerator<Uint8Array> {
140
+ if (!Number.isInteger(env.status) || env.status < 100 || env.status > 599) {
141
+ throw new HttpParseError(`invalid status: ${env.status}`);
142
+ }
143
+ const reason = env.statusText ?? "";
144
+ assertValidStatusText(reason);
145
+
146
+ // RFC 9110 §8.6: a 1xx or 204 MUST NOT carry Content-Length — there is no
147
+ // body to measure, ever, regardless of what the caller supplied. A
148
+ // response to HEAD is different: no body is sent on *this* response, but
149
+ // the Content-Length describes the body a GET would have sent, so it stays
150
+ // (M4).
151
+ const bodylessStatus = isBodylessStatus(env.status);
152
+ const bodyless = bodylessStatus || requestMethod?.toUpperCase() === "HEAD";
153
+
154
+ const owned = bodylessStatus ? [...RESPONSE_OWNED, "content-length"] : RESPONSE_OWNED;
155
+ const carried = withoutHeaders(env.headers, owned);
156
+ const declared = parseContentLength(getAll(carried, "content-length"));
157
+
158
+ let head = `HTTP/1.1 ${env.status} ${reason}\r\n`;
159
+ head += encodeHeaderLines(carried);
160
+ head += "Connection: close\r\n";
161
+ if (!bodyless && body !== undefined && declared === undefined) {
162
+ head += "Transfer-Encoding: chunked\r\n";
163
+ }
164
+ head += "\r\n";
165
+ yield encodeLatin1(head);
166
+
167
+ if (bodyless) {
168
+ await discard(body);
169
+ return;
170
+ }
171
+ if (body === undefined) {
172
+ // Same truncation guard as encodeRequest — see I1. Not reached when
173
+ // bodyless: a response to HEAD legitimately keeps a declared
174
+ // Content-Length with no body (M4), and that is not a truncation.
175
+ if (declared !== undefined && declared !== 0) {
176
+ throw new HttpParseError(`body is 0 bytes but Content-Length declares ${declared}`);
177
+ }
178
+ return;
179
+ }
180
+ yield* emitBody(body, declared);
181
+ }
@@ -0,0 +1,8 @@
1
+ /**
2
+ * Raised for any byte sequence this codec refuses to interpret. Every case is
3
+ * a refusal to guess: HTTP/1.1 parsers that guess are how request smuggling
4
+ * works.
5
+ */
6
+ export class HttpParseError extends Error {
7
+ override readonly name = "HttpParseError";
8
+ }
@@ -0,0 +1,263 @@
1
+ import type { ByteReader } from "../bytes.js";
2
+ import { HttpParseError } from "./errors.js";
3
+
4
+ export type HeaderList = [string, string][];
5
+
6
+ const TCHAR = new Set(
7
+ "!#$%&'*+-.^_`|~0123456789abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ",
8
+ );
9
+
10
+ const SP = 0x20;
11
+ const HTAB = 0x09;
12
+
13
+ export function isTokenChar(byte: number): boolean {
14
+ return byte > SP && byte < 0x7f && TCHAR.has(String.fromCharCode(byte));
15
+ }
16
+
17
+ export function isToken(value: string): boolean {
18
+ if (value.length === 0) return false;
19
+ for (const ch of value) {
20
+ if (!TCHAR.has(ch)) return false;
21
+ }
22
+ return true;
23
+ }
24
+
25
+ /**
26
+ * Header bytes are latin-1: byte-preserving, matching Node, and lossless
27
+ * across a decode/encode round-trip. Control characters are rejected
28
+ * separately by `assertValidHeaderValue`.
29
+ */
30
+ export function decodeLatin1(bytes: Uint8Array): string {
31
+ let out = "";
32
+ for (const byte of bytes) out += String.fromCharCode(byte);
33
+ return out;
34
+ }
35
+
36
+ export function encodeLatin1(text: string): Uint8Array {
37
+ const out = new Uint8Array(text.length);
38
+ for (let i = 0; i < text.length; i++) {
39
+ const code = text.charCodeAt(i);
40
+ if (code > 0xff) {
41
+ throw new HttpParseError(`not latin-1 encodable: ${JSON.stringify(text)}`);
42
+ }
43
+ out[i] = code;
44
+ }
45
+ return out;
46
+ }
47
+
48
+ // RFC 3986 `authority` grammar, restricted to `host [ ":" port ]` (no
49
+ // userinfo — RFC 9110 §7.2 excludes it from Host). Two shapes: an IP-literal
50
+ // in brackets, or a reg-name. Shared by the decoder (validating a Host header
51
+ // taken off the wire) and the encoder (validating the authority `splitTarget`
52
+ // derives from a caller-supplied url) so the grammar is defined exactly once.
53
+ const HOST_IP_LITERAL = /^\[[0-9A-Fa-f:.]+\](:\d{1,5})?$/;
54
+ const HOST_REG_NAME = /^[A-Za-z0-9._~-]+(:\d{1,5})?$/;
55
+
56
+ /**
57
+ * Validates a value that is about to become (or was read as) a `Host`
58
+ * header: untrusted input either way, so an unrecognised shape is a refusal,
59
+ * not a guess. Rejects a present-but-empty value, one carrying userinfo
60
+ * (`evil.com@good.com`), and — the reason this also runs on encode — one
61
+ * smuggling a CRLF-terminated line into the authority of a caller-supplied
62
+ * url. Never applied to trusted configuration (a codec's `opts.host`).
63
+ */
64
+ export function assertValidHost(host: string): void {
65
+ if (!HOST_IP_LITERAL.test(host) && !HOST_REG_NAME.test(host)) {
66
+ throw new HttpParseError(`invalid Host header: ${JSON.stringify(host)}`);
67
+ }
68
+ }
69
+
70
+ /**
71
+ * A request-target must be entirely visible ASCII (RFC 9110 VCHAR) — nothing
72
+ * `<= 0x20` (space and every control character, CR/LF included) and nothing
73
+ * `>= 0x7F` (DEL and beyond). Applied on encode (`splitTarget`, so a
74
+ * caller-supplied url can never inject a second request line) and on decode
75
+ * (so a relay that decodes then re-encodes can never put an embedded control
76
+ * character back on the wire).
77
+ */
78
+ export function assertValidTarget(target: string): void {
79
+ for (let i = 0; i < target.length; i++) {
80
+ const code = target.charCodeAt(i);
81
+ if (code <= SP || code >= 0x7f) {
82
+ throw new HttpParseError(`invalid request-target: ${JSON.stringify(target)}`);
83
+ }
84
+ }
85
+ }
86
+
87
+ /**
88
+ * Shared control-character check behind both `assertValidHeaderValue` and
89
+ * `assertValidStatusText` — the two were previously checked by different,
90
+ * looser rules (statusText only rejected CR/LF), which is exactly the
91
+ * asymmetry class C1 and M5 are both instances of. `label` is prepended
92
+ * verbatim to each message, so callers keep their own wording.
93
+ */
94
+ function assertNoControlChars(label: string, value: string): void {
95
+ for (let i = 0; i < value.length; i++) {
96
+ const code = value.charCodeAt(i);
97
+ if (code === 0x0d || code === 0x0a) {
98
+ throw new HttpParseError(`${label} contains CR or LF`);
99
+ }
100
+ if ((code < SP && code !== HTAB) || code === 0x7f) {
101
+ throw new HttpParseError(`${label} contains a control character`);
102
+ }
103
+ if (code > 0xff) {
104
+ throw new HttpParseError(`${label} is not latin-1 encodable`);
105
+ }
106
+ }
107
+ }
108
+
109
+ export function assertValidHeaderValue(name: string, value: string): void {
110
+ assertNoControlChars(`header "${name}" value`, value);
111
+ }
112
+
113
+ /** Same character class as a header value (M5) — CR/LF, every C0 control, and DEL. */
114
+ export function assertValidStatusText(value: string): void {
115
+ assertNoControlChars("statusText", value);
116
+ }
117
+
118
+ export function encodeHeaderLines(headers: HeaderList): string {
119
+ let out = "";
120
+ for (const [name, value] of headers) {
121
+ if (!isToken(name)) throw new HttpParseError(`invalid header name: ${JSON.stringify(name)}`);
122
+ assertValidHeaderValue(name, value);
123
+ out += `${name}: ${value}\r\n`;
124
+ }
125
+ return out;
126
+ }
127
+
128
+ /**
129
+ * Read to the blank line. `alreadyUsed` is the byte count of the start line,
130
+ * so the bound covers the whole head section rather than the headers alone.
131
+ */
132
+ export async function readHeaderSection(
133
+ reader: ByteReader,
134
+ maxHeaderBytes: number,
135
+ alreadyUsed: number,
136
+ ): Promise<HeaderList> {
137
+ const headers: HeaderList = [];
138
+ let used = alreadyUsed;
139
+ while (true) {
140
+ const remaining = maxHeaderBytes - used;
141
+ if (remaining <= 0) {
142
+ throw new HttpParseError(`head section exceeds ${maxHeaderBytes} bytes`);
143
+ }
144
+ const lineBytes = await reader.readLine(remaining);
145
+ used += lineBytes.byteLength + 2;
146
+ if (used > maxHeaderBytes) {
147
+ throw new HttpParseError(`head section exceeds ${maxHeaderBytes} bytes`);
148
+ }
149
+ if (lineBytes.byteLength === 0) return headers;
150
+ if (lineBytes[0] === SP || lineBytes[0] === HTAB) {
151
+ throw new HttpParseError("obs-fold header continuation is not accepted");
152
+ }
153
+ const line = decodeLatin1(lineBytes);
154
+ const colon = line.indexOf(":");
155
+ if (colon <= 0) throw new HttpParseError(`malformed header line: ${JSON.stringify(line)}`);
156
+ const name = line.slice(0, colon);
157
+ if (!isToken(name)) throw new HttpParseError(`invalid header name: ${JSON.stringify(name)}`);
158
+ const value = line
159
+ .slice(colon + 1)
160
+ .replace(/^[ \t]+/, "")
161
+ .replace(/[ \t]+$/, "");
162
+ assertValidHeaderValue(name, value);
163
+ headers.push([name, value]);
164
+ }
165
+ }
166
+
167
+ export function getAll(headers: HeaderList, name: string): string[] {
168
+ const lower = name.toLowerCase();
169
+ return headers.filter(([k]) => k.toLowerCase() === lower).map(([, v]) => v);
170
+ }
171
+
172
+ export function withoutHeaders(headers: HeaderList, names: string[]): HeaderList {
173
+ const drop = new Set(names.map((n) => n.toLowerCase()));
174
+ return headers.filter(([k]) => !drop.has(k.toLowerCase()));
175
+ }
176
+
177
+ /**
178
+ * Parse a `Content-Length` field into a single validated value.
179
+ *
180
+ * RFC 9110 §8.6 permits the value to be a comma-separated list of identical
181
+ * numbers (a relay may have appended one), so the list is folded; differing
182
+ * values are a refusal, not a choice. Shared by `resolveFraming` on decode and
183
+ * by the encoder's declared-length check, which previously carried its own
184
+ * copy that did NOT split on commas — so decode accepted `5, 5` while encode
185
+ * rejected it, and a relay that decoded then re-encoded threw.
186
+ *
187
+ * Returns undefined when the field is absent.
188
+ */
189
+ export function parseContentLength(values: string[]): number | undefined {
190
+ if (values.length === 0) return undefined;
191
+ const unique = new Set(values.flatMap((v) => v.split(",").map((s) => s.trim())));
192
+ if (unique.size !== 1) {
193
+ throw new HttpParseError(`conflicting Content-Length values: ${[...unique].join(", ")}`);
194
+ }
195
+ // `unique.size === 1` is checked directly above, so this spread has exactly
196
+ // one element. The `!` is what lets a consumer compiling with
197
+ // `noUncheckedIndexedAccess` build against this source: without it `raw` is
198
+ // `string | undefined` and the `.test(raw)` below is a type error in THEIR
199
+ // build, not in this package's own (which does not enable the flag, so it
200
+ // cannot catch the regression). It was lost once already, in a rebase that
201
+ // resolved toward a refactor of this function.
202
+ const raw = [...unique][0]!;
203
+ if (!/^\d{1,15}$/.test(raw)) {
204
+ throw new HttpParseError(`invalid Content-Length: ${JSON.stringify(raw)}`);
205
+ }
206
+ return Number(raw);
207
+ }
208
+
209
+ /**
210
+ * Statuses that carry no body whatever the headers say (RFC 9110 §8.6): 1xx,
211
+ * 204 and 304. Defined once because the encoder must not frame a body for
212
+ * them and the decoder must not try to read one — two lists that agreed today
213
+ * and were free to drift tomorrow.
214
+ *
215
+ * A response to HEAD is also bodyless, but that depends on the request rather
216
+ * than the status, so callers test it separately.
217
+ */
218
+ export function isBodylessStatus(status: number): boolean {
219
+ return status < 200 || status === 204 || status === 304;
220
+ }
221
+
222
+ export type BodyFraming =
223
+ | { kind: "chunked" }
224
+ | { kind: "length"; length: number }
225
+ | { kind: "none" };
226
+
227
+ /**
228
+ * RFC 9112 §6.3, with every ambiguity turned into a refusal. In particular a
229
+ * message declaring both Content-Length and Transfer-Encoding is rejected
230
+ * rather than resolved — disagreeing on which one wins is request smuggling.
231
+ */
232
+ export function resolveFraming(headers: HeaderList, version: string): BodyFraming {
233
+ const te = getAll(headers, "transfer-encoding");
234
+ const cl = getAll(headers, "content-length");
235
+
236
+ if (te.length > 0 && cl.length > 0) {
237
+ throw new HttpParseError(
238
+ "message declares both Content-Length and Transfer-Encoding; refusing (request smuggling)",
239
+ );
240
+ }
241
+
242
+ if (te.length > 0) {
243
+ const encodings = te
244
+ .join(",")
245
+ .split(",")
246
+ .map((s) => s.trim().toLowerCase())
247
+ .filter((s) => s !== "");
248
+ if (encodings.length !== 1 || encodings[0] !== "chunked") {
249
+ throw new HttpParseError(`unsupported Transfer-Encoding: ${JSON.stringify(te.join(", "))}`);
250
+ }
251
+ if (version === "HTTP/1.0") {
252
+ throw new HttpParseError("Transfer-Encoding is not valid in HTTP/1.0");
253
+ }
254
+ return { kind: "chunked" };
255
+ }
256
+
257
+ const length = parseContentLength(cl);
258
+ if (length !== undefined) {
259
+ return { kind: "length", length };
260
+ }
261
+
262
+ return { kind: "none" };
263
+ }