hugo-balancer 0.0.0-stage → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Hugo-huy2004
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md CHANGED
@@ -1,3 +1,71 @@
1
- # Temporary Holding Version
1
+ # hugo-balancer
2
2
 
3
- This version is a temporary placeholder for this package. An operational version to replace this has been submitted for review and is awaiting a staged release.
3
+ A layer-7 load balancer for HTTP and WebSocket in plain Node, with a self-healing local cluster and a load tester. Zero dependencies.
4
+
5
+ ```
6
+ npm install hugo-balancer
7
+ ```
8
+
9
+ ## Three commands
10
+
11
+ ```bash
12
+ # 1. Balance existing servers
13
+ LB_API=127.0.0.1:5011,127.0.0.1:5012,127.0.0.1:5013 npx hugo-balancer start
14
+
15
+ # 2. Run N copies of your server + the balancer, restart crashed copies, roll restarts on HUP
16
+ npx hugo-balancer cluster server.js
17
+ kill -HUP <pid> # zero-downtime: one copy at a time, N−1 always serving
18
+
19
+ # 3. Load-test it
20
+ npx hugo-balancer bench --url http://localhost:8080 --paths /api/a,/api/b --c 32 --d 20 --json out.json
21
+ ```
22
+
23
+ ## What it does
24
+
25
+ | Concern | Behaviour |
26
+ |---|---|
27
+ | Choosing a backend | `p2c-ewma` (default): two random backends, keep the one with lower latency × (in flight + 1); also `least-conn`, `round-robin`, `random` |
28
+ | Stateful tier | Paths in `LB_REALTIME_PATHS` (default `/socket.io`) go to the `realtime` pool in failover mode: always the first healthy copy |
29
+ | Dead backends | Active health checks (`rise`/`fall`) plus passive ejection after 3 consecutive errors, doubling up to 60 s |
30
+ | Retries | On another backend: always when the connection was refused (never reached the server, so POST is safe); for GET/HEAD/OPTIONS also on a broken connection or 502/503 |
31
+ | Slow backends | No response headers within `LB_HEADER_TIMEOUT_MS` → 504; latency feeds the p2c score so slow copies are avoided early |
32
+ | Bodies | Up to 1 MB buffered for retries; larger uploads stream straight through (one attempt) |
33
+ | WebSocket | Upgrade forwarded, TCP piped both ways |
34
+ | Tracing | `X-Forwarded-For/Proto/Host`, `X-Request-Id`, `X-LB-Retried` |
35
+ | Stats | `GET /__lb/stats` from loopback only: per-backend up, ejected, in flight, EWMA, totals |
36
+
37
+ ## Configuration
38
+
39
+ | Variable | Default | |
40
+ |---|---|---|
41
+ | `LB_PORT` | `8080` | listen port |
42
+ | `LB_API` | — | api backends `host:port,…` (required for `start`) |
43
+ | `LB_REALTIME` | — | realtime backends; empty = no realtime pool |
44
+ | `LB_REALTIME_PATHS` | `/socket.io` | path prefixes routed to the realtime pool |
45
+ | `LB_ALGO` | `p2c-ewma` | `p2c-ewma` · `least-conn` · `round-robin` · `random` |
46
+ | `LB_HEALTH_PATH` | `/healthz` | must answer 200 when the copy can serve |
47
+ | `LB_HC_INTERVAL_MS` / `LB_HC_TIMEOUT_MS` | `1000` / `800` | health check cadence |
48
+ | `LB_HEADER_TIMEOUT_MS` | `30000` | wait for response headers |
49
+ | `LB_MAX_RETRIES` | `2` | extra attempts per request |
50
+ | `LB_CLIENT_IP_HEADER` | — | trust this header for the client IP (only when every connection comes through that proxy) |
51
+
52
+ `cluster` also reads `LB_API_COUNT` (3), `LB_API_BASE_PORT` (5011), `LB_RT_PORT` (5021) and `LB_RT_COUNT` (1, or 0 for none). Each copy starts with `ROLE` (`api` | `realtime`), `PORT` and `INSTANCE_ID` in its environment.
53
+
54
+ **Your server, for zero-downtime restarts:** on SIGTERM, answer the health path with 503, wait a couple of health-check intervals, then close. Set `server.keepAliveTimeout` above 30 s so idle pooled connections are closed by the balancer first.
55
+
56
+ ## From code
57
+
58
+ ```js
59
+ const { createLoadBalancer, readConfig } = require('hugo-balancer');
60
+
61
+ const lb = createLoadBalancer({ ...readConfig(), api: ['127.0.0.1:5011', '127.0.0.1:5012'], realtimePaths: ['/socket.io', '/rooms'] });
62
+ await lb.listen(8080);
63
+ lb.stats(); // same as /__lb/stats
64
+ await lb.close();
65
+ ```
66
+
67
+ The selection algorithms are exported too (`createPool`, `pick`, `recordLatency`, `recordFailure`, `recordHealth`) for simulations.
68
+
69
+ ## License
70
+
71
+ MIT
@@ -0,0 +1,34 @@
1
+ #!/usr/bin/env node
2
+ // hugo-balancer start load balancer configured by LB_* variables (LB_API required)
3
+ // hugo-balancer cluster <server.js> N copies of your server + the load balancer, self-healing, HUP = rolling restart
4
+ // hugo-balancer bench --url <url> … load test: throughput, latency percentiles, which copy answered
5
+ const path = require('path');
6
+ const { createLoadBalancer, readConfig, runCluster, bench } = require('..');
7
+ const { parseArgs } = require('../src/bench');
8
+
9
+ const [cmd = 'start', ...rest] = process.argv.slice(2);
10
+ const fail = (msg) => { console.error(`hugo-balancer: ${msg}`); process.exit(1); };
11
+
12
+ if (cmd === 'start') {
13
+ const cfg = readConfig();
14
+ if (!cfg.api.length) fail('set LB_API to the api backends, e.g. LB_API=127.0.0.1:5011,127.0.0.1:5012');
15
+ const lb = createLoadBalancer(cfg);
16
+ lb.listen().then((port) => console.log(`[lb] :${port} algo=${cfg.algo} api=[${cfg.api}] realtime=[${cfg.realtime}]`));
17
+ const stop = () => lb.close().then(() => process.exit(0));
18
+ process.on('SIGTERM', stop);
19
+ process.on('SIGINT', stop);
20
+ } else if (cmd === 'cluster') {
21
+ if (!rest[0]) fail('usage: hugo-balancer cluster <server.js>');
22
+ runCluster(path.resolve(rest[0]));
23
+ } else if (cmd === 'bench') {
24
+ const args = parseArgs(rest);
25
+ bench(args).then((r) => {
26
+ const { timeline, ...summary } = r;
27
+ console.log(JSON.stringify(summary, null, 2));
28
+ const bad = timeline.map((t, s) => (t.err ? `${s}s:${t.err}` : null)).filter(Boolean);
29
+ console.log(`seconds with errors: ${bad.length ? bad.join(' ') : 'none'}`);
30
+ if (args.json) require('fs').writeFileSync(args.json, JSON.stringify({ args, ...r }, null, 2));
31
+ });
32
+ } else {
33
+ fail(`unknown command "${cmd}" — use start, cluster or bench`);
34
+ }
package/index.js ADDED
@@ -0,0 +1,6 @@
1
+ const { createLoadBalancer, readConfig } = require('./src/proxy');
2
+ const { runCluster } = require('./src/cluster');
3
+ const { bench } = require('./src/bench');
4
+ const balancer = require('./src/balancer');
5
+
6
+ module.exports = { createLoadBalancer, readConfig, runCluster, bench, ...balancer };
package/package.json CHANGED
@@ -1,6 +1,18 @@
1
1
  {
2
2
  "name": "hugo-balancer",
3
- "version": "0.0.0-stage",
4
- "stub": true,
5
- "description": "Temporary package placeholder for staged publishing"
6
- }
3
+ "version": "0.1.0",
4
+ "description": "Layer-7 load balancer for HTTP and WebSocket in plain Node: p2c-EWMA, least-conn and round-robin, health checks, outlier ejection, safe retries, a self-healing local cluster with rolling restarts, and a load tester. Zero dependencies",
5
+ "keywords": ["load-balancer", "reverse-proxy", "proxy", "websocket", "p2c", "ewma", "least-connections", "health-check", "circuit-breaker", "cluster", "rolling-restart", "benchmark"],
6
+ "homepage": "https://github.com/Hugo-huy2004/FINAL_PROJECT_HUGOMUSIC/tree/main/packages/balancer#readme",
7
+ "repository": { "type": "git", "url": "git+https://github.com/Hugo-huy2004/FINAL_PROJECT_HUGOMUSIC.git", "directory": "packages/balancer" },
8
+ "author": "Hugo-huy2004",
9
+ "license": "MIT",
10
+ "main": "index.js",
11
+ "bin": { "hugo-balancer": "bin/hugo-balancer.js" },
12
+ "files": ["index.js", "src", "bin", "README.md", "LICENSE"],
13
+ "engines": { "node": ">=18" },
14
+ "scripts": {
15
+ "test": "node checks/lb.check.js",
16
+ "prepublishOnly": "npm test"
17
+ }
18
+ }
@@ -0,0 +1,114 @@
1
+ // Backend selection — pure computation, no I/O, so every rule is unit-tested (checks/lb.check.js).
2
+ //
3
+ // Algorithms (pick one per pool to compare them under load):
4
+ // random baseline.
5
+ // round-robin one backend after another; counts requests, blind to how heavy each one is.
6
+ // least-conn the backend with the fewest requests in flight; good when durations differ wildly
7
+ // (an audio stream held for minutes next to a JSON call of a few ms).
8
+ // p2c-ewma power of two choices + peak-EWMA latency: draw two backends at random, keep the one with the
9
+ // lower latency × (in-flight + 1). Two comparisons are enough to get close to optimal, and a
10
+ // backend that starts slowing down is avoided before it fails.
11
+ //
12
+ // Each pool has a mode: 'balance' (share the load) or 'failover' (always the first healthy backend in order —
13
+ // for a stateful tier whose state lives in one process).
14
+ //
15
+ // Two ways a bad backend is detected:
16
+ // active a health check polls the health path; `rise` passes bring it up, `fall` failures take it down.
17
+ // passive outlier ejection: `ejectAfter` consecutive errors eject it for ejectBaseMs, doubling on every
18
+ // repeat up to ejectMaxMs.
19
+
20
+ const EWMA_ALPHA = 0.3;
21
+
22
+ function createBackend(addr) {
23
+ const [host, port] = addr.split(':');
24
+ return {
25
+ addr, host, port: Number(port),
26
+ up: true, // set by the active health check; starts up so there is no 503 at boot
27
+ hcOk: 0, hcFail: 0, // consecutive health-check passes / failures
28
+ inflight: 0, // requests / connections in progress
29
+ ewmaMs: 0, // time to response headers (TTFB), peak EWMA
30
+ errors: 0, // consecutive connection errors (passive)
31
+ ejections: 0, ejectedUntil: 0,
32
+ requests: 0, failures: 0, // totals, for the stats endpoint
33
+ };
34
+ }
35
+
36
+ function createPool(name, addrs, { mode = 'balance', algo = 'p2c-ewma', rise = 2, fall = 2, ejectAfter = 3, ejectBaseMs = 5000, ejectMaxMs = 60000, rand = Math.random } = {}) {
37
+ return { name, mode, algo, rise, fall, ejectAfter, ejectBaseMs, ejectMaxMs, rand, rr: 0, backends: addrs.map(createBackend) };
38
+ }
39
+
40
+ const available = (pool, now) => pool.backends.filter((b) => b.up && b.ejectedUntil <= now);
41
+
42
+ // p2c-ewma score. An unmeasured backend (ewmaMs = 0) counts as fastest, so it gets tried.
43
+ const cost = (b) => (b.ewmaMs || 1) * (b.inflight + 1);
44
+
45
+ /** Pick a backend, skipping those in `exclude` (already tried for this request). null when none is available. */
46
+ function pick(pool, { exclude = new Set(), now = Date.now() } = {}) {
47
+ const list = available(pool, now).filter((b) => !exclude.has(b));
48
+ if (!list.length) return null;
49
+ if (pool.mode === 'failover') return list[0];
50
+ switch (pool.algo) {
51
+ case 'random':
52
+ return list[Math.floor(pool.rand() * list.length)];
53
+ case 'round-robin':
54
+ return list[pool.rr++ % list.length];
55
+ case 'least-conn': {
56
+ const min = Math.min(...list.map((b) => b.inflight));
57
+ const ties = list.filter((b) => b.inflight === min);
58
+ return ties[Math.floor(pool.rand() * ties.length)];
59
+ }
60
+ case 'p2c-ewma': {
61
+ if (list.length === 1) return list[0];
62
+ const i = Math.floor(pool.rand() * list.length);
63
+ let j = Math.floor(pool.rand() * (list.length - 1));
64
+ if (j >= i) j += 1; // two DIFFERENT backends
65
+ return cost(list[i]) <= cost(list[j]) ? list[i] : list[j];
66
+ }
67
+ default:
68
+ throw new Error(`Unknown algorithm: ${pool.algo}`);
69
+ }
70
+ }
71
+
72
+ // Peak EWMA: a higher sample is taken at once (react fast when a backend slows down); a lower one decays slowly.
73
+ function recordLatency(b, ms) {
74
+ b.ewmaMs = b.ewmaMs === 0 || ms > b.ewmaMs ? ms : b.ewmaMs * (1 - EWMA_ALPHA) + ms * EWMA_ALPHA;
75
+ }
76
+
77
+ function recordSuccess(b) {
78
+ b.errors = 0;
79
+ }
80
+
81
+ /** A connection error or 502/503. Returns true when this failure ejected the backend. */
82
+ function recordFailure(pool, b, now = Date.now()) {
83
+ b.failures += 1;
84
+ b.errors += 1;
85
+ if (b.errors < pool.ejectAfter) return false;
86
+ b.errors = 0;
87
+ b.ejectedUntil = now + Math.min(pool.ejectMaxMs, pool.ejectBaseMs * 2 ** b.ejections);
88
+ b.ejections += 1;
89
+ return true;
90
+ }
91
+
92
+ /** Result of one active health check. Returns 'up' | 'down' when the state changes, null otherwise. */
93
+ function recordHealth(pool, b, ok) {
94
+ if (ok) {
95
+ b.hcFail = 0;
96
+ b.hcOk += 1;
97
+ if (!b.up && b.hcOk >= pool.rise) {
98
+ b.up = true;
99
+ b.ejections = 0; // truly recovered → forget past ejections
100
+ b.ejectedUntil = 0;
101
+ return 'up';
102
+ }
103
+ } else {
104
+ b.hcOk = 0;
105
+ b.hcFail += 1;
106
+ if (b.up && b.hcFail >= pool.fall) {
107
+ b.up = false;
108
+ return 'down';
109
+ }
110
+ }
111
+ return null;
112
+ }
113
+
114
+ module.exports = { createPool, pick, recordLatency, recordSuccess, recordFailure, recordHealth, available };
package/src/bench.js ADDED
@@ -0,0 +1,61 @@
1
+ // Load generator for load-balancing experiments (no external tool needed):
2
+ // hugo-balancer bench --url http://localhost:8080 --paths /api/a,/api/b --c 32 --d 20 [--json out.json]
3
+ // C keep-alive connections run in parallel for D seconds, each cycling through the paths. Reports throughput,
4
+ // p50/p95/p99/max latency, status codes, network errors, and which copy answered (X-Instance header).
5
+ const http = require('http');
6
+
7
+ function parseArgs(argv) {
8
+ const a = { url: 'http://localhost:8080', paths: '/', c: 32, d: 20, json: '' };
9
+ for (let i = 0; i < argv.length; i += 2) a[argv[i].replace(/^--/, '')] = argv[i + 1];
10
+ return { ...a, c: Number(a.c), d: Number(a.d), paths: a.paths.split(',') };
11
+ }
12
+
13
+ const pct = (sorted, p) => sorted.length ? sorted[Math.min(sorted.length - 1, Math.floor(p * sorted.length))] : 0;
14
+
15
+ async function bench({ url, paths, c, d }, { onTick } = {}) {
16
+ const { hostname, port } = new URL(url);
17
+ const agent = new http.Agent({ keepAlive: true, maxSockets: c });
18
+ const lat = [];
19
+ const status = {};
20
+ const instances = {};
21
+ const timeline = []; // errors per second — shows whether killing a copy mid-run caused any
22
+ let errors = 0;
23
+ let i = 0;
24
+ const start = Date.now();
25
+ const end = start + d * 1000;
26
+
27
+ const one = () => new Promise((resolve) => {
28
+ const path = paths[i++ % paths.length];
29
+ const t0 = process.hrtime.bigint();
30
+ const sec = Math.floor((Date.now() - start) / 1000);
31
+ timeline[sec] ??= { ok: 0, err: 0 };
32
+ const req = http.get({ hostname, port, path, agent }, (res) => {
33
+ res.resume();
34
+ res.on('end', () => {
35
+ lat.push(Number(process.hrtime.bigint() - t0) / 1e6);
36
+ status[res.statusCode] = (status[res.statusCode] || 0) + 1;
37
+ const inst = res.headers['x-instance'] || '-';
38
+ instances[inst] = (instances[inst] || 0) + 1;
39
+ timeline[sec][res.statusCode < 500 ? 'ok' : 'err'] += 1;
40
+ resolve();
41
+ });
42
+ });
43
+ req.on('error', () => { errors += 1; timeline[sec].err += 1; resolve(); });
44
+ });
45
+
46
+ const tick = onTick && setInterval(() => onTick(Date.now() - start), 1000);
47
+ await Promise.all(Array.from({ length: c }, async () => { while (Date.now() < end) await one(); }));
48
+ clearInterval(tick);
49
+ agent.destroy();
50
+
51
+ lat.sort((x, y) => x - y);
52
+ const total = lat.length + errors;
53
+ const r1 = (x) => Math.round(x * 10) / 10;
54
+ return {
55
+ requests: total, rps: Math.round(total / d), errors, status, instances,
56
+ latencyMs: { p50: r1(pct(lat, 0.5)), p95: r1(pct(lat, 0.95)), p99: r1(pct(lat, 0.99)), max: r1(lat.at(-1) || 0) },
57
+ timeline: timeline.map((t) => t || { ok: 0, err: 0 }),
58
+ };
59
+ }
60
+
61
+ module.exports = { bench, parseArgs };
package/src/cluster.js ADDED
@@ -0,0 +1,115 @@
1
+ // Run a whole cluster on one machine with one command: N copies of your server + an optional realtime copy +
2
+ // the load balancer in front.
3
+ //
4
+ // Each copy gets ROLE ('api' | 'realtime'), PORT and INSTANCE_ID in its environment and must answer the health
5
+ // path with 200 when ready.
6
+ // - Self-healing: a copy that dies is restarted after 1 s → 2 s → … → 30 s, so a crash loop cannot spin;
7
+ // a copy that stayed up 60 s starts again from 1 s.
8
+ // - Zero-downtime deploy: `kill -HUP <pid>` restarts the api copies one by one — SIGTERM (the copy drains
9
+ // itself), wait for it to exit, start the new one, wait for its health check — so N−1 copies always serve.
10
+ const { spawn } = require('child_process');
11
+ const http = require('http');
12
+ const { createLoadBalancer, readConfig } = require('./proxy');
13
+
14
+ const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
15
+ const log = (...a) => console.log('[cluster]', ...a);
16
+
17
+ function healthy(port, healthPath) {
18
+ return new Promise((resolve) => {
19
+ const req = http.get({ host: '127.0.0.1', port, path: healthPath, timeout: 1000, agent: false }, (r) => { r.resume(); resolve(r.statusCode === 200); });
20
+ req.on('timeout', () => req.destroy());
21
+ req.on('error', () => resolve(false));
22
+ });
23
+ }
24
+
25
+ async function waitHealthy(port, healthPath, timeoutMs = 30000) {
26
+ const end = Date.now() + timeoutMs;
27
+ while (Date.now() < end) {
28
+ if (await healthy(port, healthPath)) return true;
29
+ await sleep(300);
30
+ }
31
+ return false;
32
+ }
33
+
34
+ function createWorker(script, role, port) {
35
+ const w = { role, port, id: `${role}-${port}`, proc: null, restarts: 0, startedAt: 0, stopping: false };
36
+ w.start = () => {
37
+ w.startedAt = Date.now();
38
+ w.proc = spawn(process.execPath, [script], {
39
+ env: { ...process.env, ROLE: role, PORT: String(port), INSTANCE_ID: w.id },
40
+ stdio: ['ignore', 'inherit', 'inherit'],
41
+ });
42
+ w.proc.on('exit', async (code, signal) => {
43
+ if (w.stopping) return;
44
+ if (Date.now() - w.startedAt > 60000) w.restarts = 0;
45
+ const delay = Math.min(30000, 1000 * 2 ** w.restarts);
46
+ w.restarts += 1;
47
+ log(`${w.id} exited (code=${code} signal=${signal}) → restart in ${delay} ms`);
48
+ await sleep(delay);
49
+ if (!w.stopping) w.start();
50
+ });
51
+ };
52
+ // Graceful stop: SIGTERM, then wait until the process has fully exited.
53
+ w.stop = () => new Promise((resolve) => {
54
+ if (!w.proc || w.proc.exitCode !== null) return resolve();
55
+ w.stopping = true;
56
+ w.proc.once('exit', () => resolve());
57
+ w.proc.kill('SIGTERM');
58
+ });
59
+ return w;
60
+ }
61
+
62
+ /**
63
+ * @param {string} script path of the server entry file
64
+ * @param {object} [opts] apiCount (3), apiBasePort (5011), realtime (true), realtimePort (5021), plus any
65
+ * createLoadBalancer option (port, realtimePaths, algo…). Defaults come from LB_* environment variables.
66
+ */
67
+ async function runCluster(script, opts = {}) {
68
+ const env = process.env;
69
+ const o = {
70
+ ...readConfig(),
71
+ apiCount: Number(env.LB_API_COUNT || 3),
72
+ apiBasePort: Number(env.LB_API_BASE_PORT || 5011),
73
+ realtime: env.LB_RT_COUNT !== '0',
74
+ realtimePort: Number(env.LB_RT_PORT || 5021),
75
+ ...opts,
76
+ };
77
+ const api = Array.from({ length: o.apiCount }, (_, i) => createWorker(script, 'api', o.apiBasePort + i));
78
+ const rt = o.realtime ? [createWorker(script, 'realtime', o.realtimePort)] : [];
79
+ const workers = [...api, ...rt];
80
+ workers.forEach((w) => w.start());
81
+
82
+ const lb = createLoadBalancer({ ...o, api: api.map((w) => `127.0.0.1:${w.port}`), realtime: rt.map((w) => `127.0.0.1:${w.port}`) });
83
+ const port = await lb.listen();
84
+ log(`LB :${port} algo=${o.algo} — ${api.length} api + ${rt.length} realtime`);
85
+
86
+ let rolling = false;
87
+ process.on('SIGHUP', async () => {
88
+ if (rolling) return;
89
+ rolling = true;
90
+ log('rolling restart started');
91
+ for (const w of api) {
92
+ await w.stop();
93
+ w.stopping = false;
94
+ w.restarts = 0;
95
+ w.start();
96
+ const ok = await waitHealthy(w.port, o.healthPath);
97
+ log(`${w.id} ${ok ? 'healthy' : 'NOT healthy after 30 s — rolling restart stopped'}`);
98
+ if (!ok) break;
99
+ }
100
+ rolling = false;
101
+ log('rolling restart done');
102
+ });
103
+
104
+ const shutdown = async () => {
105
+ log('shutting down');
106
+ await Promise.all(workers.map((w) => w.stop()));
107
+ await lb.close();
108
+ process.exit(0);
109
+ };
110
+ process.on('SIGTERM', shutdown);
111
+ process.on('SIGINT', shutdown);
112
+ return lb;
113
+ }
114
+
115
+ module.exports = { runCluster };
package/src/proxy.js ADDED
@@ -0,0 +1,260 @@
1
+ // Layer-7 load balancer for HTTP and WebSocket — plain Node, no dependencies. Backend choice: ./balancer.js.
2
+ //
3
+ // Two pools:
4
+ // api stateless REST, balanced across every copy.
5
+ // realtime optional; requests under `realtimePaths` (default /socket.io) go to the first healthy copy only
6
+ // (failover), for state that lives in one process — rooms, live sessions.
7
+ //
8
+ // Connection handling:
9
+ // - Keep-alive pool per backend (http.Agent): no TCP handshake per request.
10
+ // - Retry on ANOTHER backend when a request fails before any response: always when the connection was refused
11
+ // (the request never reached the backend, so even POST is safe); for GET/HEAD/OPTIONS also when the
12
+ // connection broke or the backend answered 502/503 (e.g. it is draining).
13
+ // - Timeout waiting for response headers → 504, so a client never hangs forever.
14
+ // - X-Forwarded-For/Proto/Host and X-Request-Id tell the backend the real client and trace a request across layers.
15
+ // - WebSocket: after the backend accepts the Upgrade, the raw TCP stream is piped both ways.
16
+ const http = require('http');
17
+ const net = require('net');
18
+ const crypto = require('crypto');
19
+ const { createPool, pick, recordLatency, recordSuccess, recordFailure, recordHealth } = require('./balancer');
20
+
21
+ const IDEMPOTENT = new Set(['GET', 'HEAD', 'OPTIONS']);
22
+ const MAX_BUFFERED_BODY = 1024 * 1024; // larger bodies (uploads) are streamed straight through: one attempt, no retry
23
+ const HOP_BY_HOP = ['connection', 'keep-alive', 'proxy-connection', 'transfer-encoding', 'te', 'trailer', 'upgrade'];
24
+ const LOOPBACK = ['127.0.0.1', '::1', '::ffff:127.0.0.1'];
25
+
26
+ /** Configuration from environment variables (LB_*); every field can also be passed to createLoadBalancer directly. */
27
+ function readConfig(env = process.env) {
28
+ const list = (v) => (v || '').split(',').map((s) => s.trim()).filter(Boolean);
29
+ return {
30
+ port: Number(env.LB_PORT || 8080),
31
+ api: list(env.LB_API),
32
+ realtime: list(env.LB_REALTIME),
33
+ realtimePaths: list(env.LB_REALTIME_PATHS || '/socket.io'),
34
+ algo: env.LB_ALGO || 'p2c-ewma',
35
+ healthPath: env.LB_HEALTH_PATH || '/healthz',
36
+ hcIntervalMs: Number(env.LB_HC_INTERVAL_MS || 1000),
37
+ hcTimeoutMs: Number(env.LB_HC_TIMEOUT_MS || 800),
38
+ headerTimeoutMs: Number(env.LB_HEADER_TIMEOUT_MS || 30000),
39
+ maxRetries: Number(env.LB_MAX_RETRIES || 2),
40
+ // Behind a tunnel or another proxy that sets the real client IP in a header. Enable only when EVERY
41
+ // connection arrives through it, or clients could spoof their IP.
42
+ clientIpHeader: (env.LB_CLIENT_IP_HEADER || '').toLowerCase(),
43
+ };
44
+ }
45
+
46
+ function createLoadBalancer(options, { log = console.log } = {}) {
47
+ const cfg = { ...readConfig({}), ...options };
48
+ const pools = { api: createPool('api', cfg.api, { algo: cfg.algo }) };
49
+ if (cfg.realtime.length) pools.realtime = createPool('realtime', cfg.realtime, { mode: 'failover' });
50
+
51
+ const agents = new Map(); // backend → its own keep-alive http.Agent
52
+ const agentOf = (b) => {
53
+ if (!agents.has(b)) agents.set(b, new http.Agent({ keepAlive: true, maxSockets: 512, timeout: 30000 }));
54
+ return agents.get(b);
55
+ };
56
+ const isRealtime = (url) => cfg.realtimePaths.some((p) => url === p || url.startsWith(`${p}/`) || url.startsWith(`${p}?`));
57
+ const poolFor = (url) => (pools.realtime && isRealtime(url) ? pools.realtime : pools.api);
58
+ const clientIp = (req) => (cfg.clientIpHeader && req.headers[cfg.clientIpHeader]) || req.socket.remoteAddress;
59
+
60
+ function forwardHeaders(req) {
61
+ const h = { ...req.headers };
62
+ for (const k of HOP_BY_HOP) delete h[k];
63
+ const prior = req.headers['x-forwarded-for'];
64
+ const ip = clientIp(req);
65
+ h['x-forwarded-for'] = cfg.clientIpHeader ? ip : prior ? `${prior}, ${ip}` : ip;
66
+ h['x-forwarded-proto'] = req.headers['x-forwarded-proto'] || (req.socket.encrypted ? 'https' : 'http');
67
+ h['x-forwarded-host'] = req.headers['x-forwarded-host'] || req.headers.host || '';
68
+ h['x-request-id'] = req.headers['x-request-id'] || crypto.randomUUID();
69
+ return h;
70
+ }
71
+
72
+ // ---------- HTTP ----------
73
+ async function handle(req, res) {
74
+ if (req.url === '/__lb/stats' && LOOPBACK.includes(req.socket.remoteAddress)) {
75
+ res.setHeader('Content-Type', 'application/json');
76
+ return res.end(JSON.stringify(stats(), null, 2));
77
+ }
78
+ const pool = poolFor(req.url);
79
+ const headers = forwardHeaders(req);
80
+
81
+ // A small body is buffered so it can be resent on retry; a big one is streamed (single attempt).
82
+ let body = null;
83
+ const len = Number(req.headers['content-length'] || 0);
84
+ const hasBody = len > 0 || req.headers['transfer-encoding'];
85
+ if (hasBody && len > 0 && len <= MAX_BUFFERED_BODY) {
86
+ const chunks = [];
87
+ for await (const c of req) chunks.push(c);
88
+ body = Buffer.concat(chunks);
89
+ }
90
+ const streamBody = hasBody && !body;
91
+
92
+ const tried = new Set();
93
+ for (let attempt = 0; ; attempt += 1) {
94
+ const b = pick(pool, { exclude: tried });
95
+ if (!b) {
96
+ if (!res.headersSent) res.writeHead(503, { 'Content-Type': 'application/json', 'Retry-After': '1' });
97
+ return res.end(JSON.stringify({ message: 'Server busy, please try again shortly' }));
98
+ }
99
+ tried.add(b);
100
+ const outcome = await tryBackend(pool, b, req, res, headers, body, streamBody);
101
+ if (outcome === 'done') return;
102
+ const retryable = outcome === 'refused' || (IDEMPOTENT.has(req.method) && !streamBody);
103
+ if (!retryable || attempt >= cfg.maxRetries) {
104
+ if (!res.headersSent) res.writeHead(outcome === 'timeout' ? 504 : 502, { 'Content-Type': 'application/json' });
105
+ return res.end(JSON.stringify({ message: 'Could not connect to server' }));
106
+ }
107
+ res.setHeader('X-LB-Retried', String(attempt + 1));
108
+ }
109
+ }
110
+
111
+ // One attempt → 'done' (response sent to the client) | 'refused' (never reached the backend) |
112
+ // 'failed' (broke or 502/503 before a response) | 'timeout'.
113
+ function tryBackend(pool, b, req, res, headers, body, streamBody) {
114
+ return new Promise((resolve) => {
115
+ const started = process.hrtime.bigint();
116
+ b.inflight += 1;
117
+ b.requests += 1;
118
+ let settled = false;
119
+ const finish = (v) => { if (!settled) { settled = true; resolve(v); } };
120
+ let released = false;
121
+ const release = () => { if (!released) { released = true; b.inflight -= 1; } };
122
+
123
+ const up = http.request({
124
+ host: b.host, port: b.port, method: req.method, path: req.url, headers, agent: agentOf(b),
125
+ });
126
+ const timer = setTimeout(() => { up.destroy(new Error('header timeout')); }, cfg.headerTimeoutMs);
127
+
128
+ up.on('response', (ur) => {
129
+ clearTimeout(timer);
130
+ recordLatency(b, Number(process.hrtime.bigint() - started) / 1e6);
131
+ // Backend down or draining (502/503): for a repeatable request, try another backend instead.
132
+ if ((ur.statusCode === 502 || ur.statusCode === 503) && IDEMPOTENT.has(req.method) && !streamBody) {
133
+ ur.resume();
134
+ release();
135
+ recordFailure(pool, b);
136
+ return finish('failed');
137
+ }
138
+ recordSuccess(b);
139
+ const out = { ...ur.headers };
140
+ for (const k of HOP_BY_HOP) delete out[k];
141
+ res.writeHead(ur.statusCode, out);
142
+ ur.pipe(res);
143
+ // A media stream stays open for minutes: it counts as in flight until the response really ends —
144
+ // exactly what least-conn and p2c need to know.
145
+ res.on('close', () => { release(); if (!ur.complete) ur.destroy(); });
146
+ finish('done');
147
+ });
148
+
149
+ up.on('error', (err) => {
150
+ clearTimeout(timer);
151
+ release();
152
+ if (settled) return; // failed after the response started: the client sees a broken connection
153
+ recordFailure(pool, b);
154
+ const code = err.code;
155
+ if (err.message === 'header timeout') return finish('timeout');
156
+ finish(code === 'ECONNREFUSED' || code === 'EHOSTUNREACH' || code === 'ENOTFOUND' ? 'refused' : 'failed');
157
+ });
158
+
159
+ // Client left midway (seek, skip): cancel the backend request too.
160
+ res.on('close', () => { if (!res.writableFinished) up.destroy(); });
161
+
162
+ if (streamBody) req.pipe(up);
163
+ else up.end(body || undefined);
164
+ });
165
+ }
166
+
167
+ // ---------- WebSocket (Upgrade) ----------
168
+ // Upgraded sockets leave the HTTP server (closeAllConnections cannot see them), so they are tracked to close them.
169
+ const tunnels = new Set();
170
+ function upgrade(req, socket, head) {
171
+ const pool = poolFor(req.url);
172
+ const b = pick(pool);
173
+ if (!b) return socket.end('HTTP/1.1 503 Service Unavailable\r\n\r\n');
174
+ const headers = forwardHeaders(req);
175
+ headers.connection = 'Upgrade';
176
+ headers.upgrade = req.headers.upgrade;
177
+ let connected = false;
178
+ const upstream = net.connect(b.port, b.host, () => {
179
+ connected = true;
180
+ b.inflight += 1;
181
+ b.requests += 1;
182
+ recordSuccess(b);
183
+ const lines = [`${req.method} ${req.url} HTTP/1.1`, ...Object.entries(headers).map(([k, v]) => `${k}: ${v}`), '', ''];
184
+ upstream.write(lines.join('\r\n'));
185
+ if (head?.length) upstream.write(head);
186
+ upstream.pipe(socket).pipe(upstream);
187
+ });
188
+ let closed = false;
189
+ const tunnel = [socket, upstream];
190
+ tunnels.add(tunnel);
191
+ const close = () => {
192
+ if (closed) return;
193
+ closed = true;
194
+ tunnels.delete(tunnel);
195
+ if (connected) b.inflight -= 1;
196
+ socket.destroy();
197
+ upstream.destroy();
198
+ };
199
+ upstream.on('error', (e) => { if (e.code === 'ECONNREFUSED') recordFailure(pool, b); close(); });
200
+ socket.on('error', close);
201
+ upstream.on('close', close);
202
+ socket.on('close', close);
203
+ }
204
+
205
+ // ---------- Active health checks ----------
206
+ function checkOnce(pool, b) {
207
+ const req = http.get({ host: b.host, port: b.port, path: cfg.healthPath, timeout: cfg.hcTimeoutMs, agent: false }, (r) => {
208
+ r.resume();
209
+ report(pool, b, r.statusCode === 200);
210
+ });
211
+ req.on('timeout', () => req.destroy(new Error('timeout')));
212
+ req.on('error', () => report(pool, b, false));
213
+ }
214
+ function report(pool, b, ok) {
215
+ const change = recordHealth(pool, b, ok);
216
+ if (change) log(`[lb] ${pool.name} ${b.addr} ${change.toUpperCase()}`);
217
+ }
218
+ let hcTimer = null;
219
+ const startHealthChecks = () => {
220
+ const tick = () => { for (const pool of Object.values(pools)) for (const b of pool.backends) checkOnce(pool, b); };
221
+ tick();
222
+ hcTimer = setInterval(tick, cfg.hcIntervalMs);
223
+ };
224
+
225
+ function stats() {
226
+ const now = Date.now();
227
+ return Object.fromEntries(Object.values(pools).map((p) => [p.name, {
228
+ mode: p.mode, algo: p.mode === 'failover' ? 'failover' : p.algo,
229
+ backends: p.backends.map((b) => ({
230
+ addr: b.addr, up: b.up, ejected: b.ejectedUntil > now, inflight: b.inflight,
231
+ ewmaMs: Math.round(b.ewmaMs * 10) / 10, requests: b.requests, failures: b.failures,
232
+ })),
233
+ }]));
234
+ }
235
+
236
+ const server = http.createServer((req, res) => {
237
+ handle(req, res).catch((e) => {
238
+ log(`[lb] ${e.message}`);
239
+ if (!res.headersSent) res.writeHead(500);
240
+ res.end();
241
+ });
242
+ });
243
+ server.on('upgrade', upgrade);
244
+ server.keepAliveTimeout = 65_000;
245
+ server.headersTimeout = 66_000;
246
+
247
+ return {
248
+ server, pools, stats,
249
+ listen: (port = cfg.port) => new Promise((r) => server.listen(port, () => { startHealthChecks(); r(server.address().port); })),
250
+ close: () => new Promise((r) => {
251
+ clearInterval(hcTimer);
252
+ for (const a of agents.values()) a.destroy();
253
+ for (const t of tunnels) for (const s of t) s.destroy();
254
+ server.closeAllConnections();
255
+ server.close(r);
256
+ }),
257
+ };
258
+ }
259
+
260
+ module.exports = { createLoadBalancer, readConfig };