@rikcodes/teamclaude 1.1.20-rik.7 → 1.1.20-rik.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/crash-log.js +36 -4
- package/src/index.js +8 -2
- package/src/server.js +12 -0
- package/src/sidecar.js +111 -4
- package/src/tui.js +14 -8
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@rikcodes/teamclaude",
|
|
3
|
-
"version": "1.1.20-rik.
|
|
3
|
+
"version": "1.1.20-rik.9",
|
|
4
4
|
"description": "Multi-account proxy for Claude Code and Codex: pools Claude Max, ChatGPT/Codex, API-key and third-party backend accounts, and rotates on quota",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "src/index.js",
|
package/src/crash-log.js
CHANGED
|
@@ -1,5 +1,14 @@
|
|
|
1
1
|
import { appendFileSync } from 'node:fs';
|
|
2
2
|
|
|
3
|
+
/** Did a write fail because whatever was reading has gone?
|
|
4
|
+
*
|
|
5
|
+
* Node reports this asynchronously when the stream is non-blocking, so it
|
|
6
|
+
* arrives as an event rather than at the call site and no `try` around the
|
|
7
|
+
* write can catch it. */
|
|
8
|
+
function isBrokenPipe(err) {
|
|
9
|
+
return err?.code === 'EPIPE' || err?.code === 'ERR_STREAM_DESTROYED';
|
|
10
|
+
}
|
|
11
|
+
|
|
3
12
|
/**
|
|
4
13
|
* Write a fatal error to `path` before the process dies.
|
|
5
14
|
*
|
|
@@ -11,16 +20,39 @@ import { appendFileSync } from 'node:fs';
|
|
|
11
20
|
* Handling these events replaces Node's own behaviour, so this must do what
|
|
12
21
|
* Node would: report and exit non-zero. Continuing after an uncaught exception
|
|
13
22
|
* would leave the proxy running on unknown state.
|
|
23
|
+
*
|
|
24
|
+
* A broken pipe is the one exception, and it is deliberate. The terminal going
|
|
25
|
+
* away says nothing about the proxy's state — every account, route and inflight
|
|
26
|
+
* request is exactly as it was — so ending the process punishes every routed
|
|
27
|
+
* session for a closed pane. It happened three times in two days here: the TUI
|
|
28
|
+
* writes to a non-blocking stdout, so the failure could not be caught where it
|
|
29
|
+
* was written, and each death also orphaned a sidecar on its port. Recorded
|
|
30
|
+
* once, because a stream dying is worth knowing, then survived. Once is enough:
|
|
31
|
+
* the stream stays dead, and a storm of them would say nothing new.
|
|
14
32
|
* @param {string} path
|
|
15
33
|
* @param {{ exit?: (code: number) => void, log?: { write: (s: string) => void } }} [opts]
|
|
16
34
|
*/
|
|
17
35
|
export function installCrashHandlers(path, { exit = process.exit, log = process.stderr } = {}) {
|
|
36
|
+
let brokenPipeRecorded = false;
|
|
37
|
+
// 0600: a stack can carry request context. A write failure (read-only home,
|
|
38
|
+
// full disk) must not mask the crash itself — stderr still gets the entry.
|
|
39
|
+
const record = (/** @type {string} */ entry) => {
|
|
40
|
+
try { appendFileSync(path, entry, { mode: 0o600 }); } catch { /* report to stderr regardless */ }
|
|
41
|
+
};
|
|
18
42
|
const report = (/** @type {string} */ kind) => (/** @type {any} */ err) => {
|
|
19
43
|
const stack = err?.stack || String(err);
|
|
20
|
-
|
|
21
|
-
//
|
|
22
|
-
//
|
|
23
|
-
|
|
44
|
+
// A bare "Error: write EPIPE" names neither the stream nor the call. The
|
|
45
|
+
// code and syscall are what turn the next one of these into a diagnosis
|
|
46
|
+
// rather than an afternoon of inference.
|
|
47
|
+
const detail = [err?.code, err?.syscall].filter(Boolean).join(' ');
|
|
48
|
+
const entry = `\n=== ${new Date().toISOString()} ${kind}${detail ? ` (${detail})` : ''} ===\n${stack}\n`;
|
|
49
|
+
if (isBrokenPipe(err)) {
|
|
50
|
+
if (!brokenPipeRecorded) { brokenPipeRecorded = true; record(entry); }
|
|
51
|
+
// Deliberately not written to `log`: that is very likely the stream that
|
|
52
|
+
// just failed, and would fail again.
|
|
53
|
+
return;
|
|
54
|
+
}
|
|
55
|
+
record(entry);
|
|
24
56
|
log.write(entry);
|
|
25
57
|
exit(1);
|
|
26
58
|
};
|
package/src/index.js
CHANGED
|
@@ -323,7 +323,7 @@ async function serverCommand() {
|
|
|
323
323
|
|
|
324
324
|
// Periodically persist quota (and once more on shutdown) to the state file.
|
|
325
325
|
const persistQuotaState = () =>
|
|
326
|
-
saveState({ quota: accountManager.exportQuotaState(), clients: clientUsage.export(), usageDimensions: dimensionUsage.export() })
|
|
326
|
+
saveState({ quota: accountManager.exportQuotaState(), clients: clientUsage.export(), usageDimensions: dimensionUsage.export(), sidecars: sidecar?.exportPids() || savedState?.sidecars || {} })
|
|
327
327
|
.catch(err => console.error(`[TeamClaude] Failed to save quota state: ${err.message}`));
|
|
328
328
|
let quotaSaveInterval = null;
|
|
329
329
|
|
|
@@ -709,7 +709,13 @@ async function serverCommand() {
|
|
|
709
709
|
warmer.start();
|
|
710
710
|
|
|
711
711
|
// Launch supervised sidecars (no-op when config.sidecars is empty).
|
|
712
|
-
sidecar = new Sidecar(config.sidecars
|
|
712
|
+
sidecar = new Sidecar(config.sidecars, {
|
|
713
|
+
// A sidecar outlives a server that was killed rather than asked to stop,
|
|
714
|
+
// and goes on holding its port. The pid recorded here is the only thing
|
|
715
|
+
// that lets the next start tell its own leftover from a stranger's process.
|
|
716
|
+
savedPids: savedState?.sidecars || null,
|
|
717
|
+
onPids: () => { persistQuotaState(); },
|
|
718
|
+
});
|
|
713
719
|
sidecar.start();
|
|
714
720
|
|
|
715
721
|
// Background self-update for a backgrounded (headless) server. Skipped under
|
package/src/server.js
CHANGED
|
@@ -63,6 +63,17 @@ const RATE_LIMIT_ABSORB_MAX_SECONDS =
|
|
|
63
63
|
Number(process.env.TEAMCLAUDE_RATE_LIMIT_ABSORB_MAX_SECONDS) || 60;
|
|
64
64
|
const OAUTH_ENTITLEMENT_ERROR_CODE = 'oauth_not_allowed_for_organization';
|
|
65
65
|
const ERROR_BODY_INSPECTION_LIMIT = 64 * 1024;
|
|
66
|
+
// How long an idle keep-alive connection is held open.
|
|
67
|
+
//
|
|
68
|
+
// Node's default is 5s, but a client's connection pool may hold the same socket
|
|
69
|
+
// far longer, and whoever closes first wins: when the server does, the client
|
|
70
|
+
// finds out only by writing to a socket that is already gone, which surfaces as
|
|
71
|
+
// a request that fails in ~130ms with no upstream involvement. The Codex
|
|
72
|
+
// sidecar is such a client — reqwest's pool_idle_timeout defaults to 90s and it
|
|
73
|
+
// never overrides it — so outlive the longest pool and let the client always be
|
|
74
|
+
// the one to close. headersTimeout bounds an in-progress request's headers, not
|
|
75
|
+
// the idle gap between them (measured), so it is deliberately left alone.
|
|
76
|
+
const KEEP_ALIVE_TIMEOUT_MS = 120_000;
|
|
66
77
|
|
|
67
78
|
/** Classify only the structured organization-policy denial observed upstream.
|
|
68
79
|
* Message text and generic permission errors are deliberately not enough. */
|
|
@@ -476,6 +487,7 @@ export function createProxyServer(accountManager, config, hooks = {}, sx = null,
|
|
|
476
487
|
const egress = createEgressGuard(config, console.error);
|
|
477
488
|
const forward = createProxyRequestListener({ accountManager, upstream, logDir, hooks, sx, holdMs, config, egress, clientUsage, dimensionUsage });
|
|
478
489
|
const server = http.createServer(requestHandler);
|
|
490
|
+
server.keepAliveTimeout = KEEP_ALIVE_TIMEOUT_MS;
|
|
479
491
|
|
|
480
492
|
// What bounds a directory of one-shot dumps is deleting the expired ones, not
|
|
481
493
|
// rotating a growing file. Swept once at startup, because a backlog is usually
|
package/src/sidecar.js
CHANGED
|
@@ -11,8 +11,14 @@
|
|
|
11
11
|
// stdout is ignored (sidecars keep their own log files); stderr's last few
|
|
12
12
|
// lines are kept in a ring buffer so `getStatus()` can say WHY a sidecar is
|
|
13
13
|
// crash-looping without anyone hunting for its logs.
|
|
14
|
+
//
|
|
15
|
+
// One failure mode earned its own handling. A sidecar binds a fixed port, so a
|
|
16
|
+
// copy that outlives its server keeps that port and every later server fails to
|
|
17
|
+
// bind — reported, before this, as a bare "code 1" retried forever, while the
|
|
18
|
+
// old process quietly went on serving. Two halves: name the conflict instead of
|
|
19
|
+
// guessing at it, and reap a leftover this server can prove is its own.
|
|
14
20
|
|
|
15
|
-
import { spawn } from 'node:child_process';
|
|
21
|
+
import { spawn, spawnSync } from 'node:child_process';
|
|
16
22
|
|
|
17
23
|
/** Delay before restart attempt N (0-based): base, doubled per consecutive
|
|
18
24
|
* crash, capped. Pure so the schedule is testable without timers. */
|
|
@@ -20,6 +26,13 @@ export function restartDelayMs(restarts, { baseRestartMs, maxRestartMs }) {
|
|
|
20
26
|
return Math.min(baseRestartMs * 2 ** restarts, maxRestartMs);
|
|
21
27
|
}
|
|
22
28
|
|
|
29
|
+
/** Does this stderr line say the port is already taken? The wording is the
|
|
30
|
+
* runtime's, not ours — Rust prints the OS string, Node prints EADDRINUSE —
|
|
31
|
+
* so match the phrasings rather than one library's spelling. */
|
|
32
|
+
export function isBindConflict(line) {
|
|
33
|
+
return /address already in use|address in use|EADDRINUSE/i.test(String(line));
|
|
34
|
+
}
|
|
35
|
+
|
|
23
36
|
export class Sidecar {
|
|
24
37
|
constructor(entries, {
|
|
25
38
|
spawnFn = defaultSpawn,
|
|
@@ -28,6 +41,10 @@ export class Sidecar {
|
|
|
28
41
|
stableMs = 30_000,
|
|
29
42
|
stderrTailLines = 20,
|
|
30
43
|
log = console.log,
|
|
44
|
+
savedPids = null,
|
|
45
|
+
onPids = null,
|
|
46
|
+
readProcess = defaultReadProcess,
|
|
47
|
+
killFn = (pid, signal) => process.kill(pid, signal),
|
|
31
48
|
} = {}) {
|
|
32
49
|
this.entries = Array.isArray(entries) ? entries : [];
|
|
33
50
|
this.spawnFn = spawnFn;
|
|
@@ -36,6 +53,12 @@ export class Sidecar {
|
|
|
36
53
|
this.stableMs = stableMs;
|
|
37
54
|
this.stderrTailLines = stderrTailLines;
|
|
38
55
|
this.log = log;
|
|
56
|
+
// Pids this server recorded on a previous run, by entry name. Only ever
|
|
57
|
+
// used to recognise our own leftovers; see _reapOrphan.
|
|
58
|
+
this.savedPids = (savedPids && typeof savedPids === 'object') ? { ...savedPids } : {};
|
|
59
|
+
this.onPids = onPids;
|
|
60
|
+
this.readProcess = readProcess;
|
|
61
|
+
this.killFn = killFn;
|
|
39
62
|
this.stopping = false;
|
|
40
63
|
// Per-entry runtime state, keyed by entry (parallel array to this.entries).
|
|
41
64
|
this.states = this.entries.map(entry => ({
|
|
@@ -46,11 +69,15 @@ export class Sidecar {
|
|
|
46
69
|
lastExit: null,
|
|
47
70
|
timer: null,
|
|
48
71
|
stderrTail: [],
|
|
72
|
+
blocked: null, // stderr line proving the port is held, else null
|
|
49
73
|
}));
|
|
50
74
|
}
|
|
51
75
|
|
|
52
76
|
start() {
|
|
53
|
-
for (const state of this.states)
|
|
77
|
+
for (const state of this.states) {
|
|
78
|
+
this._reapOrphan(state);
|
|
79
|
+
this._spawn(state);
|
|
80
|
+
}
|
|
54
81
|
}
|
|
55
82
|
|
|
56
83
|
stop() {
|
|
@@ -68,10 +95,51 @@ export class Sidecar {
|
|
|
68
95
|
pid: state.child?.pid ?? null,
|
|
69
96
|
restarts: state.restarts,
|
|
70
97
|
lastExit: state.lastExit,
|
|
98
|
+
// A held port is not a crash: it says nothing is wrong with the binary
|
|
99
|
+
// and everything is wrong with the port, which is a different fix.
|
|
100
|
+
blocked: !!state.blocked,
|
|
101
|
+
blockedReason: state.blocked,
|
|
71
102
|
stderrTail: [...state.stderrTail],
|
|
72
103
|
}));
|
|
73
104
|
}
|
|
74
105
|
|
|
106
|
+
/** Last known pid per entry name, for the owner to persist.
|
|
107
|
+
*
|
|
108
|
+
* Deliberately the last pid rather than only a live one. A sidecar that is
|
|
109
|
+
* down — crashed, or blocked because something else holds its port — is
|
|
110
|
+
* exactly when the pid is worth keeping, and a record that emptied itself
|
|
111
|
+
* the moment the child exited would be absent whenever it was needed.
|
|
112
|
+
*/
|
|
113
|
+
exportPids() {
|
|
114
|
+
return { ...this.savedPids };
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/** Kill a sidecar this server started that outlived it.
|
|
118
|
+
*
|
|
119
|
+
* A server killed with SIGKILL never reaches stop(), and the child does not
|
|
120
|
+
* die with it (see defaultSpawn), so the sidecar keeps running and holding
|
|
121
|
+
* its port. Three things must all hold before anything is signalled: the pid
|
|
122
|
+
* was recorded by us, it is now reparented to init (nothing else supervises
|
|
123
|
+
* it), and it is still running the same program. A recycled pid fails the
|
|
124
|
+
* last test, and a sidecar belonging to another live server fails the second,
|
|
125
|
+
* so neither is touched. The port may take a moment to free after this; the
|
|
126
|
+
* ordinary restart backoff covers that.
|
|
127
|
+
*/
|
|
128
|
+
_reapOrphan(state) {
|
|
129
|
+
const pid = Number(this.savedPids[state.entry.name]?.pid);
|
|
130
|
+
if (!Number.isInteger(pid) || pid <= 1) return;
|
|
131
|
+
let info = null;
|
|
132
|
+
try { info = this.readProcess(pid); } catch { return; }
|
|
133
|
+
if (!info) return; // not running: already gone
|
|
134
|
+
if (info.ppid !== 1) return; // still has a parent, so not ours to reap
|
|
135
|
+
const program = state.entry.command?.[0];
|
|
136
|
+
if (!program || !String(info.command || '').includes(program)) return; // pid recycled
|
|
137
|
+
try {
|
|
138
|
+
this.killFn(pid, 'SIGTERM');
|
|
139
|
+
this.log(`[TeamClaude] Sidecar "${state.entry.name}": reaped orphan pid ${pid} left by a previous run`);
|
|
140
|
+
} catch { /* exited between the check and the signal, which is the goal anyway */ }
|
|
141
|
+
}
|
|
142
|
+
|
|
75
143
|
_spawn(state) {
|
|
76
144
|
const { entry } = state;
|
|
77
145
|
const [command, ...args] = entry.command;
|
|
@@ -89,6 +157,12 @@ export class Sidecar {
|
|
|
89
157
|
}
|
|
90
158
|
state.child = child;
|
|
91
159
|
state.startedAt = Date.now();
|
|
160
|
+
state.blocked = null; // a fresh attempt: whatever the last one hit is history
|
|
161
|
+
// Recorded now, while the pid is known. The record has to outlive the child
|
|
162
|
+
// itself: the case it exists for is this server being killed outright, and
|
|
163
|
+
// by then there is nobody left to write anything down.
|
|
164
|
+
this.savedPids[entry.name] = { pid: child.pid, command: entry.command?.[0] ?? null };
|
|
165
|
+
this._publishPids();
|
|
92
166
|
child.stderr?.on('data', (chunk) => this._recordStderr(state, chunk));
|
|
93
167
|
child.once('error', (err) => {
|
|
94
168
|
if (state.child !== child) return;
|
|
@@ -108,7 +182,12 @@ export class Sidecar {
|
|
|
108
182
|
state.lastExit = lastExit;
|
|
109
183
|
if (this.stopping) return;
|
|
110
184
|
const delay = restartDelayMs(state.restarts, this);
|
|
111
|
-
|
|
185
|
+
// Retrying still makes sense while blocked — a port held by a process we
|
|
186
|
+
// did not start frees when that process ends, and this is the only thing
|
|
187
|
+
// watching for it — but saying "down (code 1)" about it does not.
|
|
188
|
+
this.log(state.blocked
|
|
189
|
+
? `[TeamClaude] Sidecar "${state.entry.name}" cannot bind: ${state.blocked}; retrying in ${Math.round(delay / 1000)}s`
|
|
190
|
+
: `[TeamClaude] Sidecar "${state.entry.name}" down (${lastExit}); restarting in ${Math.round(delay / 1000)}s`);
|
|
112
191
|
state.restarts += 1;
|
|
113
192
|
state.timer = setTimeout(() => {
|
|
114
193
|
state.timer = null;
|
|
@@ -117,8 +196,16 @@ export class Sidecar {
|
|
|
117
196
|
state.timer.unref?.();
|
|
118
197
|
}
|
|
119
198
|
|
|
199
|
+
_publishPids() {
|
|
200
|
+
if (!this.onPids) return;
|
|
201
|
+
// Persistence is the owner's business and best-effort: failing to record a
|
|
202
|
+
// pid costs a manual reap later, never this start.
|
|
203
|
+
try { this.onPids(this.exportPids()); } catch { /* not worth failing a spawn over */ }
|
|
204
|
+
}
|
|
205
|
+
|
|
120
206
|
_recordStderr(state, chunk) {
|
|
121
207
|
const lines = String(chunk).split('\n').map(s => s.trim()).filter(Boolean);
|
|
208
|
+
for (const line of lines) if (isBindConflict(line)) state.blocked = line;
|
|
122
209
|
state.stderrTail.push(...lines);
|
|
123
210
|
if (state.stderrTail.length > this.stderrTailLines) {
|
|
124
211
|
state.stderrTail.splice(0, state.stderrTail.length - this.stderrTailLines);
|
|
@@ -127,8 +214,28 @@ export class Sidecar {
|
|
|
127
214
|
}
|
|
128
215
|
|
|
129
216
|
// Real spawner: stdout ignored (sidecars log to their own files), stderr piped
|
|
130
|
-
// for the ring buffer.
|
|
217
|
+
// for the ring buffer.
|
|
218
|
+
//
|
|
219
|
+
// The child stays in our process group (`detached` defaults to false), so an
|
|
220
|
+
// interactive ctrl-c reaches it too. That is NOT the same as dying with us:
|
|
221
|
+
// there is no PDEATHSIG on macOS, and a server killed with SIGKILL leaves the
|
|
222
|
+
// sidecar running and holding its port. stop() is the ordinary path out;
|
|
223
|
+
// Sidecar._reapOrphan covers the rest.
|
|
131
224
|
/** @param {{name?: string, command: string, args: string[], env: Record<string, string|undefined>}} spec */
|
|
132
225
|
function defaultSpawn({ command, args, env }) {
|
|
133
226
|
return spawn(command, args, { env, stdio: ['ignore', 'ignore', 'pipe'] });
|
|
134
227
|
}
|
|
228
|
+
|
|
229
|
+
/** Parent pid and command line for a pid, or null when it is not running.
|
|
230
|
+
*
|
|
231
|
+
* `ps` rather than /proc, which macOS does not have. This runs once per
|
|
232
|
+
* sidecar at startup, never on a request path.
|
|
233
|
+
* @param {number} pid
|
|
234
|
+
* @returns {{ppid: number, command: string}|null}
|
|
235
|
+
*/
|
|
236
|
+
function defaultReadProcess(pid) {
|
|
237
|
+
const r = spawnSync('ps', ['-o', 'ppid=,command=', '-p', String(pid)], { encoding: 'utf8' });
|
|
238
|
+
const out = r.status === 0 ? String(r.stdout || '').trim() : '';
|
|
239
|
+
const m = out.match(/^(\d+)\s+(.*)$/s);
|
|
240
|
+
return m ? { ppid: Number(m[1]), command: m[2].trim() } : null;
|
|
241
|
+
}
|
package/src/tui.js
CHANGED
|
@@ -586,13 +586,16 @@ export class TUI {
|
|
|
586
586
|
this._setStdoutBlocking(true);
|
|
587
587
|
// Blocking again means a failed write throws here instead of arriving as
|
|
588
588
|
// an event, and a terminal that has already gone will fail. Restoring the
|
|
589
|
-
// screen is best-effort: there is nobody left to restore it for.
|
|
590
|
-
// listener outlives this write, so a late async failure is absorbed too.
|
|
589
|
+
// screen is best-effort: there is nobody left to restore it for.
|
|
591
590
|
try { process.stdout.write(`${ESC}?25h${ESC}?1049l`); } catch { /* terminal already gone */ }
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
591
|
+
// The error listener stays. Flipping back to blocking does not make writes
|
|
592
|
+
// ALREADY QUEUED synchronous, so a paint still in flight can fail after
|
|
593
|
+
// this point, and shutdown() runs well past it — stopping the prober, the
|
|
594
|
+
// warmer and the sidecar, then awaiting a state save. An earlier version
|
|
595
|
+
// removed the listener here while claiming it outlived the write; it did
|
|
596
|
+
// not, and the proxy died of an unhandled EPIPE in exactly that window. It
|
|
597
|
+
// only sets a flag, so leaving it attached for the rest of the process
|
|
598
|
+
// costs nothing.
|
|
596
599
|
try { process.stdin.setRawMode(false); } catch {}
|
|
597
600
|
process.stdin.pause();
|
|
598
601
|
}
|
|
@@ -1707,10 +1710,13 @@ export class TUI {
|
|
|
1707
1710
|
// Matched by name: a sidecars[] entry and the account that routes to it
|
|
1708
1711
|
// are named by the same operator, and nothing else pairs them.
|
|
1709
1712
|
const proc = sidecars.find(sc => sc.name === a.name) || null;
|
|
1713
|
+
// A held port reads as a crash loop but is not one: the binary is fine and
|
|
1714
|
+
// something else owns the address, which is a different thing to go and fix.
|
|
1710
1715
|
const state = a.disabled ? red('disabled')
|
|
1711
1716
|
: a.rateLimitedUntil > Date.now() ? yellow('throttled')
|
|
1712
|
-
: proc
|
|
1713
|
-
: proc ?
|
|
1717
|
+
: proc?.blocked ? red('port in use')
|
|
1718
|
+
: proc && !proc.running ? red(`down (${proc.lastExit || 'restarting'})`)
|
|
1719
|
+
: proc ? green('up') : green('ok');
|
|
1714
1720
|
const pid = proc?.running ? dim(` pid ${proc.pid}`) : '';
|
|
1715
1721
|
const restarts = proc?.restarts ? yellow(` ${proc.restarts} restarts`) : '';
|
|
1716
1722
|
return ` ${dim('⚙')} ${a.name} ${dim('→')} ${dim(host)} ${state}${pid}${restarts}`;
|