@metamynd/agentsafe-signer 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +382 -0
- package/cli.mjs +213 -0
- package/daemon-client.mjs +97 -0
- package/daemon.mjs +558 -0
- package/governance-envelope.mjs +69 -0
- package/kek-backends.mjs +192 -0
- package/keystore.mjs +108 -0
- package/log-anchor.mjs +112 -0
- package/log-checkpoint.mjs +191 -0
- package/merkle.mjs +68 -0
- package/migrate.mjs +120 -0
- package/package.json +47 -0
- package/policy-core.mjs +601 -0
- package/secure-memory.mjs +38 -0
- package/service-installer.mjs +244 -0
- package/windows-secure-pipe.mjs +367 -0
|
@@ -0,0 +1,244 @@
|
|
|
1
|
+
// service-installer.mjs — generates the OS service unit files from
|
|
2
|
+
// docs/design/agent-key-custody-local-signer-daemon-plan.md's "OS service unit templates"
|
|
3
|
+
// section. Deliberately writes files ONLY to a local output directory the caller specifies —
|
|
4
|
+
// never to a real system location (~/.config/systemd/user, /etc/systemd/system,
|
|
5
|
+
// ~/Library/LaunchAgents, the Windows service registry) and never invokes systemctl/launchctl/
|
|
6
|
+
// sc.exe itself. It prints the exact command an operator runs to actually install the result;
|
|
7
|
+
// running that command is always a separate, deliberate, human step. See README.md "Status" for
|
|
8
|
+
// why: registering a real system service is a genuine system modification, out of scope for what
|
|
9
|
+
// this module does on its own.
|
|
10
|
+
import path from 'node:path';
|
|
11
|
+
|
|
12
|
+
/** Deterministic, filesystem-safe short id for a logical identity (agentDid/serviceDid/name). */
|
|
13
|
+
function shortId(identity) {
|
|
14
|
+
// Not crypto-sensitive — just needs to be stable and safe in a unit filename/pipe name.
|
|
15
|
+
let h = 0;
|
|
16
|
+
for (let i = 0; i < identity.length; i++) h = (Math.imul(h, 31) + identity.charCodeAt(i)) | 0;
|
|
17
|
+
return (h >>> 0).toString(16).padStart(8, '0');
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/** The node that will actually exist at ExecStart= time. `/usr/bin/node` was the old default and
|
|
21
|
+
* is absent on any host using nvm/asdf/volta (verified: the real host this was first installed on
|
|
22
|
+
* had node only under ~/.nvm), which systemd reports as a bare 203/EXEC with no explanation of
|
|
23
|
+
* which binary was missing. process.execPath is the interpreter running the generator — a far
|
|
24
|
+
* better default, and still overridable via the nodePath option. */
|
|
25
|
+
function defaultNodePath() {
|
|
26
|
+
return process.execPath || '/usr/bin/node';
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
function assertTier(tier) {
|
|
30
|
+
if (tier !== 1 && tier !== 2) throw new Error(`tier must be 1 or 2, got ${tier}`);
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
// --- Linux: systemd, socket-activated -------------------------------------------------------
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* @param {{identity:string, tier:1|2, nodePath?:string, daemonPath:string, role?:'agent'|'service', cliPath?:string}} opts
|
|
37
|
+
* @returns {{socketUnit:string, serviceUnit:string, id:string, enableCommand:string}}
|
|
38
|
+
*/
|
|
39
|
+
export function generateSystemdUnits({ identity, tier, nodePath = defaultNodePath(), daemonPath, role = 'agent', cliPath }) {
|
|
40
|
+
assertTier(tier);
|
|
41
|
+
if (!identity || !daemonPath) throw new Error('generateSystemdUnits requires { identity, daemonPath }');
|
|
42
|
+
const id = shortId(identity);
|
|
43
|
+
|
|
44
|
+
const socketUnit = [
|
|
45
|
+
`# agentsafe-signer@.socket (instance: ${id})`,
|
|
46
|
+
'[Unit]',
|
|
47
|
+
`Description=AgentSafe local signer socket for identity ${id}`,
|
|
48
|
+
'',
|
|
49
|
+
'[Socket]',
|
|
50
|
+
'ListenStream=%t/agentsafe/signer-%i.sock',
|
|
51
|
+
'SocketMode=0600',
|
|
52
|
+
'DirectoryMode=0700',
|
|
53
|
+
'RemoveOnStop=true',
|
|
54
|
+
'',
|
|
55
|
+
'[Install]',
|
|
56
|
+
'WantedBy=sockets.target',
|
|
57
|
+
'',
|
|
58
|
+
].join('\n');
|
|
59
|
+
|
|
60
|
+
// Tier 1: %h-relative ExecStart (the real logged-in user's own home) is correct here because
|
|
61
|
+
// Tier 1 runs AS that user — see the design doc's own note on why Tier 2 cannot reuse this.
|
|
62
|
+
// ONLY for a relative daemonPath, though: `%h/` + an ABSOLUTE path produced a doubled,
|
|
63
|
+
// non-existent path (`/home/u` + `/home/u/...` -> `/home/u/home/u/...`), and the CLI's own
|
|
64
|
+
// documented example passes an absolute one. Verified by installing and starting the unit on a
|
|
65
|
+
// real systemd host, where the failure was visible in the ExecStart= line systemd reported back.
|
|
66
|
+
// ExecStart runs cli.mjs, not daemon.mjs. daemon.mjs is a library module with no argv handling
|
|
67
|
+
// at all: `node daemon.mjs` loads it, returns 0 and serves nothing, so a unit pointing there
|
|
68
|
+
// starts and instantly "succeeds" while never listening. Found by installing and starting the
|
|
69
|
+
// unit on a real systemd host. cli.mjs ships alongside daemon.mjs in this package's own `files`
|
|
70
|
+
// list, so deriving it from the same directory is stable, and `cliPath` overrides it outright.
|
|
71
|
+
if (role !== 'agent' && role !== 'service') throw new Error(`role must be 'agent' or 'service', got ${role}`);
|
|
72
|
+
const resolvedCliPath = cliPath ?? daemonPath.replace(/daemon\.mjs$/, 'cli.mjs');
|
|
73
|
+
const isAbsolutePath = resolvedCliPath.startsWith('/');
|
|
74
|
+
const cliInvocationPath = tier === 1 && !isAbsolutePath ? `%h/${resolvedCliPath.replace(/^\.?\//, '')}` : resolvedCliPath;
|
|
75
|
+
// %S is systemd's own state-directory root, so this lines up with StateDirectory= below rather
|
|
76
|
+
// than hard-coding a path that has to be kept in sync with it by hand.
|
|
77
|
+
const invocation = `${nodePath} ${cliInvocationPath} start --state-dir %S/agentsafe/%i --role ${role} --identity %i`;
|
|
78
|
+
// No setpriv wrapper. An earlier pass wrapped this in `setpriv --dumpable 0 --` to deny ptrace
|
|
79
|
+
// via PR_SET_DUMPABLE=0. That option does not exist: util-linux's setpriv has no --dumpable flag
|
|
80
|
+
// (checked on util-linux 2.39.3 / Ubuntu 24.04 LTS — the string is absent from the binary
|
|
81
|
+
// entirely), so the generated unit failed at exec with "setpriv: unrecognized option
|
|
82
|
+
// '--dumpable'" and the service never started at all, making every other directive here moot.
|
|
83
|
+
// Found by actually installing and starting the unit on a real systemd host; the previous
|
|
84
|
+
// service-installer.smoke.mjs assertion passed because it only matched the generated TEXT.
|
|
85
|
+
// Tier 1's ptrace posture is therefore what the design doc's threat model already says it is —
|
|
86
|
+
// T3 "partial", resting on the host's own kernel.yama.ptrace_scope — and Tier 2's dedicated
|
|
87
|
+
// DynamicUser UID remains the directive that actually mitigates T3. LimitCORE=0 below is
|
|
88
|
+
// unaffected and still covers the core-dump half declaratively.
|
|
89
|
+
const execStart = invocation;
|
|
90
|
+
|
|
91
|
+
const serviceLines = [
|
|
92
|
+
`# agentsafe-signer@.service (instance: ${id}, Tier ${tier})`,
|
|
93
|
+
'[Unit]',
|
|
94
|
+
`Description=AgentSafe local signer daemon for identity ${id}`,
|
|
95
|
+
'Requires=agentsafe-signer@%i.socket',
|
|
96
|
+
'',
|
|
97
|
+
'[Service]',
|
|
98
|
+
// Type=exec, not notify. Nothing in this package sends sd_notify READY=1, and systemd fails a
|
|
99
|
+
// Type=notify service with `Result: protocol` when the notification never arrives — even
|
|
100
|
+
// though node is healthy and listening. Observed on a real host. Type=exec still waits for a
|
|
101
|
+
// successful execve() before considering the unit started, which is the strongest readiness
|
|
102
|
+
// guarantee available without implementing the sd_notify protocol (a real feature, tracked as
|
|
103
|
+
// an open gap in README.md rather than smuggled into this generator).
|
|
104
|
+
'Type=exec',
|
|
105
|
+
`ExecStart=${execStart}`,
|
|
106
|
+
'StateDirectory=agentsafe/%i',
|
|
107
|
+
'NoNewPrivileges=true',
|
|
108
|
+
'ProtectSystem=strict',
|
|
109
|
+
'ProtectHome=read-only',
|
|
110
|
+
'PrivateTmp=true',
|
|
111
|
+
'PrivateDevices=true',
|
|
112
|
+
'PrivateNetwork=true',
|
|
113
|
+
'ProtectKernelTunables=true',
|
|
114
|
+
'ProtectKernelModules=true',
|
|
115
|
+
'ProtectKernelLogs=true',
|
|
116
|
+
'ProtectControlGroups=true',
|
|
117
|
+
'ProtectClock=true',
|
|
118
|
+
'ProtectHostname=true',
|
|
119
|
+
'RestrictNamespaces=true',
|
|
120
|
+
'RestrictRealtime=true',
|
|
121
|
+
'RestrictSUIDSGID=true',
|
|
122
|
+
'LockPersonality=true',
|
|
123
|
+
// NOT MemoryDenyWriteExecute=true. V8 is a JIT: it needs W->X memory transitions, and this
|
|
124
|
+
// directive makes mprotect(PROT_EXEC) fail with ENOMEM inside v8::base::OS::SetPermissions
|
|
125
|
+
// during Isolate init, so node dies with SIGTRAP ("Check failed: 12 == errno") before running
|
|
126
|
+
// a single line of daemon code. Observed on a real systemd host, not inferred — the service
|
|
127
|
+
// failed with Result: core-dump until this was removed. No Node service can set it.
|
|
128
|
+
'CapabilityBoundingSet=',
|
|
129
|
+
'AmbientCapabilities=',
|
|
130
|
+
'UMask=0077',
|
|
131
|
+
'LimitCORE=0',
|
|
132
|
+
'SystemCallFilter=@system-service',
|
|
133
|
+
'SystemCallErrorNumber=EPERM',
|
|
134
|
+
'SystemCallArchitectures=native',
|
|
135
|
+
];
|
|
136
|
+
// Tier 2 only: a dedicated dynamic UID makes RemoveIPC safe (see the design doc's own
|
|
137
|
+
// reasoning for why it's unsafe at Tier 1, where the daemon shares the agent's UID).
|
|
138
|
+
if (tier === 2) serviceLines.push('DynamicUser=true', 'RemoveIPC=true');
|
|
139
|
+
serviceLines.push('', '[Install]', 'WantedBy=default.target', '');
|
|
140
|
+
|
|
141
|
+
const enableCommand =
|
|
142
|
+
tier === 1
|
|
143
|
+
? `systemctl --user daemon-reload && systemctl --user enable --now agentsafe-signer@${id}.socket`
|
|
144
|
+
: `sudo systemctl daemon-reload && sudo systemctl enable --now agentsafe-signer@${id}.socket`;
|
|
145
|
+
|
|
146
|
+
return { socketUnit, serviceUnit: serviceLines.join('\n'), id, enableCommand };
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
// --- macOS: launchd, socket-activated --------------------------------------------------------
|
|
150
|
+
|
|
151
|
+
/**
|
|
152
|
+
* @param {{identity:string, tier:1|2, nodePath?:string, daemonPath:string, socketDir:string}} opts
|
|
153
|
+
*/
|
|
154
|
+
export function generateLaunchdPlist({ identity, tier, nodePath = '/usr/local/bin/node', daemonPath, socketDir }) {
|
|
155
|
+
assertTier(tier);
|
|
156
|
+
if (!identity || !daemonPath || !socketDir) throw new Error('generateLaunchdPlist requires { identity, daemonPath, socketDir }');
|
|
157
|
+
const id = shortId(identity);
|
|
158
|
+
const label = `ai.metamynd.agentsafe-signer.${id}`;
|
|
159
|
+
const socketPath = path.posix.join(socketDir, `signer-${id}.sock`);
|
|
160
|
+
|
|
161
|
+
// launchd's plist has no LimitCORE-equivalent key at all (unlike systemd), and `ulimit` is a
|
|
162
|
+
// shell builtin, not a real binary — so, unlike setpriv above, this genuinely needs a shell
|
|
163
|
+
// hop. `sh -c 'ulimit -c 0; exec "$@"' sh <node> <daemon> --identity <id>` sets $0='sh' (unused,
|
|
164
|
+
// just a placeholder) and the rest as $1.. via "$@", then `exec` replaces the shell with node —
|
|
165
|
+
// a true exec chain, not a forked supervisor, same as setpriv's own execve() above.
|
|
166
|
+
//
|
|
167
|
+
// Disclosed, not fixed: macOS has no ptrace-deny equivalent reachable this way at all —
|
|
168
|
+
// PT_DENY_ATTACH must be called BY the target process itself via a syscall, which needs a
|
|
169
|
+
// native addon or FFI to reach from here, the same architectural limitation secure-memory.mjs's
|
|
170
|
+
// own header discloses for mlock(). Only RLIMIT_CORE is closed on macOS by this generator.
|
|
171
|
+
const plist = [
|
|
172
|
+
'<?xml version="1.0" encoding="UTF-8"?>',
|
|
173
|
+
'<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">',
|
|
174
|
+
'<plist version="1.0">',
|
|
175
|
+
'<dict>',
|
|
176
|
+
` <key>Label</key><string>${label}</string>`,
|
|
177
|
+
' <key>ProgramArguments</key>',
|
|
178
|
+
' <array>',
|
|
179
|
+
' <string>/bin/sh</string>',
|
|
180
|
+
' <string>-c</string>',
|
|
181
|
+
' <string>ulimit -c 0; exec "$@"</string>',
|
|
182
|
+
' <string>sh</string>',
|
|
183
|
+
` <string>${nodePath}</string>`,
|
|
184
|
+
` <string>${daemonPath}</string>`,
|
|
185
|
+
' <string>--identity</string>',
|
|
186
|
+
` <string>${id}</string>`,
|
|
187
|
+
' </array>',
|
|
188
|
+
' <key>Sockets</key>',
|
|
189
|
+
' <dict>',
|
|
190
|
+
' <key>Listener</key>',
|
|
191
|
+
' <dict>',
|
|
192
|
+
` <key>SockPathName</key><string>${socketPath}</string>`,
|
|
193
|
+
' <key>SockPathMode</key><integer>384</integer>',
|
|
194
|
+
' </dict>',
|
|
195
|
+
' </dict>',
|
|
196
|
+
' <key>RunAtLoad</key><false/>',
|
|
197
|
+
' <key>KeepAlive</key><false/>',
|
|
198
|
+
' <key>ProcessType</key><string>Background</string>',
|
|
199
|
+
'</dict>',
|
|
200
|
+
'</plist>',
|
|
201
|
+
'',
|
|
202
|
+
].join('\n');
|
|
203
|
+
|
|
204
|
+
const plistFileName = `${label}.plist`;
|
|
205
|
+
const enableCommand =
|
|
206
|
+
tier === 1
|
|
207
|
+
? `launchctl bootstrap gui/$(id -u) ~/Library/LaunchAgents/${plistFileName}`
|
|
208
|
+
: `# Tier 2 on macOS is a sandbox-exec Seatbelt profile layered on top of this same LaunchAgent, ` +
|
|
209
|
+
`not a separate plist mechanism — see README.md "Status" for what's not yet generated here.\n` +
|
|
210
|
+
`launchctl bootstrap gui/$(id -u) ~/Library/LaunchAgents/${plistFileName}`;
|
|
211
|
+
|
|
212
|
+
return { plist, plistFileName, id, enableCommand };
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
// --- Windows: a service under a Virtual Service Account, named pipe transport ---------------
|
|
216
|
+
|
|
217
|
+
/**
|
|
218
|
+
* Returns the exact sc.exe argv this would run — never runs it. Tier 1 has no natural analog on
|
|
219
|
+
* Windows (services are inherently system-level there), so this always targets Tier 2's Virtual
|
|
220
|
+
* Service Account, matching the design doc's own "Tier 2 by default here, not Tier 1" call.
|
|
221
|
+
* @param {{identity:string, nodePath?:string, daemonPath:string}} opts
|
|
222
|
+
*/
|
|
223
|
+
export function generateWindowsService({ identity, nodePath = 'node.exe', daemonPath }) {
|
|
224
|
+
if (!identity || !daemonPath) throw new Error('generateWindowsService requires { identity, daemonPath }');
|
|
225
|
+
const id = shortId(identity);
|
|
226
|
+
const serviceName = `AgentSafeSigner-${id}`;
|
|
227
|
+
const virtualAccount = `NT SERVICE\\${serviceName}`;
|
|
228
|
+
const binPath = `"${nodePath}" "${daemonPath}" --identity ${id}`;
|
|
229
|
+
|
|
230
|
+
const createArgv = ['sc.exe', 'create', serviceName, `binPath=${binPath}`, 'start=auto', `obj=${virtualAccount}`, 'password='];
|
|
231
|
+
const startArgv = ['sc.exe', 'start', serviceName];
|
|
232
|
+
|
|
233
|
+
return {
|
|
234
|
+
serviceName,
|
|
235
|
+
virtualAccount,
|
|
236
|
+
createArgv,
|
|
237
|
+
startArgv,
|
|
238
|
+
enableCommand: `${createArgv.map(quoteIfNeeded).join(' ')}\n${startArgv.join(' ')}`,
|
|
239
|
+
};
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
function quoteIfNeeded(arg) {
|
|
243
|
+
return /\s/.test(arg) && !arg.includes('"') ? `"${arg}"` : arg;
|
|
244
|
+
}
|
|
@@ -0,0 +1,367 @@
|
|
|
1
|
+
// windows-secure-pipe.mjs — a real, ACL-restricted named pipe on Windows, closing the gap the
|
|
2
|
+
// design doc originally flagged as blocked ("Node's net module has no API for a custom named-pipe
|
|
3
|
+
// ACL... needs a native addon"). That was wrong: the same technique already used for DPAPI
|
|
4
|
+
// (kek-backends.mjs) — shelling out to .NET via PowerShell, no native module — works here too.
|
|
5
|
+
//
|
|
6
|
+
// .NET's System.IO.Pipes.NamedPipeServerStream accepts a real PipeSecurity at construction, which
|
|
7
|
+
// Node's own net module has no equivalent for. So the pipe itself is created and owned by a small
|
|
8
|
+
// PowerShell/.NET relay process, restricted to the current Windows user, and Node talks to it
|
|
9
|
+
// over the relay's own stdin/stdout — a normal child-process pipe, nothing exotic. Once a client
|
|
10
|
+
// connects to the real named pipe, bytes flow: external caller <-> restricted named pipe (ACL
|
|
11
|
+
// enforced by Windows/.NET) <-> relay process <-> stdin/stdout <-> this module <-> daemon.mjs's
|
|
12
|
+
// existing protocol handling, unchanged.
|
|
13
|
+
//
|
|
14
|
+
// The signing socket needs to keep accepting new connections indefinitely, so a POOL of relay
|
|
15
|
+
// processes is kept warm (each an independent NamedPipeServerStream instance sharing the same
|
|
16
|
+
// pipe name — the normal Windows pattern for a multi-instance named pipe server); whichever one a
|
|
17
|
+
// client connects to gets replaced immediately so the pool stays full. The admin socket is a
|
|
18
|
+
// single, non-replenished instance that self-closes after one connection or its own timeout,
|
|
19
|
+
// matching its existing one-shot design on Linux/macOS.
|
|
20
|
+
import { spawn } from 'node:child_process';
|
|
21
|
+
import { EventEmitter } from 'node:events';
|
|
22
|
+
|
|
23
|
+
// How long a single relay attempt may take to construct its pipe and report ACL_RULE_COUNT
|
|
24
|
+
// before it's treated as stuck and killed so a fresh attempt can be made. Verified empirically
|
|
25
|
+
// (not hypothesized): under concurrent load — several relay processes constructing a
|
|
26
|
+
// NamedPipeServerStream with a custom PipeSecurity at nearly the same instant, even for
|
|
27
|
+
// DIFFERENT pipe names — one occasionally hangs indefinitely inside the .NET constructor call
|
|
28
|
+
// itself, never printing so much as ACL_RULE_COUNT and never exiting on its own. Without a bound
|
|
29
|
+
// here, that hang was unrecoverable and propagated all the way up to daemon.startSigningServer/
|
|
30
|
+
// startAdminServer's own callers, which had no way to detect or retry it either.
|
|
31
|
+
const SPAWN_TIMEOUT_MS = 1500;
|
|
32
|
+
// Bounds createSecurePipeOnce's retries specifically (the pool already retries indefinitely as
|
|
33
|
+
// part of its own "always on" design — see spawnOne's own comment) so a persistently broken
|
|
34
|
+
// environment fails loudly after a bounded number of attempts instead of retrying forever. A
|
|
35
|
+
// shorter SPAWN_TIMEOUT_MS with more attempts (rather than a longer timeout with fewer) recovers
|
|
36
|
+
// faster from the common case — the observed contention tends to clear within one or two
|
|
37
|
+
// retries — while keeping a similar worst-case total bound.
|
|
38
|
+
const MAX_SPAWN_ATTEMPTS = 8;
|
|
39
|
+
|
|
40
|
+
function relayScript(pipeName, maxInstances) {
|
|
41
|
+
return [
|
|
42
|
+
'Add-Type -AssemblyName System.Core',
|
|
43
|
+
"$ErrorActionPreference = 'Stop'",
|
|
44
|
+
'try {',
|
|
45
|
+
' $currentUser = [System.Security.Principal.WindowsIdentity]::GetCurrent().User',
|
|
46
|
+
' $security = New-Object System.IO.Pipes.PipeSecurity',
|
|
47
|
+
// FullControl, not ReadWrite — verified empirically (not assumed) that a multi-instance named
|
|
48
|
+
// pipe with an explicit PipeSecurity granting only ReadWrite fails EVERY instance after the
|
|
49
|
+
// first with "Access to the path is denied": creating an additional instance of an existing
|
|
50
|
+
// pipe needs rights beyond plain read/write, which a ReadWrite-only ACL does not grant even to
|
|
51
|
+
// the same identity that holds it. FullControl still restricts the pipe to exactly this one
|
|
52
|
+
// identity — the security property this exists for — it only changes what THAT identity may
|
|
53
|
+
// do with its own pipe, not who else can reach it.
|
|
54
|
+
' $rule = New-Object System.IO.Pipes.PipeAccessRule($currentUser, [System.IO.Pipes.PipeAccessRights]::FullControl, [System.Security.AccessControl.AccessControlType]::Allow)',
|
|
55
|
+
' $security.AddAccessRule($rule)',
|
|
56
|
+
// PipeOptions.Asynchronous, not None — the two CopyToAsync loops below run concurrently in
|
|
57
|
+
// opposite directions on this SAME handle (client->relay and relay->client). A handle opened
|
|
58
|
+
// without FILE_FLAG_OVERLAPPED (PipeOptions.None) only supports one outstanding I/O operation
|
|
59
|
+
// at a time; a concurrent Read and Write on it races and faults one of the two Tasks, which
|
|
60
|
+
// WaitAny treats as "completed" and immediately Disposes the pipe — truncating whichever
|
|
61
|
+
// direction was still in flight. Verified empirically: with None, a response written to this
|
|
62
|
+
// relay's stdin after a client request never reached the client at all.
|
|
63
|
+
` $pipe = New-Object System.IO.Pipes.NamedPipeServerStream("${pipeName}", [System.IO.Pipes.PipeDirection]::InOut, ${maxInstances}, [System.IO.Pipes.PipeTransmissionMode]::Byte, [System.IO.Pipes.PipeOptions]::Asynchronous, 65536, 65536, $security)`,
|
|
64
|
+
' $ruleCount = $pipe.GetAccessControl().GetAccessRules($true,$true,[System.Security.Principal.NTAccount]).Count',
|
|
65
|
+
' [Console]::Error.WriteLine("ACL_RULE_COUNT:$ruleCount")',
|
|
66
|
+
' [Console]::Error.WriteLine("READY")',
|
|
67
|
+
' $pipe.WaitForConnection()',
|
|
68
|
+
' [Console]::Error.WriteLine("CONNECTED")',
|
|
69
|
+
' $stdin = [Console]::OpenStandardInput()',
|
|
70
|
+
' $stdout = [Console]::OpenStandardOutput()',
|
|
71
|
+
' $t1 = $pipe.CopyToAsync($stdout)',
|
|
72
|
+
' $t2 = $stdin.CopyToAsync($pipe)',
|
|
73
|
+
' [System.Threading.Tasks.Task]::WaitAny(@($t1, $t2)) | Out-Null',
|
|
74
|
+
' $pipe.Dispose()',
|
|
75
|
+
' [Console]::Error.WriteLine("CLOSED")',
|
|
76
|
+
'} catch {',
|
|
77
|
+
' [Console]::Error.WriteLine("RELAY_ERROR:" + $_.Exception.Message)',
|
|
78
|
+
' exit 1',
|
|
79
|
+
'}',
|
|
80
|
+
].join('\n');
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
function spawnRelay(pipeName, maxInstances) {
|
|
84
|
+
return spawn('powershell.exe', ['-NoProfile', '-NonInteractive', '-Command', relayScript(pipeName, maxInstances)], {
|
|
85
|
+
windowsHide: true,
|
|
86
|
+
stdio: ['pipe', 'pipe', 'pipe'],
|
|
87
|
+
});
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/** Minimal net.Socket-shaped adapter over a relay child process's own stdio. */
|
|
91
|
+
function socketAdapter(child) {
|
|
92
|
+
return {
|
|
93
|
+
on: (event, cb) => child.stdout.on(event, cb),
|
|
94
|
+
once: (event, cb) => child.stdout.once(event, cb),
|
|
95
|
+
write: (data) => child.stdin.write(data),
|
|
96
|
+
destroy: () => {
|
|
97
|
+
try {
|
|
98
|
+
child.kill();
|
|
99
|
+
} catch {
|
|
100
|
+
/* already exited */
|
|
101
|
+
}
|
|
102
|
+
},
|
|
103
|
+
};
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
function watchRelayStderr(child, { onAclVerified, onConnected } = {}) {
|
|
107
|
+
let buf = '';
|
|
108
|
+
child.stderr.on('data', (chunk) => {
|
|
109
|
+
buf += chunk.toString('utf8');
|
|
110
|
+
let idx;
|
|
111
|
+
while ((idx = buf.indexOf('\n')) !== -1) {
|
|
112
|
+
const line = buf.slice(0, idx).trim();
|
|
113
|
+
buf = buf.slice(idx + 1);
|
|
114
|
+
if (line.startsWith('ACL_RULE_COUNT:')) onAclVerified?.(Number(line.split(':')[1]));
|
|
115
|
+
else if (line === 'CONNECTED') onConnected?.();
|
|
116
|
+
else if (line.startsWith('RELAY_ERROR:')) console.error(`[windows-secure-pipe] relay error: ${line.slice('RELAY_ERROR:'.length)}`);
|
|
117
|
+
}
|
|
118
|
+
});
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* Keeps `poolSize` relay instances warm on `pipeName`, calling `onConnection(socketLike)` once
|
|
123
|
+
* per accepted client and immediately spawning a replacement to keep the pool full. Shaped like a
|
|
124
|
+
* `net.Server` (`.close()`, emits `'close'`, `'aclVerified'` fires once per spawned instance —
|
|
125
|
+
* independent of whether it ever receives a connection — as the real test hook proving every
|
|
126
|
+
* instance, not just a lucky one, is actually restricted).
|
|
127
|
+
*/
|
|
128
|
+
export function createSecurePipePool(pipeName, onConnection, { poolSize = 4 } = {}) {
|
|
129
|
+
const emitter = new EventEmitter();
|
|
130
|
+
let stopped = false;
|
|
131
|
+
const children = new Set();
|
|
132
|
+
emitter.closed = false; // see createSecurePipeOnce's own comment on this — same synchronous escape hatch
|
|
133
|
+
|
|
134
|
+
// Ongoing replenishment after a successful start retries indefinitely on purpose (the pool's
|
|
135
|
+
// whole "always on" design — a transient failure refilling one consumed slot should just keep
|
|
136
|
+
// trying). But a pool that has NEVER once succeeded needs a bound: observed directly, not
|
|
137
|
+
// hypothesized — a persistent (not transient) construction failure can keep every attempt
|
|
138
|
+
// failing after its own SPAWN_TIMEOUT_MS, in a tight spawn->timeout->exit->respawn cycle, for
|
|
139
|
+
// 15+ minutes straight with zero signal to the caller that anything is wrong (daemon.mjs's
|
|
140
|
+
// startServer just awaits 'aclVerified' forever). STARTUP_FAILURE_LIMIT bounds only the
|
|
141
|
+
// "never even once succeeded" case; a pool that already started stays resilient indefinitely.
|
|
142
|
+
let everSucceeded = false;
|
|
143
|
+
let consecutiveFailures = 0;
|
|
144
|
+
const STARTUP_FAILURE_LIMIT = 20; // ~20 * SPAWN_TIMEOUT_MS worst case before giving up on a first-ever start
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* Resolves once THIS instance's own pipe object is constructed (its ACL confirmed) — not once
|
|
148
|
+
* it has a client. Creating several NamedPipeServerStream instances of the SAME pipe name
|
|
149
|
+
* concurrently races on Windows (observed directly: "Access to the path is denied" from the
|
|
150
|
+
* SECOND+ instance's constructor when several are spawned at once); waiting for each one to
|
|
151
|
+
* finish constructing before starting the next avoids the race entirely.
|
|
152
|
+
*
|
|
153
|
+
* Replenishment happens on the relay's own EXIT, not on `onConnected` — a connected instance
|
|
154
|
+
* still holds its slot against `poolSize` for as long as it's actively relaying (Windows only
|
|
155
|
+
* frees the slot once that instance's pipe object is disposed). Replenishing immediately on
|
|
156
|
+
* connect, before that slot is actually free, briefly asks for `poolSize + 1` live instances at
|
|
157
|
+
* once and fails with "All pipe instances are busy" — observed directly, not a hypothetical.
|
|
158
|
+
*/
|
|
159
|
+
function spawnOne() {
|
|
160
|
+
if (stopped) return Promise.resolve();
|
|
161
|
+
return new Promise((resolve) => {
|
|
162
|
+
const child = spawnRelay(pipeName, poolSize);
|
|
163
|
+
children.add(child);
|
|
164
|
+
let resolved = false;
|
|
165
|
+
let acked = false;
|
|
166
|
+
// See SPAWN_TIMEOUT_MS's own comment: a relay can hang indefinitely inside the .NET pipe
|
|
167
|
+
// constructor under concurrent load without this — killing it here just triggers the SAME
|
|
168
|
+
// 'exit' handler below that any other exit does, so replenishment/retry falls out for free.
|
|
169
|
+
const spawnTimer = setTimeout(() => {
|
|
170
|
+
if (!acked) {
|
|
171
|
+
try {
|
|
172
|
+
child.kill();
|
|
173
|
+
} catch {
|
|
174
|
+
/* already exited */
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
}, SPAWN_TIMEOUT_MS);
|
|
178
|
+
watchRelayStderr(child, {
|
|
179
|
+
onAclVerified: (n) => {
|
|
180
|
+
acked = true;
|
|
181
|
+
everSucceeded = true;
|
|
182
|
+
consecutiveFailures = 0;
|
|
183
|
+
clearTimeout(spawnTimer);
|
|
184
|
+
emitter.emit('aclVerified', n);
|
|
185
|
+
if (!resolved) {
|
|
186
|
+
resolved = true;
|
|
187
|
+
resolve();
|
|
188
|
+
}
|
|
189
|
+
},
|
|
190
|
+
onConnected: () => onConnection(socketAdapter(child)),
|
|
191
|
+
});
|
|
192
|
+
child.on('exit', () => {
|
|
193
|
+
clearTimeout(spawnTimer);
|
|
194
|
+
children.delete(child);
|
|
195
|
+
if (!acked && !everSucceeded) {
|
|
196
|
+
consecutiveFailures++;
|
|
197
|
+
if (consecutiveFailures >= STARTUP_FAILURE_LIMIT) {
|
|
198
|
+
// Never once succeeded, after a generous bounded number of attempts — stop retrying
|
|
199
|
+
// and tell the caller, instead of cycling forever with no way for anyone to notice.
|
|
200
|
+
stopped = true;
|
|
201
|
+
emitter.closed = true;
|
|
202
|
+
emitter.emit('close', new Error(`createSecurePipePool: gave up after ${consecutiveFailures} consecutive failed attempts — the pipe never became ready even once`));
|
|
203
|
+
if (!resolved) {
|
|
204
|
+
resolved = true;
|
|
205
|
+
resolve();
|
|
206
|
+
}
|
|
207
|
+
return;
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
if (!stopped) spawnOne(); // this slot just freed for real — replenish now, not earlier
|
|
211
|
+
if (!resolved) {
|
|
212
|
+
resolved = true;
|
|
213
|
+
resolve(); // don't hang the startup chain on a relay that failed before ever reporting ready
|
|
214
|
+
}
|
|
215
|
+
});
|
|
216
|
+
child.on('error', () => {
|
|
217
|
+
clearTimeout(spawnTimer);
|
|
218
|
+
children.delete(child);
|
|
219
|
+
});
|
|
220
|
+
});
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
(async () => {
|
|
224
|
+
for (let i = 0; i < poolSize; i++) {
|
|
225
|
+
if (stopped) break;
|
|
226
|
+
await spawnOne(); // sequential on purpose — see spawnOne's own comment
|
|
227
|
+
}
|
|
228
|
+
})();
|
|
229
|
+
|
|
230
|
+
// Same discipline as createSecurePipeOnce below: 'close' means every instance's pipe handle is
|
|
231
|
+
// actually gone (all children have genuinely exited), not just "kill() was requested" — a caller
|
|
232
|
+
// restarting a daemon on the same logical path right after `.close()` needs the OS to have
|
|
233
|
+
// already released the old pipe name.
|
|
234
|
+
emitter.close = () => {
|
|
235
|
+
if (stopped) return;
|
|
236
|
+
stopped = true;
|
|
237
|
+
if (children.size === 0) {
|
|
238
|
+
emitter.closed = true;
|
|
239
|
+
emitter.emit('close');
|
|
240
|
+
return;
|
|
241
|
+
}
|
|
242
|
+
let remaining = children.size;
|
|
243
|
+
for (const c of children) {
|
|
244
|
+
c.once('exit', () => {
|
|
245
|
+
remaining--;
|
|
246
|
+
if (remaining === 0) {
|
|
247
|
+
emitter.closed = true;
|
|
248
|
+
emitter.emit('close');
|
|
249
|
+
}
|
|
250
|
+
});
|
|
251
|
+
c.kill();
|
|
252
|
+
}
|
|
253
|
+
};
|
|
254
|
+
return emitter;
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
/**
|
|
258
|
+
* A single, non-replenished instance for the admin socket: closes after one connection is fully
|
|
259
|
+
* handled (the caller calls `.close()` itself, same as the existing Linux/macOS admin server) or
|
|
260
|
+
* after `timeoutMs` with no connection at all. Same `net.Server`-shaped interface as the pool.
|
|
261
|
+
*/
|
|
262
|
+
export function createSecurePipeOnce(pipeName, onConnection, { timeoutMs = 60_000 } = {}) {
|
|
263
|
+
const emitter = new EventEmitter();
|
|
264
|
+
let closed = false; // true once emitter.close() has been called, OR a terminal finish/fail happened
|
|
265
|
+
let everReady = false; // true once 'aclVerified' has fired at least once, across any attempt
|
|
266
|
+
let child = null;
|
|
267
|
+
let attempts = 0;
|
|
268
|
+
const timer = setTimeout(() => emitter.close(), timeoutMs);
|
|
269
|
+
emitter.closed = false; // readable synchronously by a caller that only gets `emitter` AFTER it
|
|
270
|
+
// already closed (see daemon.mjs's startAdminServer) — a 'close' listener attached after the
|
|
271
|
+
// event already fired would never see it; this lets that caller check directly instead.
|
|
272
|
+
|
|
273
|
+
// 'close' must mean the pipe is GENUINELY gone, not merely "kill() was called" — a caller (e.g.
|
|
274
|
+
// daemon.mjs's startAdminServer, which resolves its own promise off this event) uses it as the
|
|
275
|
+
// signal that the admin socket is safe to consider unreachable. child.kill() only requests
|
|
276
|
+
// termination; the relay process (and its held pipe handle) can still exist for a few more
|
|
277
|
+
// milliseconds until Windows actually tears it down. Verified empirically: without waiting for
|
|
278
|
+
// the child's own 'exit', a client connecting immediately after `.close()` could still reach the
|
|
279
|
+
// pipe.
|
|
280
|
+
function finish() {
|
|
281
|
+
clearTimeout(timer);
|
|
282
|
+
emitter.closed = true;
|
|
283
|
+
emitter.emit('close');
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
// Emitted (with an Error) only when every attempt exhausted MAX_SPAWN_ATTEMPTS without the pipe
|
|
287
|
+
// EVER becoming ready even once — daemon.mjs's startServer listens for this so a persistently
|
|
288
|
+
// broken environment REJECTS the caller's promise instead of hanging it forever, which nothing
|
|
289
|
+
// did before this existed (there was no failure path at all if the pipe simply never came up).
|
|
290
|
+
// A pipe that WAS ready at least once (everReady) and later closes (e.g. its own overall
|
|
291
|
+
// timeoutMs elapsed with no client ever connecting — the normal, expected admin-socket
|
|
292
|
+
// lifecycle) is NOT a failure and goes through finish() instead, even on its very last attempt.
|
|
293
|
+
function fail(err) {
|
|
294
|
+
clearTimeout(timer);
|
|
295
|
+
emitter.closed = true;
|
|
296
|
+
emitter.emit('close', err);
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
function spawnAttempt() {
|
|
300
|
+
attempts++;
|
|
301
|
+
child = spawnRelay(pipeName, 1);
|
|
302
|
+
let acked = false;
|
|
303
|
+
// See SPAWN_TIMEOUT_MS's own comment: without this, a relay stuck inside the .NET pipe
|
|
304
|
+
// constructor under concurrent load hangs forever with no signal at all.
|
|
305
|
+
const spawnTimer = setTimeout(() => {
|
|
306
|
+
if (!acked) {
|
|
307
|
+
try {
|
|
308
|
+
child.kill();
|
|
309
|
+
} catch {
|
|
310
|
+
/* already exited */
|
|
311
|
+
}
|
|
312
|
+
}
|
|
313
|
+
}, SPAWN_TIMEOUT_MS);
|
|
314
|
+
|
|
315
|
+
watchRelayStderr(child, {
|
|
316
|
+
onAclVerified: (n) => {
|
|
317
|
+
acked = true;
|
|
318
|
+
everReady = true;
|
|
319
|
+
clearTimeout(spawnTimer);
|
|
320
|
+
emitter.emit('aclVerified', n);
|
|
321
|
+
},
|
|
322
|
+
onConnected: () => onConnection(socketAdapter(child)),
|
|
323
|
+
});
|
|
324
|
+
|
|
325
|
+
const onExit = () => {
|
|
326
|
+
clearTimeout(spawnTimer);
|
|
327
|
+
if (closed) {
|
|
328
|
+
// A deliberate emitter.close() call caused this exit (its own overall timeoutMs elapsing
|
|
329
|
+
// counts as deliberate too, per the timer above) — this IS the genuine-exit signal that
|
|
330
|
+
// call was waiting for, whether or not this particular attempt ever got ready.
|
|
331
|
+
finish();
|
|
332
|
+
return;
|
|
333
|
+
}
|
|
334
|
+
if (acked) {
|
|
335
|
+
// Served its purpose (handled its one connection, or the relay script completed on its
|
|
336
|
+
// own) — not a construction failure, so no retry, just the normal close.
|
|
337
|
+
closed = true;
|
|
338
|
+
finish();
|
|
339
|
+
return;
|
|
340
|
+
}
|
|
341
|
+
// This attempt never became ready — either crashed immediately or was killed by the spawn
|
|
342
|
+
// timeout above. Retry a bounded number of times before giving up for good.
|
|
343
|
+
if (attempts < MAX_SPAWN_ATTEMPTS) {
|
|
344
|
+
spawnAttempt();
|
|
345
|
+
return;
|
|
346
|
+
}
|
|
347
|
+
closed = true;
|
|
348
|
+
fail(new Error(`createSecurePipeOnce: gave up after ${attempts} attempts — the pipe never became ready`));
|
|
349
|
+
};
|
|
350
|
+
child.on('exit', onExit);
|
|
351
|
+
child.on('error', onExit);
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
spawnAttempt();
|
|
355
|
+
|
|
356
|
+
emitter.close = () => {
|
|
357
|
+
if (closed) return;
|
|
358
|
+
closed = true;
|
|
359
|
+
try {
|
|
360
|
+
child.kill();
|
|
361
|
+
} catch {
|
|
362
|
+
finish(); // already exited before we could kill it — nothing left to wait for
|
|
363
|
+
}
|
|
364
|
+
};
|
|
365
|
+
|
|
366
|
+
return emitter;
|
|
367
|
+
}
|