kern-sandbox 0.1.19 → 0.1.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -8
- package/index.d.ts +1 -1
- package/index.js +75 -9
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
**[kern](https://github.com/getkern/kern)** is a fast, rootless sandbox and virtual resource
|
|
4
4
|
runtime for any workload, including untrusted and AI-generated code: a real, kernel-enforced box
|
|
5
|
-
that starts in **3.4 ms** from an OCI image, out of one
|
|
5
|
+
that starts in **3.4 ms** from an OCI image, out of one **1.58 MB** binary, with no daemon.
|
|
6
6
|
**kern-sandbox**
|
|
7
7
|
is its Node / TypeScript binding: run untrusted or agent-generated code in a fresh, isolated box, from Node.
|
|
8
8
|
|
|
@@ -11,7 +11,7 @@ package is on PyPI: [`kern-sandbox`](https://pypi.org/project/kern-sandbox/).
|
|
|
11
11
|
|
|
12
12
|
It is a thin, dependency-free wrapper around the [`kern`](https://github.com/getkern/kern) binary:
|
|
13
13
|
a fresh, isolated box per call, network off by default, hard resource caps, and a timeout the binding
|
|
14
|
-
itself enforces. Kernel-enforced isolation (namespaces, cgroups v2, seccomp), local, about 1.
|
|
14
|
+
itself enforces. Kernel-enforced isolation (namespaces, cgroups v2, seccomp), local, about 1.58 MB, with no cloud, no account, no VM.
|
|
15
15
|
|
|
16
16
|
```js
|
|
17
17
|
const kern = require("kern-sandbox");
|
|
@@ -98,7 +98,8 @@ A non-zero exit from *your code* is **not** a fault (`fault` stays `null`): it i
|
|
|
98
98
|
|---|---|
|
|
99
99
|
| `timeout` | the call exceeded `timeoutS`; the binding killed the box |
|
|
100
100
|
| `escape_blocked` | a syscall was blocked by the seccomp filter (SIGSYS) |
|
|
101
|
-
| `
|
|
101
|
+
| `oom` | the box was SIGKILLed and a `memoryMb` cap was **in force**: a breached `memory.max` is the cgroup OOM-killer (`memory.oom.group=1` kills the whole box). kern reports whether the cap actually bound on an unforgeable per-box channel (2nd byte of `KERN_STARTED_FD`), so this is an *enforced-cap* OOM |
|
|
102
|
+
| `killed` | a SIGKILL **not** attributed to a cgroup OOM: no `memoryMb` cap was set, or kern reported the cap did not bind here (no cgroup delegation), so it is host pressure / an external kill. Older kern (no enforcement byte) falls back to `oom` when a cap was set |
|
|
102
103
|
|
|
103
104
|
A box that fails to **start** (kern exits 125: a mount refused at runtime, an unmappable `--user`, a
|
|
104
105
|
seccomp/AppArmor/cgroup setup error, or a pull/image error) is **thrown** as a `SandboxError`, not
|
|
@@ -225,11 +226,12 @@ clear error otherwise). The Python binding uses the stdlib `tarfile` and has no
|
|
|
225
226
|
## Honest threat model
|
|
226
227
|
|
|
227
228
|
kern is a **kernel-boundary** sandbox for **your own or semi-trusted** code (CI, dev, edge, your
|
|
228
|
-
agents' code). Its seccomp filter is a **
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
229
|
+
agents' code). Its default seccomp filter is a **deny-by-default allowlist** (moby's own default
|
|
230
|
+
filter minus kern's 35 escape syscalls): right for semi-trusted agent code, **not** a hard boundary
|
|
231
|
+
against deliberately hostile multi-tenant code. For that, reach for a microVM (Firecracker / Kata) or
|
|
232
|
+
gVisor. The wider denylist is the opt-out (`KERN_SECCOMP=denylist`), and `securityProfile: "untrusted"`
|
|
233
|
+
bundles the allowlist with `--cap-drop ALL` + `--read-only`. See the project's
|
|
234
|
+
[SECURITY.md](https://github.com/getkern/kern/blob/main/SECURITY.md).
|
|
233
235
|
|
|
234
236
|
## License
|
|
235
237
|
|
package/index.d.ts
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
// Run LLM/agent-generated code in a fast, local, daemonless kernel sandbox.
|
|
3
3
|
|
|
4
4
|
/** What stopped the code at the SANDBOX level. Reported as data on a result, never thrown. */
|
|
5
|
-
export type SandboxFaultType = "timeout" | "escape_blocked" | "killed" | "startup_failed";
|
|
5
|
+
export type SandboxFaultType = "timeout" | "oom" | "escape_blocked" | "killed" | "startup_failed";
|
|
6
6
|
|
|
7
7
|
export interface SandboxFault {
|
|
8
8
|
type: SandboxFaultType;
|
package/index.js
CHANGED
|
@@ -1043,6 +1043,7 @@ class Sandbox {
|
|
|
1043
1043
|
return new Promise((resolve, reject) => {
|
|
1044
1044
|
let child;
|
|
1045
1045
|
let boxStarted = false;
|
|
1046
|
+
let capSignal = 0; // 2nd started byte: 0 undetermined/old-kern, 1 memory cap enforced, 2 not enforced
|
|
1046
1047
|
try {
|
|
1047
1048
|
// detached: own process group, so we can signal the box + kern as a unit (killpg).
|
|
1048
1049
|
// The 4th stdio slot is fd 3: the child (kern) writes the started byte, the parent reads it.
|
|
@@ -1058,9 +1059,11 @@ class Sandbox {
|
|
|
1058
1059
|
|
|
1059
1060
|
const startedCh = child.stdio[3];
|
|
1060
1061
|
if (startedCh) {
|
|
1061
|
-
//
|
|
1062
|
+
// Byte 0 (0x01) = the box started; stream end with no byte = never started / old kern. Byte 1
|
|
1063
|
+
// (a NEWER kern only, same atomic write) = the memory-cap enforcement signal; absent = 0.
|
|
1062
1064
|
startedCh.on("data", (b) => {
|
|
1063
1065
|
if (b.length && b[0] === 1) boxStarted = true;
|
|
1066
|
+
if (b.length >= 2) capSignal = b[1];
|
|
1064
1067
|
});
|
|
1065
1068
|
startedCh.on("error", () => {});
|
|
1066
1069
|
}
|
|
@@ -1087,7 +1090,7 @@ class Sandbox {
|
|
|
1087
1090
|
const stdout = out.buffer().toString("utf8");
|
|
1088
1091
|
const stderr = err.buffer().toString("utf8");
|
|
1089
1092
|
const rc = toRc(code, signal);
|
|
1090
|
-
let fault = this._classify(rc, signal, stderr, timedOut, timeoutS);
|
|
1093
|
+
let fault = this._classify(rc, signal, stderr, timedOut, timeoutS, capSignal);
|
|
1091
1094
|
if (boxStarted && fault && fault.type === "startup_failed") {
|
|
1092
1095
|
// kern signalled the box STARTED, so a `startup_failed` here is only the stderr heuristic
|
|
1093
1096
|
// matching a marker the WORKLOAD wrote (code-based faults are decided first). The box
|
|
@@ -1161,7 +1164,7 @@ class Sandbox {
|
|
|
1161
1164
|
}
|
|
1162
1165
|
}
|
|
1163
1166
|
|
|
1164
|
-
_classify(rc, signal, stderr, timedOut, timeoutS) {
|
|
1167
|
+
_classify(rc, signal, stderr, timedOut, timeoutS, capSignal = 0) {
|
|
1165
1168
|
// ORDER IS A SECURITY PROPERTY: deterministic-by-exit-code classes are decided BEFORE the stderr
|
|
1166
1169
|
// heuristic, because stderr is a channel the workload controls.
|
|
1167
1170
|
if (timedOut)
|
|
@@ -1171,8 +1174,23 @@ class Sandbox {
|
|
|
1171
1174
|
);
|
|
1172
1175
|
if (rc === EXIT_SIGSYS || signal === "SIGSYS")
|
|
1173
1176
|
return sandboxFault("escape_blocked", "a syscall was blocked by the seccomp filter (SIGSYS)");
|
|
1174
|
-
if (rc === EXIT_SIGKILL || signal === "SIGKILL")
|
|
1175
|
-
|
|
1177
|
+
if (rc === EXIT_SIGKILL || signal === "SIGKILL") {
|
|
1178
|
+
// A memory-capped box SIGKILLed is the cgroup OOM-killer - what a breached memory.max does (kern
|
|
1179
|
+
// sets memory.oom.group=1, so the whole box dies at once). `capSignal` is kern's UNFORGEABLE
|
|
1180
|
+
// enforcement byte (2nd byte of KERN_STARTED_FD, not the workload's stderr): 1 = enforced, 2 =
|
|
1181
|
+
// requested but NOT enforced here, 0 = undetermined (old kern / no --memory). Claim `oom` when a
|
|
1182
|
+
// --memory cap was set AND kern did not report it unenforced (capSignal !== 2): enforced (1) is a
|
|
1183
|
+
// certain cgroup OOM, undetermined (0) keeps the pre-signal heuristic. When kern reports the cap did
|
|
1184
|
+
// not bind (2), a SIGKILL cannot be attributed to the box's cgroup - keep the honest `killed`.
|
|
1185
|
+
if (this.memoryMb !== null && capSignal !== 2)
|
|
1186
|
+
return sandboxFault("oom", "the box exceeded its memory cap and was OOM-killed (SIGKILL, exit 137)");
|
|
1187
|
+
if (capSignal === 2)
|
|
1188
|
+
return sandboxFault(
|
|
1189
|
+
"killed",
|
|
1190
|
+
"the box was SIGKILLed, but its memory cap was not enforced here (no cgroup delegation), so it is not attributed to a cgroup OOM",
|
|
1191
|
+
);
|
|
1192
|
+
return sandboxFault("killed", "the box was killed (SIGKILL); no memory cap was set to attribute it to OOM");
|
|
1193
|
+
}
|
|
1176
1194
|
if (rc === EXIT_SIGTERM || signal === "SIGTERM")
|
|
1177
1195
|
return sandboxFault("timeout", "the box exceeded its time limit (reaped by kern's timeout backstop)");
|
|
1178
1196
|
// Box-not-started: a non-zero exit whose stderr carries kern's OWN setup markers (printed by the
|
|
@@ -1637,6 +1655,10 @@ class Kernel {
|
|
|
1637
1655
|
this._waiters = []; // FIFO of { resolve, timer }; one reply per request keeps them in order
|
|
1638
1656
|
this._stderr = Buffer.alloc(0);
|
|
1639
1657
|
this._dead = false;
|
|
1658
|
+
// kern's memory-cap enforcement byte (2nd byte of KERN_STARTED_FD). For a RESIDENT box kern writes
|
|
1659
|
+
// it only at box teardown (a cell kills the kernel), so it arrives ~concurrent with the death we
|
|
1660
|
+
// detect on stdout; read once, bounded, on death (`_readCapSignal`). 0 = undetermined / old kern.
|
|
1661
|
+
this._capSignal = 0;
|
|
1640
1662
|
}
|
|
1641
1663
|
|
|
1642
1664
|
async _open() {
|
|
@@ -1648,14 +1670,21 @@ class Kernel {
|
|
|
1648
1670
|
this._name = uniqueName();
|
|
1649
1671
|
this._childEnv = { ...process.env };
|
|
1650
1672
|
if (!sbx.enforceLimits) this._childEnv.KERN_NO_SCOPE = "1";
|
|
1673
|
+
this._childEnv.KERN_STARTED_FD = "3"; // same unforgeable channel; here for the enforcement byte only
|
|
1651
1674
|
const argv = [
|
|
1652
1675
|
...sbx._baseArgv(this._name, { network: sbx.network, timeoutS: KERNEL_BACKSTOP_S }),
|
|
1653
1676
|
"--", "python3", "-S", `${WORKSPACE}/${this._driver}`,
|
|
1654
1677
|
];
|
|
1655
|
-
// detached: own process group so we can killpg the box + kern as a unit, like _spawn.
|
|
1678
|
+
// detached: own process group so we can killpg the box + kern as a unit, like _spawn. fd 3 carries
|
|
1679
|
+
// the started/enforcement bytes; the workload never holds it.
|
|
1656
1680
|
this._child = spawn(argv[0], argv.slice(1), {
|
|
1657
|
-
env: this._childEnv, detached: true, stdio: ["pipe", "pipe", "pipe"],
|
|
1681
|
+
env: this._childEnv, detached: true, stdio: ["pipe", "pipe", "pipe", "pipe"],
|
|
1658
1682
|
});
|
|
1683
|
+
const startedCh = this._child.stdio[3];
|
|
1684
|
+
if (startedCh) {
|
|
1685
|
+
startedCh.on("data", (b) => { if (b.length >= 2) this._capSignal = b[1]; });
|
|
1686
|
+
startedCh.on("error", () => {});
|
|
1687
|
+
}
|
|
1659
1688
|
this._child.on("error", () => { this._dead = true; this._flush(null); });
|
|
1660
1689
|
this._child.on("close", () => { this._dead = true; this._flush(null); });
|
|
1661
1690
|
this._child.stdout.on("data", (d) => this._onData(d));
|
|
@@ -1754,8 +1783,8 @@ class Kernel {
|
|
|
1754
1783
|
return this._teardownResult("killed", `the kernel reply exceeded the ${this._cap}-byte cap`, started);
|
|
1755
1784
|
if (reply === null) {
|
|
1756
1785
|
const err = this._stderr.toString("utf8");
|
|
1757
|
-
const kind =
|
|
1758
|
-
return this._teardownResult(kind, err.trim() ||
|
|
1786
|
+
const [kind, dflt] = this._kernelDeathFault(err, await this._readCapSignal());
|
|
1787
|
+
return this._teardownResult(kind, err.trim() || dflt, started);
|
|
1759
1788
|
}
|
|
1760
1789
|
return this._resultFromReply(reply, started);
|
|
1761
1790
|
}
|
|
@@ -1801,6 +1830,43 @@ class Kernel {
|
|
|
1801
1830
|
});
|
|
1802
1831
|
}
|
|
1803
1832
|
|
|
1833
|
+
/** Why the resident kernel box died mid-cell, as `[type, defaultMessage]`. A kern setup marker on
|
|
1834
|
+
* stderr means it never came up (startup_failed). Otherwise this is the runCode counterpart of the
|
|
1835
|
+
* one-shot _classify SIGKILL branch - a kernel death has no per-cell exit code. `capSignal` is kern's
|
|
1836
|
+
* unforgeable enforcement byte (0 = old kern / undetermined, 1 = memory cap enforced, 2 = requested
|
|
1837
|
+
* but NOT enforced): with a memoryMb cap AND not-reported-unenforced (`!== 2`), the cgroup OOM-killer
|
|
1838
|
+
* is the cause -> `oom`; when kern reports the cap did not bind (2), the kill is not attributable to
|
|
1839
|
+
* the box's cgroup -> `killed`; uncapped is also `killed`. */
|
|
1840
|
+
_kernelDeathFault(err, capSignal = 0) {
|
|
1841
|
+
if (looksLikeStartupFailure(err)) return ["startup_failed", "the kernel box failed to start"];
|
|
1842
|
+
if (this._sbx.memoryMb !== null && capSignal !== 2)
|
|
1843
|
+
return ["oom", "the kernel box was OOM-killed (it exceeded its memory cap)"];
|
|
1844
|
+
if (capSignal === 2)
|
|
1845
|
+
return [
|
|
1846
|
+
"killed",
|
|
1847
|
+
"the kernel box was SIGKILLed, but its memory cap was not enforced here (no cgroup delegation), so it is not attributed to a cgroup OOM",
|
|
1848
|
+
];
|
|
1849
|
+
return ["killed", "the kernel box exited"];
|
|
1850
|
+
}
|
|
1851
|
+
|
|
1852
|
+
/** kern's memory-cap enforcement byte for the resident box, read ONCE on kernel death. kern writes
|
|
1853
|
+
* the two-byte KERN_STARTED_FD signal only at box teardown (a resident box exits when a cell kills it),
|
|
1854
|
+
* ~concurrent with the death detected on stdout. The fd-3 `data` handler in `_open` records the byte
|
|
1855
|
+
* as it arrives; this awaits a BOUNDED window (the fd's own `end`, or 1 s) so the read is deterministic
|
|
1856
|
+
* rather than a race, then returns the byte (0 = EOF / old kern / not yet -> the memoryMb heuristic). */
|
|
1857
|
+
async _readCapSignal() {
|
|
1858
|
+
const ch = this._child && this._child.stdio && this._child.stdio[3];
|
|
1859
|
+
if (!ch) return 0;
|
|
1860
|
+
if (this._capSignal === 0 && !ch.destroyed) {
|
|
1861
|
+
await new Promise((res) => {
|
|
1862
|
+
const t = setTimeout(res, 1000);
|
|
1863
|
+
ch.once("end", () => { clearTimeout(t); res(); });
|
|
1864
|
+
ch.once("error", () => { clearTimeout(t); res(); });
|
|
1865
|
+
});
|
|
1866
|
+
}
|
|
1867
|
+
return this._capSignal;
|
|
1868
|
+
}
|
|
1869
|
+
|
|
1804
1870
|
_teardownResult(type, message, started) {
|
|
1805
1871
|
this._kill();
|
|
1806
1872
|
// Same rule as the one-shot path: a box that never STARTED (the kernel failed to boot) throws, it
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "kern-sandbox",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.21",
|
|
4
4
|
"description": "kern is a fast, rootless sandbox and virtual resource runtime for any workload, including untrusted and AI-generated code; kern-sandbox is its Node/TypeScript binding. Run untrusted or agent-generated code (Python/JS/Bash) in a real, kernel-enforced box in single-digit milliseconds, with no cloud, no account and no VM.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"sandbox",
|