@skydiveai/pi-extensions 0.1.0-beta.1233 → 0.1.0-beta.1235
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.mjs +414 -261
- package/package.json +1 -1
package/dist/index.mjs
CHANGED
|
@@ -19,11 +19,11 @@ import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-http";
|
|
|
19
19
|
import { Resource } from "@opentelemetry/resources";
|
|
20
20
|
import { BatchSpanProcessor, NodeTracerProvider } from "@opentelemetry/sdk-trace-node";
|
|
21
21
|
import { ATTR_SERVICE_NAME } from "@opentelemetry/semantic-conventions";
|
|
22
|
-
import { hc } from "hono/client";
|
|
23
|
-
import { parse } from "yaml";
|
|
24
22
|
import { execFile } from "node:child_process";
|
|
25
23
|
import { availableParallelism } from "node:os";
|
|
26
24
|
import { promisify } from "node:util";
|
|
25
|
+
import { hc } from "hono/client";
|
|
26
|
+
import { parse } from "yaml";
|
|
27
27
|
import { quote } from "shell-quote";
|
|
28
28
|
import { createWriteStream } from "node:fs";
|
|
29
29
|
import { finished } from "node:stream/promises";
|
|
@@ -259,7 +259,7 @@ function createHealthHandler({ metadata }) {
|
|
|
259
259
|
* read on the hot path before every LLM call), it falls back to the default
|
|
260
260
|
* for that knob and logs once.
|
|
261
261
|
*/
|
|
262
|
-
const log$
|
|
262
|
+
const log$15 = logger.child({ module: "context-management-config" });
|
|
263
263
|
const DEFAULT_CONTEXT_MANAGEMENT_CONFIG = {
|
|
264
264
|
enabled: false,
|
|
265
265
|
perResultMaxBytes: 16 * 1024,
|
|
@@ -307,7 +307,7 @@ function resolveContextManagementConfig(env = process.env) {
|
|
|
307
307
|
maxModelCallsPerTurn: env.SKYDIVE_CTX_MAX_MODEL_CALLS
|
|
308
308
|
});
|
|
309
309
|
if (!parsed.success) {
|
|
310
|
-
log$
|
|
310
|
+
log$15.warn({
|
|
311
311
|
event: "context_management_config_invalid",
|
|
312
312
|
err: parsed.error
|
|
313
313
|
}, "falling back to default context-management config");
|
|
@@ -441,7 +441,7 @@ const CAPABILITY_SOUL_NUDGE = "New capability gained — once the current task i
|
|
|
441
441
|
* or `ToolDefinition[]`. Files starting with `_` or `.` are skipped, so
|
|
442
442
|
* `tools/_example.ts` documents the shape without registering.
|
|
443
443
|
*/
|
|
444
|
-
const log$
|
|
444
|
+
const log$14 = logger.child({ module: "local-tools-extension" });
|
|
445
445
|
const TOOLS_DIRNAME = "tools";
|
|
446
446
|
const fileState = /* @__PURE__ */ new Map();
|
|
447
447
|
let pendingLocalToolsUpdate = null;
|
|
@@ -585,7 +585,7 @@ async function reconcileAndQueue({ pi, dir, reason }) {
|
|
|
585
585
|
dir
|
|
586
586
|
});
|
|
587
587
|
if (reason !== "session_start" && summaryHasChanges$1(summary)) pendingLocalToolsUpdate = summary;
|
|
588
|
-
log$
|
|
588
|
+
log$14.info({
|
|
589
589
|
event: "local_tools_reconcile",
|
|
590
590
|
reason,
|
|
591
591
|
total_tools: summary.totalTools,
|
|
@@ -607,7 +607,7 @@ const localToolsExtension = (pi) => {
|
|
|
607
607
|
reason: "session_start"
|
|
608
608
|
});
|
|
609
609
|
} catch (err) {
|
|
610
|
-
log$
|
|
610
|
+
log$14.error({
|
|
611
611
|
err,
|
|
612
612
|
event: "local_tools_reconcile_failed"
|
|
613
613
|
}, "local tools reconcile failed");
|
|
@@ -619,7 +619,7 @@ const localToolsExtension = (pi) => {
|
|
|
619
619
|
try {
|
|
620
620
|
current = await listToolFiles(dir);
|
|
621
621
|
} catch (err) {
|
|
622
|
-
log$
|
|
622
|
+
log$14.warn({
|
|
623
623
|
err,
|
|
624
624
|
event: "local_tools_listing_failed"
|
|
625
625
|
}, "tools/ listing failed");
|
|
@@ -641,7 +641,7 @@ const localToolsExtension = (pi) => {
|
|
|
641
641
|
reason: "auto_reload"
|
|
642
642
|
});
|
|
643
643
|
} catch (err) {
|
|
644
|
-
log$
|
|
644
|
+
log$14.error({
|
|
645
645
|
err,
|
|
646
646
|
event: "local_tools_auto_reload_failed"
|
|
647
647
|
}, "auto-reload after tools/ change failed");
|
|
@@ -879,7 +879,7 @@ async function loadMcpConfig(path) {
|
|
|
879
879
|
* Clients are keyed by JSON-stringified config and reused across
|
|
880
880
|
* reloads — only changed configs reconnect.
|
|
881
881
|
*/
|
|
882
|
-
const log$
|
|
882
|
+
const log$13 = logger.child({ module: "mcp-extension" });
|
|
883
883
|
async function closeConnected(connected) {
|
|
884
884
|
try {
|
|
885
885
|
await connected.client.close();
|
|
@@ -1301,7 +1301,7 @@ var McpExtension = class {
|
|
|
1301
1301
|
});
|
|
1302
1302
|
this.lastConfigMtimeMs = await readConfigMtimeMs(configPath);
|
|
1303
1303
|
if (reason !== "session_start" && summaryHasChanges(summary)) this.pendingMcpUpdate = summary;
|
|
1304
|
-
log$
|
|
1304
|
+
log$13.info({
|
|
1305
1305
|
event: "mcp_reconcile",
|
|
1306
1306
|
reason,
|
|
1307
1307
|
total_tools: summary.totalTools,
|
|
@@ -1324,7 +1324,7 @@ var McpExtension = class {
|
|
|
1324
1324
|
reason: "session_start"
|
|
1325
1325
|
});
|
|
1326
1326
|
} catch (err) {
|
|
1327
|
-
log$
|
|
1327
|
+
log$13.error({
|
|
1328
1328
|
err,
|
|
1329
1329
|
event: "mcp_reconcile_failed"
|
|
1330
1330
|
}, "MCP reconcile failed");
|
|
@@ -1336,7 +1336,7 @@ var McpExtension = class {
|
|
|
1336
1336
|
try {
|
|
1337
1337
|
mtime = await readConfigMtimeMs(configPath);
|
|
1338
1338
|
} catch (err) {
|
|
1339
|
-
log$
|
|
1339
|
+
log$13.warn({
|
|
1340
1340
|
err,
|
|
1341
1341
|
event: "mcp_mtime_check_failed"
|
|
1342
1342
|
}, "mtime check on mcp.config.json failed");
|
|
@@ -1350,7 +1350,7 @@ var McpExtension = class {
|
|
|
1350
1350
|
reason: "auto_reload"
|
|
1351
1351
|
});
|
|
1352
1352
|
} catch (err) {
|
|
1353
|
-
log$
|
|
1353
|
+
log$13.error({
|
|
1354
1354
|
err,
|
|
1355
1355
|
event: "mcp_auto_reload_failed"
|
|
1356
1356
|
}, "auto-reload after mcp.config.json change failed");
|
|
@@ -1713,6 +1713,370 @@ const bashDefaultTimeoutExtension = (pi) => {
|
|
|
1713
1713
|
});
|
|
1714
1714
|
};
|
|
1715
1715
|
//#endregion
|
|
1716
|
+
//#region src/extensions/resource-pressure-warning.ts
|
|
1717
|
+
/**
|
|
1718
|
+
* Mid-run resource-pressure warning to the agent.
|
|
1719
|
+
*
|
|
1720
|
+
* The sandbox already detects pressure — the boot scripts cap the
|
|
1721
|
+
* user-workload cgroup (memory.high/memory.max) and watchers log warn/crit
|
|
1722
|
+
* edges for memory and disk — but nothing told the *agent*, so a turn burned
|
|
1723
|
+
* straight to the OOM kill (or a full disk) and only learned about it from
|
|
1724
|
+
* the post-mortem notice. This extension closes that gap in-process: while a
|
|
1725
|
+
* turn is active it polls the agent cgroup and the root filesystem and, the
|
|
1726
|
+
* first time usage crosses a warn threshold, folds a system notification into
|
|
1727
|
+
* the open turn so the agent can checkpoint, shed work (constrain
|
|
1728
|
+
* parallelism, kill a background hog, clean scratch space), or request a
|
|
1729
|
+
* bigger tier BEFORE the kill.
|
|
1730
|
+
*
|
|
1731
|
+
* The notification is triggered by the two conditions that actually kill
|
|
1732
|
+
* work — memory near the cgroup hard cap, disk near full — and reports a
|
|
1733
|
+
* snapshot of all the relevant stats (memory, CPU utilization, disk) so the
|
|
1734
|
+
* agent can tell which resource is the problem and how much headroom the
|
|
1735
|
+
* others have.
|
|
1736
|
+
*
|
|
1737
|
+
* Edge-triggered, once per trigger per turn: the fired flags reset on
|
|
1738
|
+
* agent_start, so a turn that rides a threshold gets one warning per
|
|
1739
|
+
* resource, not a stream. Polling only runs while the agent is active — an
|
|
1740
|
+
* idle sandbox's resource usage is not the agent's problem and there is no
|
|
1741
|
+
* open turn to deliver into anyway.
|
|
1742
|
+
*
|
|
1743
|
+
* Best-effort throughout: any read failure (cgroup absent, controller not
|
|
1744
|
+
* delegated, non-cgroup-v2 host, df missing) reads as "no signal" for that
|
|
1745
|
+
* stat and the extension warns on what it can see — it must never break a
|
|
1746
|
+
* turn over an observability feature.
|
|
1747
|
+
*/
|
|
1748
|
+
const execFileAsync = promisify(execFile);
|
|
1749
|
+
const log$12 = logger.child({ module: "resource-pressure-warning" });
|
|
1750
|
+
const POLL_INTERVAL_MS = 1e4;
|
|
1751
|
+
function envOverride(name) {
|
|
1752
|
+
for (const prefix of ["SKYDIVE_", "ANYONE_"]) {
|
|
1753
|
+
const value = process.env[`${prefix}${name}`];
|
|
1754
|
+
if (value != null && value !== "") return value;
|
|
1755
|
+
}
|
|
1756
|
+
return null;
|
|
1757
|
+
}
|
|
1758
|
+
function cgroupDir() {
|
|
1759
|
+
return envOverride("AGENT_CGROUP") ?? "/sys/fs/cgroup/agent";
|
|
1760
|
+
}
|
|
1761
|
+
function diskRoot() {
|
|
1762
|
+
return envOverride("DISK_ROOT") ?? "/";
|
|
1763
|
+
}
|
|
1764
|
+
/**
|
|
1765
|
+
* Read a cgroup v2 scalar file. Returns a number, or null for "max"
|
|
1766
|
+
* (uncapped), an empty/absent file, or any read/parse error — an uncapped or
|
|
1767
|
+
* unreadable limit means there is nothing meaningful to warn against.
|
|
1768
|
+
*/
|
|
1769
|
+
async function readScalar(file) {
|
|
1770
|
+
try {
|
|
1771
|
+
const raw = (await readFile(`${cgroupDir()}/${file}`, "utf8")).trim();
|
|
1772
|
+
if (raw === "" || raw === "max") return null;
|
|
1773
|
+
const n = Number(raw);
|
|
1774
|
+
return Number.isFinite(n) ? n : null;
|
|
1775
|
+
} catch (_error) {
|
|
1776
|
+
return null;
|
|
1777
|
+
}
|
|
1778
|
+
}
|
|
1779
|
+
/**
|
|
1780
|
+
* Read a cgroup v2 "flat keyed" file (one `key value` pair per line, e.g.
|
|
1781
|
+
* cpu.stat) and return the counter for `key`, or null when absent.
|
|
1782
|
+
*/
|
|
1783
|
+
async function readKeyedCounter(file, key) {
|
|
1784
|
+
try {
|
|
1785
|
+
const raw = await readFile(`${cgroupDir()}/${file}`, "utf8");
|
|
1786
|
+
for (const line of raw.split("\n")) {
|
|
1787
|
+
const [k, v] = line.trim().split(/\s+/);
|
|
1788
|
+
if (k === key) {
|
|
1789
|
+
const n = Number(v);
|
|
1790
|
+
return Number.isFinite(n) ? n : null;
|
|
1791
|
+
}
|
|
1792
|
+
}
|
|
1793
|
+
return null;
|
|
1794
|
+
} catch (_error) {
|
|
1795
|
+
return null;
|
|
1796
|
+
}
|
|
1797
|
+
}
|
|
1798
|
+
/**
|
|
1799
|
+
* Live memory usage as an integer percent of the hard cap, or null when
|
|
1800
|
+
* either side is unreadable/uncapped. Exported for tests.
|
|
1801
|
+
*/
|
|
1802
|
+
async function readMemUsePct() {
|
|
1803
|
+
const [current, max] = await Promise.all([readScalar("memory.current"), readScalar("memory.max")]);
|
|
1804
|
+
if (current === null || max === null || max <= 0) return null;
|
|
1805
|
+
return {
|
|
1806
|
+
pct: Math.floor(current / max * 100),
|
|
1807
|
+
currentBytes: current,
|
|
1808
|
+
maxBytes: max
|
|
1809
|
+
};
|
|
1810
|
+
}
|
|
1811
|
+
/**
|
|
1812
|
+
* Root filesystem used% (df -P Capacity column), or null on any failure.
|
|
1813
|
+
* Exported for tests.
|
|
1814
|
+
*/
|
|
1815
|
+
async function readDiskUsePct() {
|
|
1816
|
+
try {
|
|
1817
|
+
const { stdout } = await execFileAsync("df", ["-P", diskRoot()]);
|
|
1818
|
+
const dataRow = stdout.trim().split("\n")[1];
|
|
1819
|
+
if (dataRow == null) return null;
|
|
1820
|
+
const capacity = dataRow.trim().split(/\s+/)[4];
|
|
1821
|
+
if (capacity == null) return null;
|
|
1822
|
+
const pct = Number(capacity.replace("%", ""));
|
|
1823
|
+
return Number.isFinite(pct) ? pct : null;
|
|
1824
|
+
} catch (_error) {
|
|
1825
|
+
return null;
|
|
1826
|
+
}
|
|
1827
|
+
}
|
|
1828
|
+
/**
|
|
1829
|
+
* CPU utilization sampler. cgroup v2 exposes cumulative CPU time
|
|
1830
|
+
* (cpu.stat usage_usec); utilization is the delta between two samples over
|
|
1831
|
+
* the wall time between them, normalized by core count. The first call after
|
|
1832
|
+
* construction has no previous sample and returns null.
|
|
1833
|
+
*/
|
|
1834
|
+
function createCpuSampler() {
|
|
1835
|
+
let prevUsageUsec = null;
|
|
1836
|
+
let prevAtMs = null;
|
|
1837
|
+
return async () => {
|
|
1838
|
+
const usage = await readKeyedCounter("cpu.stat", "usage_usec");
|
|
1839
|
+
const now = Date.now();
|
|
1840
|
+
const prev = prevUsageUsec;
|
|
1841
|
+
const prevAt = prevAtMs;
|
|
1842
|
+
prevUsageUsec = usage;
|
|
1843
|
+
prevAtMs = now;
|
|
1844
|
+
if (usage === null || prev === null || prevAt === null) return null;
|
|
1845
|
+
const wallUsec = (now - prevAt) * 1e3;
|
|
1846
|
+
if (wallUsec <= 0) return null;
|
|
1847
|
+
const cores = availableParallelism();
|
|
1848
|
+
const pct = Math.round((usage - prev) / (wallUsec * cores) * 100);
|
|
1849
|
+
return Math.max(0, Math.min(100, pct));
|
|
1850
|
+
};
|
|
1851
|
+
}
|
|
1852
|
+
function fmtMb(bytes) {
|
|
1853
|
+
return Math.round(bytes / 1024 / 1024);
|
|
1854
|
+
}
|
|
1855
|
+
/** The model-facing warning text. Exported for tests. */
|
|
1856
|
+
function resourcePressureWarningText(trigger, { mem, cpuPct, diskPct }) {
|
|
1857
|
+
const stats = [];
|
|
1858
|
+
if (mem) stats.push(`memory ${mem.pct}% of cap (${fmtMb(mem.currentBytes)}/${fmtMb(mem.maxBytes)} MB)`);
|
|
1859
|
+
if (cpuPct !== null) stats.push(`CPU ${cpuPct}%`);
|
|
1860
|
+
if (diskPct !== null) stats.push(`disk ${diskPct}% full`);
|
|
1861
|
+
const lead = trigger === "memory" ? `Your sandbox is at ${mem?.pct}% of its memory cap. If usage keeps climbing, the kernel will kill the offending process and this turn may die with it.` : `Your sandbox's disk is ${diskPct}% full. If it fills completely, writes will start failing and this turn may die with them.`;
|
|
1862
|
+
const remedy = trigger === "memory" ? "checkpoint in-flight work (commit and push), then reduce the footprint — constrain parallelism, run heavy steps sequentially, or kill background processes you no longer need." : "checkpoint in-flight work (commit and push), then free space — clean build artifacts, caches, and scratch files you no longer need.";
|
|
1863
|
+
return `<system_notification>${lead} Current usage: ${stats.join(", ")}. Act now: ${remedy} If the workload genuinely needs more resources, request a bigger sandbox with \`platform compute request\`. This is an automated resource warning, not a message from the user; continue the task, adjusted.</system_notification>`;
|
|
1864
|
+
}
|
|
1865
|
+
const resourcePressureWarningExtension = (pi) => {
|
|
1866
|
+
let agentActive = false;
|
|
1867
|
+
let warnedMemThisTurn = false;
|
|
1868
|
+
let warnedDiskThisTurn = false;
|
|
1869
|
+
let timer = null;
|
|
1870
|
+
const sampleCpu = createCpuSampler();
|
|
1871
|
+
async function checkOnce() {
|
|
1872
|
+
if (!agentActive || warnedMemThisTurn && warnedDiskThisTurn) return;
|
|
1873
|
+
const [mem, cpuPct, diskPct] = await Promise.all([
|
|
1874
|
+
readMemUsePct(),
|
|
1875
|
+
sampleCpu(),
|
|
1876
|
+
readDiskUsePct()
|
|
1877
|
+
]);
|
|
1878
|
+
let trigger = null;
|
|
1879
|
+
if (!warnedMemThisTurn && mem !== null && mem.pct >= 80) {
|
|
1880
|
+
trigger = "memory";
|
|
1881
|
+
warnedMemThisTurn = true;
|
|
1882
|
+
} else if (!warnedDiskThisTurn && diskPct !== null && diskPct >= 80) {
|
|
1883
|
+
trigger = "disk";
|
|
1884
|
+
warnedDiskThisTurn = true;
|
|
1885
|
+
}
|
|
1886
|
+
if (trigger === null) return;
|
|
1887
|
+
log$12.warn({
|
|
1888
|
+
trigger,
|
|
1889
|
+
mem,
|
|
1890
|
+
cpuPct,
|
|
1891
|
+
diskPct
|
|
1892
|
+
}, "resource pressure warning delivered to agent");
|
|
1893
|
+
await pi.sendMessage({
|
|
1894
|
+
customType: "anyone-resource-pressure-warning",
|
|
1895
|
+
content: resourcePressureWarningText(trigger, {
|
|
1896
|
+
mem,
|
|
1897
|
+
cpuPct,
|
|
1898
|
+
diskPct
|
|
1899
|
+
}),
|
|
1900
|
+
display: false
|
|
1901
|
+
}, {
|
|
1902
|
+
triggerTurn: true,
|
|
1903
|
+
deliverAs: "followUp"
|
|
1904
|
+
});
|
|
1905
|
+
}
|
|
1906
|
+
pi.on("agent_start", async () => {
|
|
1907
|
+
agentActive = true;
|
|
1908
|
+
warnedMemThisTurn = false;
|
|
1909
|
+
warnedDiskThisTurn = false;
|
|
1910
|
+
if (!timer) {
|
|
1911
|
+
timer = setInterval(() => {
|
|
1912
|
+
checkOnce().catch((err) => {
|
|
1913
|
+
log$12.error({ err }, "resource pressure check failed");
|
|
1914
|
+
});
|
|
1915
|
+
}, POLL_INTERVAL_MS);
|
|
1916
|
+
timer.unref?.();
|
|
1917
|
+
}
|
|
1918
|
+
});
|
|
1919
|
+
pi.on("agent_end", async () => {
|
|
1920
|
+
agentActive = false;
|
|
1921
|
+
if (timer) {
|
|
1922
|
+
clearInterval(timer);
|
|
1923
|
+
timer = null;
|
|
1924
|
+
}
|
|
1925
|
+
});
|
|
1926
|
+
};
|
|
1927
|
+
//#endregion
|
|
1928
|
+
//#region src/extensions/disk-guard.ts
|
|
1929
|
+
const log$11 = logger.child({ module: "disk-guard" });
|
|
1930
|
+
/**
|
|
1931
|
+
* In-band bypass. The guard is a safety net, not a jail: when the agent knows
|
|
1932
|
+
* a flagged command is genuinely safe (writing to a different mount, a tiny
|
|
1933
|
+
* bounded download, a delete-then-clone one-liner, an emergency it accepts the
|
|
1934
|
+
* risk on) it can force the command through by appending this marker as a
|
|
1935
|
+
* trailing shell comment. Kept as a comment so it never changes what the
|
|
1936
|
+
* command does, and matched case-insensitively with flexible spacing so the
|
|
1937
|
+
* agent doesn't have to reproduce it byte-for-byte.
|
|
1938
|
+
*/
|
|
1939
|
+
const BYPASS_MARKER = /#\s*disk-guard:\s*allow\b/i;
|
|
1940
|
+
/** The exact marker text the block message tells the agent to append. */
|
|
1941
|
+
const BYPASS_HINT = "# disk-guard: allow";
|
|
1942
|
+
/**
|
|
1943
|
+
* Harness-level kill switch: set DISK_GUARD_DISABLE=1 to turn the guard off
|
|
1944
|
+
* entirely. This is the "I own my harness, let me opt out" knob — an agent
|
|
1945
|
+
* that boots its own harness can disable the guard for its whole process
|
|
1946
|
+
* without a code roll, and it's also the fleet-wide escape hatch if the
|
|
1947
|
+
* classifier ever misfires and blocks real work. The bare name is honored
|
|
1948
|
+
* first; the SKYDIVE_/ANYONE_ prefixes are accepted too for consistency with
|
|
1949
|
+
* the other env overrides. Empty/unset/"0"/"false" leave the guard on.
|
|
1950
|
+
*/
|
|
1951
|
+
function guardDisabledByEnv() {
|
|
1952
|
+
for (const name of [
|
|
1953
|
+
"DISK_GUARD_DISABLE",
|
|
1954
|
+
"SKYDIVE_DISK_GUARD_DISABLE",
|
|
1955
|
+
"ANYONE_DISK_GUARD_DISABLE"
|
|
1956
|
+
]) {
|
|
1957
|
+
const value = process.env[name];
|
|
1958
|
+
if (value != null && value !== "" && value !== "0" && value !== "false") return true;
|
|
1959
|
+
}
|
|
1960
|
+
return false;
|
|
1961
|
+
}
|
|
1962
|
+
/** True when the command carries the in-band bypass marker. */
|
|
1963
|
+
function hasBypassMarker(command) {
|
|
1964
|
+
return BYPASS_MARKER.test(command);
|
|
1965
|
+
}
|
|
1966
|
+
/**
|
|
1967
|
+
* Commands that reclaim space or merely inspect it. If any of these verbs
|
|
1968
|
+
* appears in the command line, we never block — otherwise the guard would trap
|
|
1969
|
+
* the agent by blocking the exact command it needs to dig out. Matched as
|
|
1970
|
+
* whole words so `remove-item` etc. don't accidentally match `rm`.
|
|
1971
|
+
*/
|
|
1972
|
+
const RECLAIM_PATTERNS = [
|
|
1973
|
+
/\brm\b/,
|
|
1974
|
+
/\brmdir\b/,
|
|
1975
|
+
/\bdf\b/,
|
|
1976
|
+
/\bdu\b/,
|
|
1977
|
+
/\bncdu\b/,
|
|
1978
|
+
/\bfind\b[^|]*\s-delete\b/,
|
|
1979
|
+
/\btruncate\b/,
|
|
1980
|
+
/\bgit\s+(gc|prune|clean|worktree\s+remove|worktree\s+prune)\b/,
|
|
1981
|
+
/\b(yarn|npm|pnpm|bun)\s+.*\b(cache\s+clean|cache\s+clear|store\s+prune)\b/,
|
|
1982
|
+
/\bcache\s+(clean|clear|prune)\b/,
|
|
1983
|
+
/\b(docker|podman)\s+.*\bprune\b/,
|
|
1984
|
+
/\bapt(-get)?\s+clean\b/,
|
|
1985
|
+
/\bjournalctl\b[^|]*--vacuum/
|
|
1986
|
+
];
|
|
1987
|
+
/**
|
|
1988
|
+
* File extensions that mean a download is actually LARGE — archives, disk
|
|
1989
|
+
* images, compiled/binary artifacts, model weights, media. A curl/wget is only
|
|
1990
|
+
* gated when it writes one of these; an API/page fetch to a `.json`/`.html`/
|
|
1991
|
+
* `.txt` file is tiny and must not be blocked. Derived from 4,144 real
|
|
1992
|
+
* commands: ~64% of `curl -o` uses were tiny fetches, only ~4% large.
|
|
1993
|
+
*/
|
|
1994
|
+
const BIG_DOWNLOAD_EXT = "(?:tar\\.gz|tgz|tar|zip|iso|gz|bz2|xz|zst|deb|rpm|pkg|dmg|whl|jar|7z|img|mp4|mov|avi|mkv|onnx|gguf|safetensors|bin|node)";
|
|
1995
|
+
/**
|
|
1996
|
+
* Commands that consume a meaningful amount of disk. Kept deliberately tight
|
|
1997
|
+
* and high-precision: validated against 4,144 real commands from the last 7
|
|
1998
|
+
* days, the earlier "writes a file" heuristic flagged 82% of everything (a
|
|
1999
|
+
* `curl -o /tmp/x.json` API call is not a disk event). This set flags ~33%,
|
|
2000
|
+
* almost all genuinely large — real installs, clones, big-archive downloads,
|
|
2001
|
+
* extractions. What was DROPPED and why:
|
|
2002
|
+
* - `git fetch` / `git pull` — incremental on an existing clone, usually tiny.
|
|
2003
|
+
* - `git checkout` — overwhelmingly `git checkout <ref> -- <file>` or a
|
|
2004
|
+
* branch switch, ~zero net growth; the rare full materialization isn't
|
|
2005
|
+
* worth the false-positive rate.
|
|
2006
|
+
* - bare `curl -o` / `wget -o` — see BIG_DOWNLOAD_EXT above.
|
|
2007
|
+
* - loose `… build` — matched `--mode=skip-build`, `oxfmt … build`, prose.
|
|
2008
|
+
* The remaining big-disk op in escher is `git clone` and `git worktree add`
|
|
2009
|
+
* (which is really a checkout), both kept.
|
|
2010
|
+
*/
|
|
2011
|
+
const SPACE_HUNGRY_PATTERNS = [
|
|
2012
|
+
/\bgit\s+clone\b/,
|
|
2013
|
+
/\bgit\s+worktree\s+add\b/,
|
|
2014
|
+
/\b(yarn|npm|pnpm|bun)\s+(install|add|ci)\b/,
|
|
2015
|
+
/\byarn\s*$/,
|
|
2016
|
+
/\byarn\s+--(?!version|help)\S/,
|
|
2017
|
+
/\bpip3?\s+install\b/,
|
|
2018
|
+
/\bapt(-get)?\s+install\b/,
|
|
2019
|
+
/\bnpm\s+pack\b/,
|
|
2020
|
+
/\bdocker\s+(build|pull)\b/,
|
|
2021
|
+
new RegExp(`\\b(?:curl|wget)\\b[^\\n]*\\s-[a-zA-Z]*[oO]\\b[^\\n]*\\.${BIG_DOWNLOAD_EXT}\\b`, "i"),
|
|
2022
|
+
new RegExp(`\\b(?:curl|wget)\\b[^\\n]*\\.${BIG_DOWNLOAD_EXT}\\b[^\\n]*\\s-[a-zA-Z]*[oO]\\b`, "i"),
|
|
2023
|
+
/\btar\s+[^\n|]*x[^\n|]*f/,
|
|
2024
|
+
/\bunzip\b/,
|
|
2025
|
+
/\bdd\b[^\n|]*\bof=/
|
|
2026
|
+
];
|
|
2027
|
+
/**
|
|
2028
|
+
* True when the command reclaims or inspects space — these are always allowed,
|
|
2029
|
+
* even on a 100%-full box, so the agent can dig itself out.
|
|
2030
|
+
*/
|
|
2031
|
+
function isReclaimCommand(command) {
|
|
2032
|
+
return RECLAIM_PATTERNS.some((re) => re.test(command));
|
|
2033
|
+
}
|
|
2034
|
+
/**
|
|
2035
|
+
* True when the command is likely to consume a meaningful amount of disk.
|
|
2036
|
+
* A reclaim/inspect command is never space-hungry — the reclaim check wins so a
|
|
2037
|
+
* `git worktree remove` or a `yarn cache clean` is never mistaken for growth.
|
|
2038
|
+
*/
|
|
2039
|
+
function isSpaceHungryCommand(command) {
|
|
2040
|
+
if (isReclaimCommand(command)) return false;
|
|
2041
|
+
return SPACE_HUNGRY_PATTERNS.some((re) => re.test(command));
|
|
2042
|
+
}
|
|
2043
|
+
/**
|
|
2044
|
+
* The decision, factored out and pure so it's exhaustively testable without a
|
|
2045
|
+
* real filesystem. Block only when we have a disk reading, it's at/above the
|
|
2046
|
+
* critical threshold, the command is space-hungry (and not a reclaim), and the
|
|
2047
|
+
* agent hasn't explicitly opted out with the bypass marker.
|
|
2048
|
+
*/
|
|
2049
|
+
function shouldBlockForDisk(command, diskPct) {
|
|
2050
|
+
if (diskPct === null) return false;
|
|
2051
|
+
if (diskPct < 95) return false;
|
|
2052
|
+
if (hasBypassMarker(command)) return false;
|
|
2053
|
+
return isSpaceHungryCommand(command);
|
|
2054
|
+
}
|
|
2055
|
+
/** The agent-facing explanation returned as the blocked tool result. */
|
|
2056
|
+
function diskBlockReason(command, diskPct) {
|
|
2057
|
+
return `Blocked: the sandbox disk is ${diskPct}% full and this command (\`${command.trim().slice(0, 120)}\`) writes a large amount, so it would fail partway with ENOSPC and leave a corrupt result. Reclaim space FIRST, then retry. Free ONLY what THIS conversation created — scratch/build output you wrote this run, downloads you're done with, and worktrees/branches whose work you've already committed and pushed (\`git worktree remove\`, \`yarn cache clean\`, delete your own scratch). Do NOT blindly wipe /tmp or delete a clone/worktree you don't recognize — other conversations share this box. Check headroom with \`df -h /\` and \`du -sh ~/workspace/* 2>/dev/null\`. If you genuinely can't free enough, stop and tell the user you're blocked on disk rather than retrying the write. If you're certain this command is safe anyway (writes elsewhere, tiny bounded size, delete-then-write), force it through by appending \` ${BYPASS_HINT}\` to the command.`;
|
|
2058
|
+
}
|
|
2059
|
+
const diskGuardExtension = (pi) => {
|
|
2060
|
+
pi.on("tool_call", async (event) => {
|
|
2061
|
+
if (event.toolName !== "bash") return;
|
|
2062
|
+
if (guardDisabledByEnv()) return;
|
|
2063
|
+
const command = event.input.command;
|
|
2064
|
+
if (typeof command !== "string" || command.length === 0) return;
|
|
2065
|
+
if (hasBypassMarker(command)) return;
|
|
2066
|
+
if (!isSpaceHungryCommand(command)) return;
|
|
2067
|
+
const diskPct = await readDiskUsePct();
|
|
2068
|
+
if (!shouldBlockForDisk(command, diskPct)) return;
|
|
2069
|
+
log$11.warn({
|
|
2070
|
+
diskPct,
|
|
2071
|
+
command: command.slice(0, 200)
|
|
2072
|
+
}, "blocked space-hungry bash command on near-full disk");
|
|
2073
|
+
return {
|
|
2074
|
+
block: true,
|
|
2075
|
+
reason: diskBlockReason(command, diskPct)
|
|
2076
|
+
};
|
|
2077
|
+
});
|
|
2078
|
+
};
|
|
2079
|
+
//#endregion
|
|
1716
2080
|
//#region src/channel-context-ref.ts
|
|
1717
2081
|
/**
|
|
1718
2082
|
* The worker injects a small reference — `{ channel, messageId, runId }` —
|
|
@@ -1775,7 +2139,7 @@ const HEARTBEAT_THROTTLE_MS = 6e4;
|
|
|
1775
2139
|
const TOOL_HEARTBEAT_INTERVAL_MS = 5e3;
|
|
1776
2140
|
const MAX_TOOL_HEARTBEATS = 1440 * 60 * 1e3 / TOOL_HEARTBEAT_INTERVAL_MS;
|
|
1777
2141
|
const DAEMON_URL = "http://localhost:38994";
|
|
1778
|
-
const log$
|
|
2142
|
+
const log$10 = logger.child({ module: "platform-ext" });
|
|
1779
2143
|
function sandboxClient() {
|
|
1780
2144
|
const apiUrl = apiBaseUrl();
|
|
1781
2145
|
if (!apiUrl) return null;
|
|
@@ -1843,7 +2207,7 @@ async function fetchHarnessFlags() {
|
|
|
1843
2207
|
try {
|
|
1844
2208
|
const res = await client["feature-flags"].$get();
|
|
1845
2209
|
if (!res.ok) {
|
|
1846
|
-
log$
|
|
2210
|
+
log$10.debug({
|
|
1847
2211
|
status: res.status,
|
|
1848
2212
|
event: "feature_flags_fetch_failed"
|
|
1849
2213
|
}, "feature-flags fetch failed");
|
|
@@ -1851,7 +2215,7 @@ async function fetchHarnessFlags() {
|
|
|
1851
2215
|
}
|
|
1852
2216
|
return { contextManagement: (await res.json()).contextManagement ?? null };
|
|
1853
2217
|
} catch (err) {
|
|
1854
|
-
log$
|
|
2218
|
+
log$10.debug({
|
|
1855
2219
|
err,
|
|
1856
2220
|
event: "feature_flags_fetch_error"
|
|
1857
2221
|
}, "feature-flags request errored");
|
|
@@ -1862,7 +2226,7 @@ function postHeartbeat({ messageId }) {
|
|
|
1862
2226
|
const client = sandboxClient();
|
|
1863
2227
|
if (!client) return;
|
|
1864
2228
|
client.heartbeat.$post({ json: { messageId } }).catch((err) => {
|
|
1865
|
-
log$
|
|
2229
|
+
log$10.debug({
|
|
1866
2230
|
err,
|
|
1867
2231
|
event: "heartbeat_failed"
|
|
1868
2232
|
}, "heartbeat failed");
|
|
@@ -1874,7 +2238,7 @@ async function resolveConversationFromApi(messageId) {
|
|
|
1874
2238
|
try {
|
|
1875
2239
|
const res = await client["message-conversation"].$get({ query: { messageId } });
|
|
1876
2240
|
if (!res.ok) {
|
|
1877
|
-
log$
|
|
2241
|
+
log$10.warn({
|
|
1878
2242
|
status: res.status,
|
|
1879
2243
|
messageId,
|
|
1880
2244
|
event: "resolve_conversation_failed"
|
|
@@ -1883,7 +2247,7 @@ async function resolveConversationFromApi(messageId) {
|
|
|
1883
2247
|
}
|
|
1884
2248
|
return (await res.json()).conversationId ?? null;
|
|
1885
2249
|
} catch (err) {
|
|
1886
|
-
log$
|
|
2250
|
+
log$10.warn({
|
|
1887
2251
|
err,
|
|
1888
2252
|
messageId,
|
|
1889
2253
|
event: "resolve_conversation_error"
|
|
@@ -1968,7 +2332,7 @@ function createToolHeartbeat({ messageId }) {
|
|
|
1968
2332
|
}
|
|
1969
2333
|
heartbeatCount++;
|
|
1970
2334
|
if (heartbeatCount > MAX_TOOL_HEARTBEATS) {
|
|
1971
|
-
log$
|
|
2335
|
+
log$10.warn({
|
|
1972
2336
|
heartbeatCount,
|
|
1973
2337
|
activeToolCalls: [...activeToolCalls]
|
|
1974
2338
|
}, "tool heartbeat max reached, stopping");
|
|
@@ -1999,7 +2363,7 @@ function postToDaemon(path, body) {
|
|
|
1999
2363
|
headers: { "content-type": "application/json" },
|
|
2000
2364
|
body: JSON.stringify(body)
|
|
2001
2365
|
}).catch((err) => {
|
|
2002
|
-
log$
|
|
2366
|
+
log$10.debug({
|
|
2003
2367
|
err,
|
|
2004
2368
|
path,
|
|
2005
2369
|
event: "daemon_post_failed"
|
|
@@ -2008,7 +2372,7 @@ function postToDaemon(path, body) {
|
|
|
2008
2372
|
}
|
|
2009
2373
|
function createPlatformExtensions({ sessionId, channelContext }) {
|
|
2010
2374
|
return (pi) => {
|
|
2011
|
-
log$
|
|
2375
|
+
log$10.info({
|
|
2012
2376
|
sessionId,
|
|
2013
2377
|
hasChannelContext: Boolean(channelContext)
|
|
2014
2378
|
}, "platform extension initialized");
|
|
@@ -2057,7 +2421,7 @@ function createPlatformExtensions({ sessionId, channelContext }) {
|
|
|
2057
2421
|
});
|
|
2058
2422
|
});
|
|
2059
2423
|
pi.on("agent_end", () => {
|
|
2060
|
-
log$
|
|
2424
|
+
log$10.info({ sessionId }, "session ending");
|
|
2061
2425
|
postToDaemon("/session/end", { sessionId });
|
|
2062
2426
|
});
|
|
2063
2427
|
};
|
|
@@ -2083,7 +2447,7 @@ function createPlatformExtensions({ sessionId, channelContext }) {
|
|
|
2083
2447
|
* alive and an indeterminate result (no api url / transient failure) leaves the
|
|
2084
2448
|
* last-known values untouched so a blip can't silently flip behavior.
|
|
2085
2449
|
*/
|
|
2086
|
-
const log$
|
|
2450
|
+
const log$9 = logger.child({ module: "feature-flags-poll" });
|
|
2087
2451
|
const FLAG_POLL_INTERVAL_MS = 6e4;
|
|
2088
2452
|
let contextManagement = null;
|
|
2089
2453
|
const subscribers = { contextManagement: /* @__PURE__ */ new Set() };
|
|
@@ -2118,7 +2482,7 @@ function apply(name, next) {
|
|
|
2118
2482
|
if (next !== prev) for (const cb of subscribers[name]) try {
|
|
2119
2483
|
cb(next);
|
|
2120
2484
|
} catch (err) {
|
|
2121
|
-
log$
|
|
2485
|
+
log$9.warn({
|
|
2122
2486
|
err,
|
|
2123
2487
|
flag: name
|
|
2124
2488
|
}, "flag subscriber threw");
|
|
@@ -2130,7 +2494,7 @@ async function pollOnce() {
|
|
|
2130
2494
|
if (!flags) return;
|
|
2131
2495
|
apply("contextManagement", flags.contextManagement ?? null);
|
|
2132
2496
|
} catch (err) {
|
|
2133
|
-
log$
|
|
2497
|
+
log$9.debug({ err }, "feature-flag poll threw");
|
|
2134
2498
|
}
|
|
2135
2499
|
}
|
|
2136
2500
|
/**
|
|
@@ -2237,7 +2601,7 @@ function transformContextMessages(messages, config, now) {
|
|
|
2237
2601
|
}
|
|
2238
2602
|
//#endregion
|
|
2239
2603
|
//#region src/extensions/context-management.ts
|
|
2240
|
-
const log$
|
|
2604
|
+
const log$8 = logger.child({ module: "context-management-extension" });
|
|
2241
2605
|
function isAnthropicMessagesPayload(payload) {
|
|
2242
2606
|
if (typeof payload !== "object" || payload === null) return false;
|
|
2243
2607
|
const candidate = payload;
|
|
@@ -2302,13 +2666,13 @@ function createContextManagementExtension() {
|
|
|
2302
2666
|
setContextManagementFlagOverride(getPolledFlag("contextManagement"));
|
|
2303
2667
|
onFlagChange("contextManagement", (enabled) => {
|
|
2304
2668
|
setContextManagementFlagOverride(enabled);
|
|
2305
|
-
log$
|
|
2669
|
+
log$8.info({
|
|
2306
2670
|
event: "context_management_flag_update",
|
|
2307
2671
|
enabled
|
|
2308
2672
|
}, "context-management flag updated from platform");
|
|
2309
2673
|
});
|
|
2310
2674
|
startFeatureFlagPoller();
|
|
2311
|
-
log$
|
|
2675
|
+
log$8.info({
|
|
2312
2676
|
event: "context_management_registered",
|
|
2313
2677
|
enabled: initial.enabled,
|
|
2314
2678
|
flagSource: hasFlagSource(),
|
|
@@ -2320,13 +2684,13 @@ function createContextManagementExtension() {
|
|
|
2320
2684
|
const { messages } = event;
|
|
2321
2685
|
try {
|
|
2322
2686
|
const result = transformContextIfEnabled(messages, getContextManagementConfig(), Date.now());
|
|
2323
|
-
if (result.stats && (result.stats.clearedResults > 0 || result.stats.trimmedResults > 0)) log$
|
|
2687
|
+
if (result.stats && (result.stats.clearedResults > 0 || result.stats.trimmedResults > 0)) log$8.info({
|
|
2324
2688
|
event: "context_management_applied",
|
|
2325
2689
|
...result.stats
|
|
2326
2690
|
}, "trimmed/cleared tool output before LLM call");
|
|
2327
2691
|
return { messages: result.messages };
|
|
2328
2692
|
} catch (err) {
|
|
2329
|
-
log$
|
|
2693
|
+
log$8.error({
|
|
2330
2694
|
err,
|
|
2331
2695
|
event: "context_management_transform_failed"
|
|
2332
2696
|
}, "context transform failed; passing messages through unchanged");
|
|
@@ -2338,7 +2702,7 @@ function createContextManagementExtension() {
|
|
|
2338
2702
|
}
|
|
2339
2703
|
//#endregion
|
|
2340
2704
|
//#region src/extensions/current-time.ts
|
|
2341
|
-
const log$
|
|
2705
|
+
const log$7 = logger.child({ module: "current-time-extension" });
|
|
2342
2706
|
const PI_DATE_LINE = /^Current date:.*$/m;
|
|
2343
2707
|
function formatCurrentTimeLine(now) {
|
|
2344
2708
|
return `Current date: ${now.getUTCFullYear()}-${String(now.getUTCMonth() + 1).padStart(2, "0")}-${String(now.getUTCDate()).padStart(2, "0")} (${new Intl.DateTimeFormat("en-US", {
|
|
@@ -2351,7 +2715,7 @@ const currentTimeExtension = (pi) => {
|
|
|
2351
2715
|
const line = formatCurrentTimeLine(/* @__PURE__ */ new Date());
|
|
2352
2716
|
const base = event.systemPrompt;
|
|
2353
2717
|
if (PI_DATE_LINE.test(base)) {
|
|
2354
|
-
log$
|
|
2718
|
+
log$7.info({ event: "pi_date_line_present" }, "pi base prompt carries its own 'Current date:' line again; replacing it in place (pi prompt format may have changed)");
|
|
2355
2719
|
return { systemPrompt: base.replace(PI_DATE_LINE, line) };
|
|
2356
2720
|
}
|
|
2357
2721
|
return { systemPrompt: `${base}\n${line}` };
|
|
@@ -2588,7 +2952,7 @@ function renderIndex(entries) {
|
|
|
2588
2952
|
}
|
|
2589
2953
|
//#endregion
|
|
2590
2954
|
//#region src/extensions/memory.ts
|
|
2591
|
-
const log$
|
|
2955
|
+
const log$6 = logger.child({ module: "memory-extension" });
|
|
2592
2956
|
/**
|
|
2593
2957
|
* The standing instructions for the memory system. Always injected (even with
|
|
2594
2958
|
* an empty `.memory/`) so the agent knows it can persist notes. `users/` is
|
|
@@ -2622,7 +2986,7 @@ const memoryExtension = (pi) => {
|
|
|
2622
2986
|
index
|
|
2623
2987
|
});
|
|
2624
2988
|
} catch (err) {
|
|
2625
|
-
log$
|
|
2989
|
+
log$6.warn({
|
|
2626
2990
|
err,
|
|
2627
2991
|
event: "memory_index_failed"
|
|
2628
2992
|
}, "memory index build failed; injecting instructions only");
|
|
@@ -2636,7 +3000,7 @@ const memoryExtension = (pi) => {
|
|
|
2636
3000
|
};
|
|
2637
3001
|
//#endregion
|
|
2638
3002
|
//#region src/extensions/platform-memory.ts
|
|
2639
|
-
const log$
|
|
3003
|
+
const log$5 = logger.child({ module: "platform-memory-extension" });
|
|
2640
3004
|
/**
|
|
2641
3005
|
* Resolve the human on this turn via the API, keyed by the message id.
|
|
2642
3006
|
* `/sandbox/channel-context` only returns a sender for a platform-known
|
|
@@ -2649,13 +3013,13 @@ const log$6 = logger.child({ module: "platform-memory-extension" });
|
|
|
2649
3013
|
async function resolveTurnUser(messageId) {
|
|
2650
3014
|
const client = sandboxClient();
|
|
2651
3015
|
if (!client) {
|
|
2652
|
-
log$
|
|
3016
|
+
log$5.debug({ event: "resolve_turn_user_no_api_url" }, "no API url in env; withholding user memory");
|
|
2653
3017
|
return null;
|
|
2654
3018
|
}
|
|
2655
3019
|
try {
|
|
2656
3020
|
const res = await client["channel-context"].$get({ query: { messageId } });
|
|
2657
3021
|
if (!res.ok) {
|
|
2658
|
-
log$
|
|
3022
|
+
log$5.warn({
|
|
2659
3023
|
event: "resolve_turn_user_failed",
|
|
2660
3024
|
status: res.status
|
|
2661
3025
|
}, "channel-context returned non-ok; withholding user memory");
|
|
@@ -2668,7 +3032,7 @@ async function resolveTurnUser(messageId) {
|
|
|
2668
3032
|
displayName: sender.displayName
|
|
2669
3033
|
};
|
|
2670
3034
|
} catch (err) {
|
|
2671
|
-
log$
|
|
3035
|
+
log$5.warn({
|
|
2672
3036
|
err,
|
|
2673
3037
|
event: "resolve_turn_user_failed"
|
|
2674
3038
|
}, "failed to resolve current user; withholding user memory");
|
|
@@ -2715,7 +3079,7 @@ function createPlatformMemoryExtension({ channelContext }) {
|
|
|
2715
3079
|
user
|
|
2716
3080
|
});
|
|
2717
3081
|
} catch (err) {
|
|
2718
|
-
log$
|
|
3082
|
+
log$5.warn({
|
|
2719
3083
|
err,
|
|
2720
3084
|
event: "user_memory_index_failed"
|
|
2721
3085
|
}, "user memory index build failed; skipping injection");
|
|
@@ -2730,7 +3094,7 @@ function createPlatformMemoryExtension({ channelContext }) {
|
|
|
2730
3094
|
}
|
|
2731
3095
|
//#endregion
|
|
2732
3096
|
//#region src/extensions/self-trace.ts
|
|
2733
|
-
const log$
|
|
3097
|
+
const log$4 = logger.child({ module: "self-trace-extension" });
|
|
2734
3098
|
/**
|
|
2735
3099
|
* Reports the agent's own execution as OpenTelemetry spans:
|
|
2736
3100
|
* agent.session → agent.run → agent.turn.N → tool.NAME, with token/cost
|
|
@@ -2755,7 +3119,7 @@ const selfTraceExtension = (pi) => {
|
|
|
2755
3119
|
sessionSpan = tracer.startSpan("agent.session", { attributes: { "agent.model": modelId } }, remoteCtx);
|
|
2756
3120
|
sessionCtx = trace.setSpan(remoteCtx, sessionSpan);
|
|
2757
3121
|
const sc = sessionSpan.spanContext();
|
|
2758
|
-
log$
|
|
3122
|
+
log$4.info({
|
|
2759
3123
|
event: "self_trace_session_start",
|
|
2760
3124
|
trace_id: sc.traceId,
|
|
2761
3125
|
span_id: sc.spanId,
|
|
@@ -2867,13 +3231,13 @@ const selfTraceExtension = (pi) => {
|
|
|
2867
3231
|
* Lives in the harness package — soul.md is content from the agent's
|
|
2868
3232
|
* own git repo, not from the platform — so its handling stays here.
|
|
2869
3233
|
*/
|
|
2870
|
-
const log$
|
|
3234
|
+
const log$3 = logger.child({ module: "soul-extension" });
|
|
2871
3235
|
async function readSoul(cwd) {
|
|
2872
3236
|
try {
|
|
2873
3237
|
return (await readFile(join(cwd, "soul.md"), "utf8")).trim() || null;
|
|
2874
3238
|
} catch (err) {
|
|
2875
3239
|
if (err?.code === "ENOENT") return null;
|
|
2876
|
-
log$
|
|
3240
|
+
log$3.warn({
|
|
2877
3241
|
err,
|
|
2878
3242
|
event: "soul_read_failed"
|
|
2879
3243
|
}, "soul.md read failed");
|
|
@@ -2905,7 +3269,7 @@ const soulExtension = (pi) => {
|
|
|
2905
3269
|
};
|
|
2906
3270
|
//#endregion
|
|
2907
3271
|
//#region src/extensions/subagent/index.ts
|
|
2908
|
-
const log$
|
|
3272
|
+
const log$2 = logger.child({ module: "subagent-ext" });
|
|
2909
3273
|
const MAX_TASKS = 8;
|
|
2910
3274
|
const TaskItem = Type.Object({
|
|
2911
3275
|
task: Type.String({ description: "The task to delegate to a subagent run." }),
|
|
@@ -2965,7 +3329,7 @@ function buildTool(messageId) {
|
|
|
2965
3329
|
tasks: spawnTasks
|
|
2966
3330
|
});
|
|
2967
3331
|
const { taskIds } = spawned;
|
|
2968
|
-
log$
|
|
3332
|
+
log$2.info({
|
|
2969
3333
|
event: "subagent_spawned",
|
|
2970
3334
|
count: taskIds.length
|
|
2971
3335
|
}, "subagent tasks queued");
|
|
@@ -2988,7 +3352,7 @@ function buildTool(messageId) {
|
|
|
2988
3352
|
};
|
|
2989
3353
|
} catch (err) {
|
|
2990
3354
|
const message = err instanceof Error ? err.message : String(err);
|
|
2991
|
-
log$
|
|
3355
|
+
log$2.warn({
|
|
2992
3356
|
err,
|
|
2993
3357
|
event: "subagent_spawn_failed"
|
|
2994
3358
|
}, "subagent spawn failed");
|
|
@@ -3019,7 +3383,7 @@ function createSubagentExtension({ channelContext }) {
|
|
|
3019
3383
|
if (registered) return;
|
|
3020
3384
|
registered = true;
|
|
3021
3385
|
pi.registerTool(buildTool(messageId));
|
|
3022
|
-
log$
|
|
3386
|
+
log$2.info({ event: "subagent_enabled" }, "subagent tool registered");
|
|
3023
3387
|
};
|
|
3024
3388
|
pi.on("session_start", () => {
|
|
3025
3389
|
registerOnce();
|
|
@@ -3027,218 +3391,6 @@ function createSubagentExtension({ channelContext }) {
|
|
|
3027
3391
|
};
|
|
3028
3392
|
}
|
|
3029
3393
|
//#endregion
|
|
3030
|
-
//#region src/extensions/resource-pressure-warning.ts
|
|
3031
|
-
/**
|
|
3032
|
-
* Mid-run resource-pressure warning to the agent.
|
|
3033
|
-
*
|
|
3034
|
-
* The sandbox already detects pressure — the boot scripts cap the
|
|
3035
|
-
* user-workload cgroup (memory.high/memory.max) and watchers log warn/crit
|
|
3036
|
-
* edges for memory and disk — but nothing told the *agent*, so a turn burned
|
|
3037
|
-
* straight to the OOM kill (or a full disk) and only learned about it from
|
|
3038
|
-
* the post-mortem notice. This extension closes that gap in-process: while a
|
|
3039
|
-
* turn is active it polls the agent cgroup and the root filesystem and, the
|
|
3040
|
-
* first time usage crosses a warn threshold, folds a system notification into
|
|
3041
|
-
* the open turn so the agent can checkpoint, shed work (constrain
|
|
3042
|
-
* parallelism, kill a background hog, clean scratch space), or request a
|
|
3043
|
-
* bigger tier BEFORE the kill.
|
|
3044
|
-
*
|
|
3045
|
-
* The notification is triggered by the two conditions that actually kill
|
|
3046
|
-
* work — memory near the cgroup hard cap, disk near full — and reports a
|
|
3047
|
-
* snapshot of all the relevant stats (memory, CPU utilization, disk) so the
|
|
3048
|
-
* agent can tell which resource is the problem and how much headroom the
|
|
3049
|
-
* others have.
|
|
3050
|
-
*
|
|
3051
|
-
* Edge-triggered, once per trigger per turn: the fired flags reset on
|
|
3052
|
-
* agent_start, so a turn that rides a threshold gets one warning per
|
|
3053
|
-
* resource, not a stream. Polling only runs while the agent is active — an
|
|
3054
|
-
* idle sandbox's resource usage is not the agent's problem and there is no
|
|
3055
|
-
* open turn to deliver into anyway.
|
|
3056
|
-
*
|
|
3057
|
-
* Best-effort throughout: any read failure (cgroup absent, controller not
|
|
3058
|
-
* delegated, non-cgroup-v2 host, df missing) reads as "no signal" for that
|
|
3059
|
-
* stat and the extension warns on what it can see — it must never break a
|
|
3060
|
-
* turn over an observability feature.
|
|
3061
|
-
*/
|
|
3062
|
-
const execFileAsync = promisify(execFile);
|
|
3063
|
-
const log$2 = logger.child({ module: "resource-pressure-warning" });
|
|
3064
|
-
const POLL_INTERVAL_MS = 1e4;
|
|
3065
|
-
function envOverride(name) {
|
|
3066
|
-
for (const prefix of ["SKYDIVE_", "ANYONE_"]) {
|
|
3067
|
-
const value = process.env[`${prefix}${name}`];
|
|
3068
|
-
if (value != null && value !== "") return value;
|
|
3069
|
-
}
|
|
3070
|
-
return null;
|
|
3071
|
-
}
|
|
3072
|
-
function cgroupDir() {
|
|
3073
|
-
return envOverride("AGENT_CGROUP") ?? "/sys/fs/cgroup/agent";
|
|
3074
|
-
}
|
|
3075
|
-
function diskRoot() {
|
|
3076
|
-
return envOverride("DISK_ROOT") ?? "/";
|
|
3077
|
-
}
|
|
3078
|
-
/**
|
|
3079
|
-
* Read a cgroup v2 scalar file. Returns a number, or null for "max"
|
|
3080
|
-
* (uncapped), an empty/absent file, or any read/parse error — an uncapped or
|
|
3081
|
-
* unreadable limit means there is nothing meaningful to warn against.
|
|
3082
|
-
*/
|
|
3083
|
-
async function readScalar(file) {
|
|
3084
|
-
try {
|
|
3085
|
-
const raw = (await readFile(`${cgroupDir()}/${file}`, "utf8")).trim();
|
|
3086
|
-
if (raw === "" || raw === "max") return null;
|
|
3087
|
-
const n = Number(raw);
|
|
3088
|
-
return Number.isFinite(n) ? n : null;
|
|
3089
|
-
} catch (_error) {
|
|
3090
|
-
return null;
|
|
3091
|
-
}
|
|
3092
|
-
}
|
|
3093
|
-
/**
|
|
3094
|
-
* Read a cgroup v2 "flat keyed" file (one `key value` pair per line, e.g.
|
|
3095
|
-
* cpu.stat) and return the counter for `key`, or null when absent.
|
|
3096
|
-
*/
|
|
3097
|
-
async function readKeyedCounter(file, key) {
|
|
3098
|
-
try {
|
|
3099
|
-
const raw = await readFile(`${cgroupDir()}/${file}`, "utf8");
|
|
3100
|
-
for (const line of raw.split("\n")) {
|
|
3101
|
-
const [k, v] = line.trim().split(/\s+/);
|
|
3102
|
-
if (k === key) {
|
|
3103
|
-
const n = Number(v);
|
|
3104
|
-
return Number.isFinite(n) ? n : null;
|
|
3105
|
-
}
|
|
3106
|
-
}
|
|
3107
|
-
return null;
|
|
3108
|
-
} catch (_error) {
|
|
3109
|
-
return null;
|
|
3110
|
-
}
|
|
3111
|
-
}
|
|
3112
|
-
/**
|
|
3113
|
-
* Live memory usage as an integer percent of the hard cap, or null when
|
|
3114
|
-
* either side is unreadable/uncapped. Exported for tests.
|
|
3115
|
-
*/
|
|
3116
|
-
async function readMemUsePct() {
|
|
3117
|
-
const [current, max] = await Promise.all([readScalar("memory.current"), readScalar("memory.max")]);
|
|
3118
|
-
if (current === null || max === null || max <= 0) return null;
|
|
3119
|
-
return {
|
|
3120
|
-
pct: Math.floor(current / max * 100),
|
|
3121
|
-
currentBytes: current,
|
|
3122
|
-
maxBytes: max
|
|
3123
|
-
};
|
|
3124
|
-
}
|
|
3125
|
-
/**
|
|
3126
|
-
* Root filesystem used% (df -P Capacity column), or null on any failure.
|
|
3127
|
-
* Exported for tests.
|
|
3128
|
-
*/
|
|
3129
|
-
async function readDiskUsePct() {
|
|
3130
|
-
try {
|
|
3131
|
-
const { stdout } = await execFileAsync("df", ["-P", diskRoot()]);
|
|
3132
|
-
const dataRow = stdout.trim().split("\n")[1];
|
|
3133
|
-
if (dataRow == null) return null;
|
|
3134
|
-
const capacity = dataRow.trim().split(/\s+/)[4];
|
|
3135
|
-
if (capacity == null) return null;
|
|
3136
|
-
const pct = Number(capacity.replace("%", ""));
|
|
3137
|
-
return Number.isFinite(pct) ? pct : null;
|
|
3138
|
-
} catch (_error) {
|
|
3139
|
-
return null;
|
|
3140
|
-
}
|
|
3141
|
-
}
|
|
3142
|
-
/**
|
|
3143
|
-
* CPU utilization sampler. cgroup v2 exposes cumulative CPU time
|
|
3144
|
-
* (cpu.stat usage_usec); utilization is the delta between two samples over
|
|
3145
|
-
* the wall time between them, normalized by core count. The first call after
|
|
3146
|
-
* construction has no previous sample and returns null.
|
|
3147
|
-
*/
|
|
3148
|
-
function createCpuSampler() {
|
|
3149
|
-
let prevUsageUsec = null;
|
|
3150
|
-
let prevAtMs = null;
|
|
3151
|
-
return async () => {
|
|
3152
|
-
const usage = await readKeyedCounter("cpu.stat", "usage_usec");
|
|
3153
|
-
const now = Date.now();
|
|
3154
|
-
const prev = prevUsageUsec;
|
|
3155
|
-
const prevAt = prevAtMs;
|
|
3156
|
-
prevUsageUsec = usage;
|
|
3157
|
-
prevAtMs = now;
|
|
3158
|
-
if (usage === null || prev === null || prevAt === null) return null;
|
|
3159
|
-
const wallUsec = (now - prevAt) * 1e3;
|
|
3160
|
-
if (wallUsec <= 0) return null;
|
|
3161
|
-
const cores = availableParallelism();
|
|
3162
|
-
const pct = Math.round((usage - prev) / (wallUsec * cores) * 100);
|
|
3163
|
-
return Math.max(0, Math.min(100, pct));
|
|
3164
|
-
};
|
|
3165
|
-
}
|
|
3166
|
-
function fmtMb(bytes) {
|
|
3167
|
-
return Math.round(bytes / 1024 / 1024);
|
|
3168
|
-
}
|
|
3169
|
-
/** The model-facing warning text. Exported for tests. */
|
|
3170
|
-
function resourcePressureWarningText(trigger, { mem, cpuPct, diskPct }) {
|
|
3171
|
-
const stats = [];
|
|
3172
|
-
if (mem) stats.push(`memory ${mem.pct}% of cap (${fmtMb(mem.currentBytes)}/${fmtMb(mem.maxBytes)} MB)`);
|
|
3173
|
-
if (cpuPct !== null) stats.push(`CPU ${cpuPct}%`);
|
|
3174
|
-
if (diskPct !== null) stats.push(`disk ${diskPct}% full`);
|
|
3175
|
-
const lead = trigger === "memory" ? `Your sandbox is at ${mem?.pct}% of its memory cap. If usage keeps climbing, the kernel will kill the offending process and this turn may die with it.` : `Your sandbox's disk is ${diskPct}% full. If it fills completely, writes will start failing and this turn may die with them.`;
|
|
3176
|
-
const remedy = trigger === "memory" ? "checkpoint in-flight work (commit and push), then reduce the footprint — constrain parallelism, run heavy steps sequentially, or kill background processes you no longer need." : "checkpoint in-flight work (commit and push), then free space — clean build artifacts, caches, and scratch files you no longer need.";
|
|
3177
|
-
return `<system_notification>${lead} Current usage: ${stats.join(", ")}. Act now: ${remedy} If the workload genuinely needs more resources, request a bigger sandbox with \`platform compute request\`. This is an automated resource warning, not a message from the user; continue the task, adjusted.</system_notification>`;
|
|
3178
|
-
}
|
|
3179
|
-
const resourcePressureWarningExtension = (pi) => {
|
|
3180
|
-
let agentActive = false;
|
|
3181
|
-
let warnedMemThisTurn = false;
|
|
3182
|
-
let warnedDiskThisTurn = false;
|
|
3183
|
-
let timer = null;
|
|
3184
|
-
const sampleCpu = createCpuSampler();
|
|
3185
|
-
async function checkOnce() {
|
|
3186
|
-
if (!agentActive || warnedMemThisTurn && warnedDiskThisTurn) return;
|
|
3187
|
-
const [mem, cpuPct, diskPct] = await Promise.all([
|
|
3188
|
-
readMemUsePct(),
|
|
3189
|
-
sampleCpu(),
|
|
3190
|
-
readDiskUsePct()
|
|
3191
|
-
]);
|
|
3192
|
-
let trigger = null;
|
|
3193
|
-
if (!warnedMemThisTurn && mem !== null && mem.pct >= 80) {
|
|
3194
|
-
trigger = "memory";
|
|
3195
|
-
warnedMemThisTurn = true;
|
|
3196
|
-
} else if (!warnedDiskThisTurn && diskPct !== null && diskPct >= 80) {
|
|
3197
|
-
trigger = "disk";
|
|
3198
|
-
warnedDiskThisTurn = true;
|
|
3199
|
-
}
|
|
3200
|
-
if (trigger === null) return;
|
|
3201
|
-
log$2.warn({
|
|
3202
|
-
trigger,
|
|
3203
|
-
mem,
|
|
3204
|
-
cpuPct,
|
|
3205
|
-
diskPct
|
|
3206
|
-
}, "resource pressure warning delivered to agent");
|
|
3207
|
-
await pi.sendMessage({
|
|
3208
|
-
customType: "anyone-resource-pressure-warning",
|
|
3209
|
-
content: resourcePressureWarningText(trigger, {
|
|
3210
|
-
mem,
|
|
3211
|
-
cpuPct,
|
|
3212
|
-
diskPct
|
|
3213
|
-
}),
|
|
3214
|
-
display: false
|
|
3215
|
-
}, {
|
|
3216
|
-
triggerTurn: true,
|
|
3217
|
-
deliverAs: "followUp"
|
|
3218
|
-
});
|
|
3219
|
-
}
|
|
3220
|
-
pi.on("agent_start", async () => {
|
|
3221
|
-
agentActive = true;
|
|
3222
|
-
warnedMemThisTurn = false;
|
|
3223
|
-
warnedDiskThisTurn = false;
|
|
3224
|
-
if (!timer) {
|
|
3225
|
-
timer = setInterval(() => {
|
|
3226
|
-
checkOnce().catch((err) => {
|
|
3227
|
-
log$2.error({ err }, "resource pressure check failed");
|
|
3228
|
-
});
|
|
3229
|
-
}, POLL_INTERVAL_MS);
|
|
3230
|
-
timer.unref?.();
|
|
3231
|
-
}
|
|
3232
|
-
});
|
|
3233
|
-
pi.on("agent_end", async () => {
|
|
3234
|
-
agentActive = false;
|
|
3235
|
-
if (timer) {
|
|
3236
|
-
clearInterval(timer);
|
|
3237
|
-
timer = null;
|
|
3238
|
-
}
|
|
3239
|
-
});
|
|
3240
|
-
};
|
|
3241
|
-
//#endregion
|
|
3242
3394
|
//#region src/extensions/tool-call-env.ts
|
|
3243
3395
|
const TOOL_CALL_ID_VAR = "TOOL_CALL_ID";
|
|
3244
3396
|
function shellQuoteValue(value) {
|
|
@@ -3994,6 +4146,7 @@ const all = [
|
|
|
3994
4146
|
localToolsExtension,
|
|
3995
4147
|
toolCallEnvExtension,
|
|
3996
4148
|
bashDefaultTimeoutExtension,
|
|
4149
|
+
diskGuardExtension,
|
|
3997
4150
|
toolCallSummaryExtension
|
|
3998
4151
|
];
|
|
3999
4152
|
/**
|