@skydiveai/pi-extensions 0.1.0-beta.121 → 0.1.0-beta.1210
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.mjs +464 -114
- package/package.json +1 -7
package/dist/index.mjs
CHANGED
|
@@ -21,6 +21,9 @@ import { BatchSpanProcessor, NodeTracerProvider } from "@opentelemetry/sdk-trace
|
|
|
21
21
|
import { ATTR_SERVICE_NAME } from "@opentelemetry/semantic-conventions";
|
|
22
22
|
import { hc } from "hono/client";
|
|
23
23
|
import { parse } from "yaml";
|
|
24
|
+
import { execFile } from "node:child_process";
|
|
25
|
+
import { availableParallelism } from "node:os";
|
|
26
|
+
import { promisify } from "node:util";
|
|
24
27
|
import { quote } from "shell-quote";
|
|
25
28
|
import { createWriteStream } from "node:fs";
|
|
26
29
|
import { finished } from "node:stream/promises";
|
|
@@ -256,7 +259,7 @@ function createHealthHandler({ metadata }) {
|
|
|
256
259
|
* read on the hot path before every LLM call), it falls back to the default
|
|
257
260
|
* for that knob and logs once.
|
|
258
261
|
*/
|
|
259
|
-
const log$
|
|
262
|
+
const log$14 = logger.child({ module: "context-management-config" });
|
|
260
263
|
const DEFAULT_CONTEXT_MANAGEMENT_CONFIG = {
|
|
261
264
|
enabled: false,
|
|
262
265
|
perResultMaxBytes: 16 * 1024,
|
|
@@ -304,7 +307,7 @@ function resolveContextManagementConfig(env = process.env) {
|
|
|
304
307
|
maxModelCallsPerTurn: env.SKYDIVE_CTX_MAX_MODEL_CALLS
|
|
305
308
|
});
|
|
306
309
|
if (!parsed.success) {
|
|
307
|
-
log$
|
|
310
|
+
log$14.warn({
|
|
308
311
|
event: "context_management_config_invalid",
|
|
309
312
|
err: parsed.error
|
|
310
313
|
}, "falling back to default context-management config");
|
|
@@ -438,7 +441,7 @@ const CAPABILITY_SOUL_NUDGE = "New capability gained — once the current task i
|
|
|
438
441
|
* or `ToolDefinition[]`. Files starting with `_` or `.` are skipped, so
|
|
439
442
|
* `tools/_example.ts` documents the shape without registering.
|
|
440
443
|
*/
|
|
441
|
-
const log$
|
|
444
|
+
const log$13 = logger.child({ module: "local-tools-extension" });
|
|
442
445
|
const TOOLS_DIRNAME = "tools";
|
|
443
446
|
const fileState = /* @__PURE__ */ new Map();
|
|
444
447
|
let pendingLocalToolsUpdate = null;
|
|
@@ -582,7 +585,7 @@ async function reconcileAndQueue({ pi, dir, reason }) {
|
|
|
582
585
|
dir
|
|
583
586
|
});
|
|
584
587
|
if (reason !== "session_start" && summaryHasChanges$1(summary)) pendingLocalToolsUpdate = summary;
|
|
585
|
-
log$
|
|
588
|
+
log$13.info({
|
|
586
589
|
event: "local_tools_reconcile",
|
|
587
590
|
reason,
|
|
588
591
|
total_tools: summary.totalTools,
|
|
@@ -604,7 +607,7 @@ const localToolsExtension = (pi) => {
|
|
|
604
607
|
reason: "session_start"
|
|
605
608
|
});
|
|
606
609
|
} catch (err) {
|
|
607
|
-
log$
|
|
610
|
+
log$13.error({
|
|
608
611
|
err,
|
|
609
612
|
event: "local_tools_reconcile_failed"
|
|
610
613
|
}, "local tools reconcile failed");
|
|
@@ -616,7 +619,7 @@ const localToolsExtension = (pi) => {
|
|
|
616
619
|
try {
|
|
617
620
|
current = await listToolFiles(dir);
|
|
618
621
|
} catch (err) {
|
|
619
|
-
log$
|
|
622
|
+
log$13.warn({
|
|
620
623
|
err,
|
|
621
624
|
event: "local_tools_listing_failed"
|
|
622
625
|
}, "tools/ listing failed");
|
|
@@ -638,7 +641,7 @@ const localToolsExtension = (pi) => {
|
|
|
638
641
|
reason: "auto_reload"
|
|
639
642
|
});
|
|
640
643
|
} catch (err) {
|
|
641
|
-
log$
|
|
644
|
+
log$13.error({
|
|
642
645
|
err,
|
|
643
646
|
event: "local_tools_auto_reload_failed"
|
|
644
647
|
}, "auto-reload after tools/ change failed");
|
|
@@ -692,6 +695,19 @@ const STDERR_BUFFER_BYTES = 4096;
|
|
|
692
695
|
* indistinguishable from any other transport problem. Walk the cause
|
|
693
696
|
* chain so the agent sees the real underlying error.
|
|
694
697
|
*/
|
|
698
|
+
/**
|
|
699
|
+
* True when an error from an http MCP transport (connect, listTools, or a tool
|
|
700
|
+
* call) is an authentication failure. With no authProvider configured the SDK
|
|
701
|
+
* surfaces a 401 as `StreamableHTTPError(401)`; older paths translate it to
|
|
702
|
+
* `UnauthorizedError`. A dead/expired OAuth token (the proxy can no longer
|
|
703
|
+
* mint one) shows up here on the NEXT request against a previously-connected
|
|
704
|
+
* client — not just at connect — so reconcile must re-classify such a failure
|
|
705
|
+
* as `pending_auth` instead of a generic `failed`, keeping the "waiting on
|
|
706
|
+
* auth" report consistent with `platform auth`.
|
|
707
|
+
*/
|
|
708
|
+
function isUnauthorizedError(err) {
|
|
709
|
+
return err instanceof UnauthorizedError || err instanceof StreamableHTTPError && err.code === 401;
|
|
710
|
+
}
|
|
695
711
|
function formatError(err) {
|
|
696
712
|
if (!(err instanceof Error)) return String(err);
|
|
697
713
|
const parts = [err.message];
|
|
@@ -713,7 +729,7 @@ async function connectHttp(_id, config, client) {
|
|
|
713
729
|
stderr: null
|
|
714
730
|
};
|
|
715
731
|
} catch (err) {
|
|
716
|
-
if (err
|
|
732
|
+
if (isUnauthorizedError(err)) return {
|
|
717
733
|
status: "pending_auth",
|
|
718
734
|
client,
|
|
719
735
|
stderr: "",
|
|
@@ -863,7 +879,7 @@ async function loadMcpConfig(path) {
|
|
|
863
879
|
* Clients are keyed by JSON-stringified config and reused across
|
|
864
880
|
* reloads — only changed configs reconnect.
|
|
865
881
|
*/
|
|
866
|
-
const log$
|
|
882
|
+
const log$12 = logger.child({ module: "mcp-extension" });
|
|
867
883
|
async function closeConnected(connected) {
|
|
868
884
|
try {
|
|
869
885
|
await connected.client.close();
|
|
@@ -1190,6 +1206,64 @@ var McpExtension = class {
|
|
|
1190
1206
|
try {
|
|
1191
1207
|
mcpTools = (await connected.client.listTools()).tools;
|
|
1192
1208
|
} catch (err) {
|
|
1209
|
+
if (isUnauthorizedError(err) && serverConfig.transport === "http") {
|
|
1210
|
+
await closeConnected(connected);
|
|
1211
|
+
const retry = await connectClient(id, serverConfig, { connectTimeoutMs });
|
|
1212
|
+
if (retry.status === "pending_auth") return {
|
|
1213
|
+
id,
|
|
1214
|
+
store: {
|
|
1215
|
+
client: retry.client,
|
|
1216
|
+
configKey,
|
|
1217
|
+
status: "pending_auth",
|
|
1218
|
+
stderrBuffer: null,
|
|
1219
|
+
cliHint: retry.cliHint
|
|
1220
|
+
},
|
|
1221
|
+
serverStatus: {
|
|
1222
|
+
status: "pending_auth",
|
|
1223
|
+
stderr: "",
|
|
1224
|
+
cliHint: retry.cliHint
|
|
1225
|
+
},
|
|
1226
|
+
change: null,
|
|
1227
|
+
error: null,
|
|
1228
|
+
tools: null
|
|
1229
|
+
};
|
|
1230
|
+
if (retry.status === "failed") return {
|
|
1231
|
+
id,
|
|
1232
|
+
store: null,
|
|
1233
|
+
serverStatus: {
|
|
1234
|
+
status: "failed",
|
|
1235
|
+
error: retry.error,
|
|
1236
|
+
stderr: retry.stderr
|
|
1237
|
+
},
|
|
1238
|
+
change: null,
|
|
1239
|
+
error: {
|
|
1240
|
+
serverId: id,
|
|
1241
|
+
message: retry.error
|
|
1242
|
+
},
|
|
1243
|
+
tools: null
|
|
1244
|
+
};
|
|
1245
|
+
if (retry.status === "connected") {
|
|
1246
|
+
connected = {
|
|
1247
|
+
client: retry.client,
|
|
1248
|
+
configKey,
|
|
1249
|
+
status: "connected",
|
|
1250
|
+
stderrBuffer: retry.stderr,
|
|
1251
|
+
cliHint: null
|
|
1252
|
+
};
|
|
1253
|
+
mcpTools = (await connected.client.listTools()).tools;
|
|
1254
|
+
return {
|
|
1255
|
+
id,
|
|
1256
|
+
store: connected,
|
|
1257
|
+
serverStatus: { status: "connected" },
|
|
1258
|
+
change: action === "reused" ? "refreshed" : action,
|
|
1259
|
+
error: null,
|
|
1260
|
+
tools: {
|
|
1261
|
+
client: connected.client,
|
|
1262
|
+
list: mcpTools
|
|
1263
|
+
}
|
|
1264
|
+
};
|
|
1265
|
+
}
|
|
1266
|
+
}
|
|
1193
1267
|
const message = err instanceof Error ? err.message : String(err);
|
|
1194
1268
|
const stderr = connected.stderrBuffer?.read() ?? "";
|
|
1195
1269
|
return {
|
|
@@ -1227,7 +1301,7 @@ var McpExtension = class {
|
|
|
1227
1301
|
});
|
|
1228
1302
|
this.lastConfigMtimeMs = await readConfigMtimeMs(configPath);
|
|
1229
1303
|
if (reason !== "session_start" && summaryHasChanges(summary)) this.pendingMcpUpdate = summary;
|
|
1230
|
-
log$
|
|
1304
|
+
log$12.info({
|
|
1231
1305
|
event: "mcp_reconcile",
|
|
1232
1306
|
reason,
|
|
1233
1307
|
total_tools: summary.totalTools,
|
|
@@ -1250,7 +1324,7 @@ var McpExtension = class {
|
|
|
1250
1324
|
reason: "session_start"
|
|
1251
1325
|
});
|
|
1252
1326
|
} catch (err) {
|
|
1253
|
-
log$
|
|
1327
|
+
log$12.error({
|
|
1254
1328
|
err,
|
|
1255
1329
|
event: "mcp_reconcile_failed"
|
|
1256
1330
|
}, "MCP reconcile failed");
|
|
@@ -1262,7 +1336,7 @@ var McpExtension = class {
|
|
|
1262
1336
|
try {
|
|
1263
1337
|
mtime = await readConfigMtimeMs(configPath);
|
|
1264
1338
|
} catch (err) {
|
|
1265
|
-
log$
|
|
1339
|
+
log$12.warn({
|
|
1266
1340
|
err,
|
|
1267
1341
|
event: "mcp_mtime_check_failed"
|
|
1268
1342
|
}, "mtime check on mcp.config.json failed");
|
|
@@ -1276,7 +1350,7 @@ var McpExtension = class {
|
|
|
1276
1350
|
reason: "auto_reload"
|
|
1277
1351
|
});
|
|
1278
1352
|
} catch (err) {
|
|
1279
|
-
log$
|
|
1353
|
+
log$12.error({
|
|
1280
1354
|
err,
|
|
1281
1355
|
event: "mcp_auto_reload_failed"
|
|
1282
1356
|
}, "auto-reload after mcp.config.json change failed");
|
|
@@ -1641,14 +1715,14 @@ const bashDefaultTimeoutExtension = (pi) => {
|
|
|
1641
1715
|
//#endregion
|
|
1642
1716
|
//#region src/channel-context-ref.ts
|
|
1643
1717
|
/**
|
|
1644
|
-
* The worker injects
|
|
1645
|
-
* sandbox env (`SKYDIVE_CHANNEL_CONTEXT`) rather than the full context.
|
|
1718
|
+
* The worker injects a small reference — `{ channel, messageId, runId }` —
|
|
1719
|
+
* into the sandbox env (`SKYDIVE_CHANNEL_CONTEXT`) rather than the full context.
|
|
1646
1720
|
*
|
|
1647
1721
|
* The canonical `ChannelContextRef` type + `parseChannelContextRef` live in
|
|
1648
1722
|
* `@createinc/anyone-channels`, but the harness (`@skydiveai/*`) keeps zero
|
|
1649
1723
|
* `@createinc/*` dependencies — importing that package would pull the whole
|
|
1650
|
-
* platform channel stack (Slack/email/Linq SDKs, messaging)
|
|
1651
|
-
*
|
|
1724
|
+
* platform channel stack (Slack/email/Linq SDKs, messaging) just to read one
|
|
1725
|
+
* field. So we validate the field this consumer needs locally instead.
|
|
1652
1726
|
*/
|
|
1653
1727
|
const channelContextRefSchema = z.object({ messageId: z.string().nullable() });
|
|
1654
1728
|
/**
|
|
@@ -1680,20 +1754,84 @@ function apiBaseUrl() {
|
|
|
1680
1754
|
}
|
|
1681
1755
|
//#endregion
|
|
1682
1756
|
//#region src/extensions/platform.ts
|
|
1757
|
+
/**
|
|
1758
|
+
* Platform extension — bridges the agent harness to the Skydive platform daemon.
|
|
1759
|
+
*
|
|
1760
|
+
* Responsibilities:
|
|
1761
|
+
* - Heartbeat: periodic POST to the API so the sandbox manager knows the
|
|
1762
|
+
* agent is alive. Throttled to once per minute, triggered by tool events.
|
|
1763
|
+
* - Session tracking: registers the session with the daemon on start,
|
|
1764
|
+
* streams tool_call / tool_result events so the daemon can track which
|
|
1765
|
+
* session is actively executing, and signals session end on agent_end.
|
|
1766
|
+
* - Channel context: passes the SKYDIVE_CHANNEL_CONTEXT (containing the
|
|
1767
|
+
* messageId) to the daemon so file writes can be attributed to the
|
|
1768
|
+
* correct conversation.
|
|
1769
|
+
*
|
|
1770
|
+
* All daemon POSTs are fire-and-forget — failures are logged but never
|
|
1771
|
+
* block the agent. The daemon may not be running (e.g. local dev without
|
|
1772
|
+
* a sandbox), and that's fine.
|
|
1773
|
+
*/
|
|
1683
1774
|
const HEARTBEAT_THROTTLE_MS = 6e4;
|
|
1684
1775
|
const TOOL_HEARTBEAT_INTERVAL_MS = 5e3;
|
|
1685
1776
|
const MAX_TOOL_HEARTBEATS = 1440 * 60 * 1e3 / TOOL_HEARTBEAT_INTERVAL_MS;
|
|
1686
1777
|
const DAEMON_URL = "http://localhost:38994";
|
|
1687
|
-
const log$
|
|
1778
|
+
const log$11 = logger.child({ module: "platform-ext" });
|
|
1688
1779
|
function sandboxClient() {
|
|
1689
1780
|
const apiUrl = apiBaseUrl();
|
|
1690
1781
|
if (!apiUrl) return null;
|
|
1691
1782
|
return hc(`${apiUrl}/api/v1/sandbox`);
|
|
1692
1783
|
}
|
|
1693
1784
|
/**
|
|
1694
|
-
*
|
|
1695
|
-
*
|
|
1696
|
-
*
|
|
1785
|
+
* Is this box still an unclaimed warm-pool sandbox? (ANY-6000, the
|
|
1786
|
+
* feature-flags half of the ANY-5184 pool 403 wave.)
|
|
1787
|
+
*
|
|
1788
|
+
* `GET /sandbox/feature-flags` is agent-only, so the shared poller's request
|
|
1789
|
+
* from a pool box can only 403 — a guaranteed-failing GET every 60s for the
|
|
1790
|
+
* life of the pool phase. The discriminator is the sandbox token's `type`
|
|
1791
|
+
* claim, read UNVERIFIED (this box never holds the signing secret): not an
|
|
1792
|
+
* authorization decision, only "should I bother calling?", and the api still
|
|
1793
|
+
* authorizes every request.
|
|
1794
|
+
*
|
|
1795
|
+
* Read per call from the daemon's persisted env file, NOT process.env:
|
|
1796
|
+
* claiming a pool box rebinds the token in place (the daemon rewrites this
|
|
1797
|
+
* file) while the harness's process.env keeps the boot snapshot, so a
|
|
1798
|
+
* process-env gate would leave a claimed box permanently skipping — trading a
|
|
1799
|
+
* wasted request for silently frozen flags, which is strictly worse. "Cannot
|
|
1800
|
+
* tell" (no file, no token, unparseable payload) reports false so the poll
|
|
1801
|
+
* proceeds.
|
|
1802
|
+
*/
|
|
1803
|
+
const daemonEnvIdentitySchema = z.object({
|
|
1804
|
+
ANYONE_SANDBOX_TOKEN: z.string().optional(),
|
|
1805
|
+
SKYDIVE_SANDBOX_TOKEN: z.string().optional()
|
|
1806
|
+
}).passthrough();
|
|
1807
|
+
const tokenTypeSchema = z.object({ type: z.string() }).passthrough();
|
|
1808
|
+
async function isPoolIdentity() {
|
|
1809
|
+
try {
|
|
1810
|
+
const override = process.env.ANYONE_DAEMON_ENV_CACHE;
|
|
1811
|
+
const candidates = override ? [override] : ["/run/anyone-system/daemon-env.json", "/tmp/.anyone/daemon-env.json"];
|
|
1812
|
+
let raw = null;
|
|
1813
|
+
for (const file of candidates) {
|
|
1814
|
+
raw = await readFile(file, "utf8").catch(() => null);
|
|
1815
|
+
if (raw !== null) break;
|
|
1816
|
+
}
|
|
1817
|
+
if (raw === null) return false;
|
|
1818
|
+
const env = daemonEnvIdentitySchema.safeParse(JSON.parse(raw));
|
|
1819
|
+
if (!env.success) return false;
|
|
1820
|
+
const token = env.data.ANYONE_SANDBOX_TOKEN ?? env.data.SKYDIVE_SANDBOX_TOKEN;
|
|
1821
|
+
if (typeof token !== "string" || token === "") return false;
|
|
1822
|
+
const payload = token.split(".")[1];
|
|
1823
|
+
if (!payload) return false;
|
|
1824
|
+
const claims = tokenTypeSchema.safeParse(JSON.parse(Buffer.from(payload, "base64url").toString("utf8")));
|
|
1825
|
+
return claims.success && claims.data.type === "onboarding-pool";
|
|
1826
|
+
} catch (_err) {
|
|
1827
|
+
return false;
|
|
1828
|
+
}
|
|
1829
|
+
}
|
|
1830
|
+
/**
|
|
1831
|
+
* Fetch every harness feature flag in one GET (`{ contextManagement, ... }`
|
|
1832
|
+
* — see apps/anyone/api/src/routes/sandbox-feature-flags.ts). Returns
|
|
1833
|
+
* null when indeterminate (no api url, the request failed, or the box is an
|
|
1834
|
+
* unclaimed pool sandbox whose token the route would 403) so the shared
|
|
1697
1835
|
* poller keeps the last-known values rather than flipping on a transient error.
|
|
1698
1836
|
* This is the single fetch behind `feature-flags-poll.ts`; extensions read the
|
|
1699
1837
|
* polled values there instead of issuing their own GET.
|
|
@@ -1701,22 +1839,19 @@ function sandboxClient() {
|
|
|
1701
1839
|
async function fetchHarnessFlags() {
|
|
1702
1840
|
const client = sandboxClient();
|
|
1703
1841
|
if (!client) return null;
|
|
1842
|
+
if (await isPoolIdentity()) return null;
|
|
1704
1843
|
try {
|
|
1705
1844
|
const res = await client["feature-flags"].$get();
|
|
1706
1845
|
if (!res.ok) {
|
|
1707
|
-
log$
|
|
1846
|
+
log$11.debug({
|
|
1708
1847
|
status: res.status,
|
|
1709
1848
|
event: "feature_flags_fetch_failed"
|
|
1710
1849
|
}, "feature-flags fetch failed");
|
|
1711
1850
|
return null;
|
|
1712
1851
|
}
|
|
1713
|
-
|
|
1714
|
-
return {
|
|
1715
|
-
contextManagement: body.contextManagement ?? null,
|
|
1716
|
-
subagent: body.subagent ?? null
|
|
1717
|
-
};
|
|
1852
|
+
return { contextManagement: (await res.json()).contextManagement ?? null };
|
|
1718
1853
|
} catch (err) {
|
|
1719
|
-
log$
|
|
1854
|
+
log$11.debug({
|
|
1720
1855
|
err,
|
|
1721
1856
|
event: "feature_flags_fetch_error"
|
|
1722
1857
|
}, "feature-flags request errored");
|
|
@@ -1727,7 +1862,7 @@ function postHeartbeat({ messageId }) {
|
|
|
1727
1862
|
const client = sandboxClient();
|
|
1728
1863
|
if (!client) return;
|
|
1729
1864
|
client.heartbeat.$post({ json: { messageId } }).catch((err) => {
|
|
1730
|
-
log$
|
|
1865
|
+
log$11.debug({
|
|
1731
1866
|
err,
|
|
1732
1867
|
event: "heartbeat_failed"
|
|
1733
1868
|
}, "heartbeat failed");
|
|
@@ -1739,7 +1874,7 @@ async function resolveConversationFromApi(messageId) {
|
|
|
1739
1874
|
try {
|
|
1740
1875
|
const res = await client["message-conversation"].$get({ query: { messageId } });
|
|
1741
1876
|
if (!res.ok) {
|
|
1742
|
-
log$
|
|
1877
|
+
log$11.warn({
|
|
1743
1878
|
status: res.status,
|
|
1744
1879
|
messageId,
|
|
1745
1880
|
event: "resolve_conversation_failed"
|
|
@@ -1748,7 +1883,7 @@ async function resolveConversationFromApi(messageId) {
|
|
|
1748
1883
|
}
|
|
1749
1884
|
return (await res.json()).conversationId ?? null;
|
|
1750
1885
|
} catch (err) {
|
|
1751
|
-
log$
|
|
1886
|
+
log$11.warn({
|
|
1752
1887
|
err,
|
|
1753
1888
|
messageId,
|
|
1754
1889
|
event: "resolve_conversation_error"
|
|
@@ -1772,8 +1907,19 @@ async function postSubagentSpawn({ messageId, tasks }) {
|
|
|
1772
1907
|
messageId,
|
|
1773
1908
|
tasks
|
|
1774
1909
|
} });
|
|
1775
|
-
if (!res.ok)
|
|
1776
|
-
|
|
1910
|
+
if (!res.ok) {
|
|
1911
|
+
let detail = "";
|
|
1912
|
+
try {
|
|
1913
|
+
const errBody = await res.json();
|
|
1914
|
+
if (errBody && typeof errBody.error === "string") detail = `: ${errBody.error}`;
|
|
1915
|
+
} catch {}
|
|
1916
|
+
throw new Error(`subagent-spawn POST failed (${res.status})${detail}`);
|
|
1917
|
+
}
|
|
1918
|
+
const body = await res.json();
|
|
1919
|
+
return {
|
|
1920
|
+
taskIds: body.taskIds,
|
|
1921
|
+
tasks: body.tasks ?? []
|
|
1922
|
+
};
|
|
1777
1923
|
}
|
|
1778
1924
|
function createHeartbeatThrottle({ messageId }) {
|
|
1779
1925
|
let lastAt = 0;
|
|
@@ -1822,7 +1968,7 @@ function createToolHeartbeat({ messageId }) {
|
|
|
1822
1968
|
}
|
|
1823
1969
|
heartbeatCount++;
|
|
1824
1970
|
if (heartbeatCount > MAX_TOOL_HEARTBEATS) {
|
|
1825
|
-
log$
|
|
1971
|
+
log$11.warn({
|
|
1826
1972
|
heartbeatCount,
|
|
1827
1973
|
activeToolCalls: [...activeToolCalls]
|
|
1828
1974
|
}, "tool heartbeat max reached, stopping");
|
|
@@ -1853,7 +1999,7 @@ function postToDaemon(path, body) {
|
|
|
1853
1999
|
headers: { "content-type": "application/json" },
|
|
1854
2000
|
body: JSON.stringify(body)
|
|
1855
2001
|
}).catch((err) => {
|
|
1856
|
-
log$
|
|
2002
|
+
log$11.debug({
|
|
1857
2003
|
err,
|
|
1858
2004
|
path,
|
|
1859
2005
|
event: "daemon_post_failed"
|
|
@@ -1862,7 +2008,7 @@ function postToDaemon(path, body) {
|
|
|
1862
2008
|
}
|
|
1863
2009
|
function createPlatformExtensions({ sessionId, channelContext }) {
|
|
1864
2010
|
return (pi) => {
|
|
1865
|
-
log$
|
|
2011
|
+
log$11.info({
|
|
1866
2012
|
sessionId,
|
|
1867
2013
|
hasChannelContext: Boolean(channelContext)
|
|
1868
2014
|
}, "platform extension initialized");
|
|
@@ -1911,7 +2057,7 @@ function createPlatformExtensions({ sessionId, channelContext }) {
|
|
|
1911
2057
|
});
|
|
1912
2058
|
});
|
|
1913
2059
|
pi.on("agent_end", () => {
|
|
1914
|
-
log$
|
|
2060
|
+
log$11.info({ sessionId }, "session ending");
|
|
1915
2061
|
postToDaemon("/session/end", { sessionId });
|
|
1916
2062
|
});
|
|
1917
2063
|
};
|
|
@@ -1922,37 +2068,29 @@ function createPlatformExtensions({ sessionId, channelContext }) {
|
|
|
1922
2068
|
* Shared harness feature-flag poll.
|
|
1923
2069
|
*
|
|
1924
2070
|
* The api exposes one `/feature-flags` GET that returns every harness flag in a
|
|
1925
|
-
* single response (`{ contextManagement,
|
|
2071
|
+
* single response (`{ contextManagement, commandFlags }` — see
|
|
1926
2072
|
* apps/anyone/api/src/routes/sandbox-feature-flags.ts). Rather than each
|
|
1927
2073
|
* extension issuing its own GET — and, worse, a *blocking* GET on the
|
|
1928
2074
|
* pre-first-token `session_start` path — a single background poller fetches
|
|
1929
2075
|
* that response once per interval and fans the values out to every subscriber.
|
|
1930
2076
|
*
|
|
1931
|
-
* Why one poller:
|
|
1932
|
-
*
|
|
1933
|
-
*
|
|
1934
|
-
*
|
|
1935
|
-
* every session, flag on or off. Reading the last-polled value instead keeps
|
|
1936
|
-
* the hot path allocation-only. A cold cache reads as `null` (fail-open to
|
|
1937
|
-
* unregistered); a newly-flipped flag takes effect on the next poll, matching
|
|
1938
|
-
* how context-management already treats its flag.
|
|
2077
|
+
* Why one poller: context-management consumes the `contextManagement` flag
|
|
2078
|
+
* without a blocking GET on the pre-first-token `session_start` path. Reading
|
|
2079
|
+
* the last-polled value keeps the hot path allocation-only; a cold cache reads
|
|
2080
|
+
* as `null` and a newly-flipped flag takes effect on the next poll.
|
|
1939
2081
|
*
|
|
1940
2082
|
* The poll is fire-and-forget and self-unref'd — it never keeps the process
|
|
1941
2083
|
* alive and an indeterminate result (no api url / transient failure) leaves the
|
|
1942
2084
|
* last-known values untouched so a blip can't silently flip behavior.
|
|
1943
2085
|
*/
|
|
1944
|
-
const log$
|
|
2086
|
+
const log$10 = logger.child({ module: "feature-flags-poll" });
|
|
1945
2087
|
const FLAG_POLL_INTERVAL_MS = 6e4;
|
|
1946
2088
|
let contextManagement = null;
|
|
1947
|
-
|
|
1948
|
-
const subscribers = {
|
|
1949
|
-
contextManagement: /* @__PURE__ */ new Set(),
|
|
1950
|
-
subagent: /* @__PURE__ */ new Set()
|
|
1951
|
-
};
|
|
2089
|
+
const subscribers = { contextManagement: /* @__PURE__ */ new Set() };
|
|
1952
2090
|
let pollerStarted = false;
|
|
1953
2091
|
let firstPollSettled = false;
|
|
1954
2092
|
let resolveFirstPoll = null;
|
|
1955
|
-
|
|
2093
|
+
new Promise((resolve) => {
|
|
1956
2094
|
resolveFirstPoll = resolve;
|
|
1957
2095
|
});
|
|
1958
2096
|
function markFirstPollSettled() {
|
|
@@ -1961,20 +2099,8 @@ function markFirstPollSettled() {
|
|
|
1961
2099
|
resolveFirstPoll?.();
|
|
1962
2100
|
}
|
|
1963
2101
|
/** Last-polled value of a flag, or `null` if not yet resolved. */
|
|
1964
|
-
function getPolledFlag(
|
|
1965
|
-
return
|
|
1966
|
-
}
|
|
1967
|
-
/**
|
|
1968
|
-
* Await the first poll already kicked by `startFeatureFlagPoller` (never a new
|
|
1969
|
-
* GET). Resolves when that poll settles, immediately if it already has, or
|
|
1970
|
-
* immediately when there's no flag source to poll. Callers on the hot path
|
|
1971
|
-
* should race this against their own short timeout so a slow/failed flag
|
|
1972
|
-
* service cannot delay first-token; a timeout just means the caller reads the
|
|
1973
|
-
* still-cold cache and falls back to its default, exactly as before.
|
|
1974
|
-
*/
|
|
1975
|
-
function awaitFirstFlagPoll() {
|
|
1976
|
-
if (firstPollSettled || !hasFlagSource()) return Promise.resolve();
|
|
1977
|
-
return firstPollPromise;
|
|
2102
|
+
function getPolledFlag(_name) {
|
|
2103
|
+
return contextManagement;
|
|
1978
2104
|
}
|
|
1979
2105
|
/**
|
|
1980
2106
|
* Subscribe to changes of a flag. The callback fires only on a *transition*
|
|
@@ -1987,13 +2113,12 @@ function onFlagChange(name, cb) {
|
|
|
1987
2113
|
}
|
|
1988
2114
|
function apply(name, next) {
|
|
1989
2115
|
if (next === null) return;
|
|
1990
|
-
const prev =
|
|
1991
|
-
|
|
1992
|
-
else subagent = next;
|
|
2116
|
+
const prev = contextManagement;
|
|
2117
|
+
contextManagement = next;
|
|
1993
2118
|
if (next !== prev) for (const cb of subscribers[name]) try {
|
|
1994
2119
|
cb(next);
|
|
1995
2120
|
} catch (err) {
|
|
1996
|
-
log$
|
|
2121
|
+
log$10.warn({
|
|
1997
2122
|
err,
|
|
1998
2123
|
flag: name
|
|
1999
2124
|
}, "flag subscriber threw");
|
|
@@ -2004,9 +2129,8 @@ async function pollOnce() {
|
|
|
2004
2129
|
const flags = await fetchHarnessFlags();
|
|
2005
2130
|
if (!flags) return;
|
|
2006
2131
|
apply("contextManagement", flags.contextManagement ?? null);
|
|
2007
|
-
apply("subagent", flags.subagent ?? null);
|
|
2008
2132
|
} catch (err) {
|
|
2009
|
-
log$
|
|
2133
|
+
log$10.debug({ err }, "feature-flag poll threw");
|
|
2010
2134
|
}
|
|
2011
2135
|
}
|
|
2012
2136
|
/**
|
|
@@ -2113,7 +2237,7 @@ function transformContextMessages(messages, config, now) {
|
|
|
2113
2237
|
}
|
|
2114
2238
|
//#endregion
|
|
2115
2239
|
//#region src/extensions/context-management.ts
|
|
2116
|
-
const log$
|
|
2240
|
+
const log$9 = logger.child({ module: "context-management-extension" });
|
|
2117
2241
|
function isAnthropicMessagesPayload(payload) {
|
|
2118
2242
|
if (typeof payload !== "object" || payload === null) return false;
|
|
2119
2243
|
const candidate = payload;
|
|
@@ -2178,13 +2302,13 @@ function createContextManagementExtension() {
|
|
|
2178
2302
|
setContextManagementFlagOverride(getPolledFlag("contextManagement"));
|
|
2179
2303
|
onFlagChange("contextManagement", (enabled) => {
|
|
2180
2304
|
setContextManagementFlagOverride(enabled);
|
|
2181
|
-
log$
|
|
2305
|
+
log$9.info({
|
|
2182
2306
|
event: "context_management_flag_update",
|
|
2183
2307
|
enabled
|
|
2184
2308
|
}, "context-management flag updated from platform");
|
|
2185
2309
|
});
|
|
2186
2310
|
startFeatureFlagPoller();
|
|
2187
|
-
log$
|
|
2311
|
+
log$9.info({
|
|
2188
2312
|
event: "context_management_registered",
|
|
2189
2313
|
enabled: initial.enabled,
|
|
2190
2314
|
flagSource: hasFlagSource(),
|
|
@@ -2196,13 +2320,13 @@ function createContextManagementExtension() {
|
|
|
2196
2320
|
const { messages } = event;
|
|
2197
2321
|
try {
|
|
2198
2322
|
const result = transformContextIfEnabled(messages, getContextManagementConfig(), Date.now());
|
|
2199
|
-
if (result.stats && (result.stats.clearedResults > 0 || result.stats.trimmedResults > 0)) log$
|
|
2323
|
+
if (result.stats && (result.stats.clearedResults > 0 || result.stats.trimmedResults > 0)) log$9.info({
|
|
2200
2324
|
event: "context_management_applied",
|
|
2201
2325
|
...result.stats
|
|
2202
2326
|
}, "trimmed/cleared tool output before LLM call");
|
|
2203
2327
|
return { messages: result.messages };
|
|
2204
2328
|
} catch (err) {
|
|
2205
|
-
log$
|
|
2329
|
+
log$9.error({
|
|
2206
2330
|
err,
|
|
2207
2331
|
event: "context_management_transform_failed"
|
|
2208
2332
|
}, "context transform failed; passing messages through unchanged");
|
|
@@ -2214,7 +2338,7 @@ function createContextManagementExtension() {
|
|
|
2214
2338
|
}
|
|
2215
2339
|
//#endregion
|
|
2216
2340
|
//#region src/extensions/current-time.ts
|
|
2217
|
-
const log$
|
|
2341
|
+
const log$8 = logger.child({ module: "current-time-extension" });
|
|
2218
2342
|
const PI_DATE_LINE = /^Current date:.*$/m;
|
|
2219
2343
|
function formatCurrentTimeLine(now) {
|
|
2220
2344
|
return `Current date: ${now.getUTCFullYear()}-${String(now.getUTCMonth() + 1).padStart(2, "0")}-${String(now.getUTCDate()).padStart(2, "0")} (${new Intl.DateTimeFormat("en-US", {
|
|
@@ -2227,7 +2351,7 @@ const currentTimeExtension = (pi) => {
|
|
|
2227
2351
|
const line = formatCurrentTimeLine(/* @__PURE__ */ new Date());
|
|
2228
2352
|
const base = event.systemPrompt;
|
|
2229
2353
|
if (PI_DATE_LINE.test(base)) {
|
|
2230
|
-
log$
|
|
2354
|
+
log$8.info({ event: "pi_date_line_present" }, "pi base prompt carries its own 'Current date:' line again; replacing it in place (pi prompt format may have changed)");
|
|
2231
2355
|
return { systemPrompt: base.replace(PI_DATE_LINE, line) };
|
|
2232
2356
|
}
|
|
2233
2357
|
return { systemPrompt: `${base}\n${line}` };
|
|
@@ -2464,7 +2588,7 @@ function renderIndex(entries) {
|
|
|
2464
2588
|
}
|
|
2465
2589
|
//#endregion
|
|
2466
2590
|
//#region src/extensions/memory.ts
|
|
2467
|
-
const log$
|
|
2591
|
+
const log$7 = logger.child({ module: "memory-extension" });
|
|
2468
2592
|
/**
|
|
2469
2593
|
* The standing instructions for the memory system. Always injected (even with
|
|
2470
2594
|
* an empty `.memory/`) so the agent knows it can persist notes. `users/` is
|
|
@@ -2498,7 +2622,7 @@ const memoryExtension = (pi) => {
|
|
|
2498
2622
|
index
|
|
2499
2623
|
});
|
|
2500
2624
|
} catch (err) {
|
|
2501
|
-
log$
|
|
2625
|
+
log$7.warn({
|
|
2502
2626
|
err,
|
|
2503
2627
|
event: "memory_index_failed"
|
|
2504
2628
|
}, "memory index build failed; injecting instructions only");
|
|
@@ -2512,7 +2636,7 @@ const memoryExtension = (pi) => {
|
|
|
2512
2636
|
};
|
|
2513
2637
|
//#endregion
|
|
2514
2638
|
//#region src/extensions/platform-memory.ts
|
|
2515
|
-
const log$
|
|
2639
|
+
const log$6 = logger.child({ module: "platform-memory-extension" });
|
|
2516
2640
|
/**
|
|
2517
2641
|
* Resolve the human on this turn via the API, keyed by the message id.
|
|
2518
2642
|
* `/sandbox/channel-context` only returns a sender for a platform-known
|
|
@@ -2525,13 +2649,13 @@ const log$5 = logger.child({ module: "platform-memory-extension" });
|
|
|
2525
2649
|
async function resolveTurnUser(messageId) {
|
|
2526
2650
|
const client = sandboxClient();
|
|
2527
2651
|
if (!client) {
|
|
2528
|
-
log$
|
|
2652
|
+
log$6.debug({ event: "resolve_turn_user_no_api_url" }, "no API url in env; withholding user memory");
|
|
2529
2653
|
return null;
|
|
2530
2654
|
}
|
|
2531
2655
|
try {
|
|
2532
2656
|
const res = await client["channel-context"].$get({ query: { messageId } });
|
|
2533
2657
|
if (!res.ok) {
|
|
2534
|
-
log$
|
|
2658
|
+
log$6.warn({
|
|
2535
2659
|
event: "resolve_turn_user_failed",
|
|
2536
2660
|
status: res.status
|
|
2537
2661
|
}, "channel-context returned non-ok; withholding user memory");
|
|
@@ -2544,7 +2668,7 @@ async function resolveTurnUser(messageId) {
|
|
|
2544
2668
|
displayName: sender.displayName
|
|
2545
2669
|
};
|
|
2546
2670
|
} catch (err) {
|
|
2547
|
-
log$
|
|
2671
|
+
log$6.warn({
|
|
2548
2672
|
err,
|
|
2549
2673
|
event: "resolve_turn_user_failed"
|
|
2550
2674
|
}, "failed to resolve current user; withholding user memory");
|
|
@@ -2591,7 +2715,7 @@ function createPlatformMemoryExtension({ channelContext }) {
|
|
|
2591
2715
|
user
|
|
2592
2716
|
});
|
|
2593
2717
|
} catch (err) {
|
|
2594
|
-
log$
|
|
2718
|
+
log$6.warn({
|
|
2595
2719
|
err,
|
|
2596
2720
|
event: "user_memory_index_failed"
|
|
2597
2721
|
}, "user memory index build failed; skipping injection");
|
|
@@ -2606,7 +2730,7 @@ function createPlatformMemoryExtension({ channelContext }) {
|
|
|
2606
2730
|
}
|
|
2607
2731
|
//#endregion
|
|
2608
2732
|
//#region src/extensions/self-trace.ts
|
|
2609
|
-
const log$
|
|
2733
|
+
const log$5 = logger.child({ module: "self-trace-extension" });
|
|
2610
2734
|
/**
|
|
2611
2735
|
* Reports the agent's own execution as OpenTelemetry spans:
|
|
2612
2736
|
* agent.session → agent.run → agent.turn.N → tool.NAME, with token/cost
|
|
@@ -2631,7 +2755,7 @@ const selfTraceExtension = (pi) => {
|
|
|
2631
2755
|
sessionSpan = tracer.startSpan("agent.session", { attributes: { "agent.model": modelId } }, remoteCtx);
|
|
2632
2756
|
sessionCtx = trace.setSpan(remoteCtx, sessionSpan);
|
|
2633
2757
|
const sc = sessionSpan.spanContext();
|
|
2634
|
-
log$
|
|
2758
|
+
log$5.info({
|
|
2635
2759
|
event: "self_trace_session_start",
|
|
2636
2760
|
trace_id: sc.traceId,
|
|
2637
2761
|
span_id: sc.spanId,
|
|
@@ -2743,13 +2867,13 @@ const selfTraceExtension = (pi) => {
|
|
|
2743
2867
|
* Lives in the harness package — soul.md is content from the agent's
|
|
2744
2868
|
* own git repo, not from the platform — so its handling stays here.
|
|
2745
2869
|
*/
|
|
2746
|
-
const log$
|
|
2870
|
+
const log$4 = logger.child({ module: "soul-extension" });
|
|
2747
2871
|
async function readSoul(cwd) {
|
|
2748
2872
|
try {
|
|
2749
2873
|
return (await readFile(join(cwd, "soul.md"), "utf8")).trim() || null;
|
|
2750
2874
|
} catch (err) {
|
|
2751
2875
|
if (err?.code === "ENOENT") return null;
|
|
2752
|
-
log$
|
|
2876
|
+
log$4.warn({
|
|
2753
2877
|
err,
|
|
2754
2878
|
event: "soul_read_failed"
|
|
2755
2879
|
}, "soul.md read failed");
|
|
@@ -2763,7 +2887,7 @@ function soulSection(cwd, soul) {
|
|
|
2763
2887
|
|
|
2764
2888
|
**\`soul.md\` is where behavior lives.** Any standing instruction about how you should act — a rule a user wants you to follow going forward, a tone or format preference, a workflow convention, a "from now on, always/never …" — belongs here, not in \`.memory/\`. Memory records *what happened* (facts, events, findings); soul defines *how you behave*. When a user gives you a durable behavioral rule, write it to \`soul.md\`. If you find behavioral rules that ended up in \`.memory/\`, treat that as misfiled and move them here.
|
|
2765
2889
|
|
|
2766
|
-
Keep it current. When you gain a durable new capability — a tool you build, a skill or integration you set up, a service you connect — or a user hands you a lasting behavioral rule, record it in \`soul.md\` so a future conversation knows it's part of you rather than rediscovering it from scratch.
|
|
2890
|
+
Keep it current. When you gain a durable new capability — a tool you build, a skill or integration you set up, a service you connect, a secret or auth credential you wire in — or a user hands you a lasting behavioral rule, record it in \`soul.md\` so a future conversation knows it's part of you rather than rediscovering it from scratch. Do this the moment you gain the capability, and for a credential that means the moment it verifies with a real call, not after a human points out that you forgot. Connecting a capability is itself a durable change worth recording, not merely a step toward the task in front of you. Edit \`soul.md\` (then \`git add soul.md && git commit && git push\`) to redefine yourself; picked up on the next message.
|
|
2767
2891
|
|
|
2768
2892
|
${soul ? soul : "_(empty — write to `soul.md` to define your persona)_"}`;
|
|
2769
2893
|
}
|
|
@@ -2781,8 +2905,7 @@ const soulExtension = (pi) => {
|
|
|
2781
2905
|
};
|
|
2782
2906
|
//#endregion
|
|
2783
2907
|
//#region src/extensions/subagent/index.ts
|
|
2784
|
-
const log$
|
|
2785
|
-
const COLD_START_FLAG_WAIT_MS = 750;
|
|
2908
|
+
const log$3 = logger.child({ module: "subagent-ext" });
|
|
2786
2909
|
const MAX_TASKS = 8;
|
|
2787
2910
|
const TaskItem = Type.Object({
|
|
2788
2911
|
task: Type.String({ description: "The task to delegate to a subagent run." }),
|
|
@@ -2791,8 +2914,14 @@ const TaskItem = Type.Object({
|
|
|
2791
2914
|
maxLength: 120
|
|
2792
2915
|
}),
|
|
2793
2916
|
persona: Type.Optional(Type.String({ description: "Optional extra system prompt / role for this task, applied ON TOP of the child run's own default persona (your full identity and soul are still there underneath). Omit to run with just your default persona." })),
|
|
2794
|
-
model: Type.Optional(Type.String({ description: "Optional model id to run this subagent on
|
|
2917
|
+
model: Type.Optional(Type.String({ description: "Optional model id to run this subagent on. PREFER A LOWER-COST, FASTER MODEL when the task is well-scoped and does not need your full reasoning depth — most delegated subtasks (searching, summarizing, mechanical edits, gathering or reformatting data, running a check) run just as well on a lighter model and cost far less. Reserve a top-tier model for subtasks that genuinely need deep reasoning or careful judgment. Must be a real catalogued model id. Omit to inherit your own model. If you are locked to a Google-compliant model, only compliant models are accepted." })),
|
|
2918
|
+
timeoutMinutes: Type.Optional(Type.Integer({
|
|
2919
|
+
description: "Optional wall-clock timeout for this subagent, in minutes. If the run is still going after this long it is ended and you are rewoken with a timeout result, so a hung subagent can never strand you. Omit for the default (30 minutes). Raise it for genuinely long work (a big migration, a large audit); lower it for a quick lookup. Range 1-360.",
|
|
2920
|
+
minimum: 1,
|
|
2921
|
+
maximum: 360
|
|
2922
|
+
}))
|
|
2795
2923
|
});
|
|
2924
|
+
const DEFAULT_SUBAGENT_TIMEOUT_MS = 30 * 6e4;
|
|
2796
2925
|
const SubagentParams = Type.Object({ tasks: Type.Array(TaskItem, {
|
|
2797
2926
|
description: "One or more tasks to delegate. Each spawns an isolated subagent run linked to this conversation; they run in parallel and each rewakes you with its result when it finishes.",
|
|
2798
2927
|
minItems: 1,
|
|
@@ -2807,7 +2936,9 @@ function buildTool(messageId) {
|
|
|
2807
2936
|
"Delegate one or more tasks to subagent runs — fresh isolated copies of yourself, each with its own context window, linked to this conversation.",
|
|
2808
2937
|
"Use it to parallelize independent work, to keep a large or noisy subtask out of your own context, or to run a task under a specialized persona.",
|
|
2809
2938
|
"Fire-and-forget: this returns immediately after queueing. It does NOT wait for results. Each subagent runs on its own and, when it finishes, sends you its result on this thread — so queue the work, then keep going or end your turn. To chain, re-delegate after a result lands.",
|
|
2810
|
-
"Pass tasks: [{ task, title, persona?, model? }]. title is a short 3-6 word name for the task — it is shown to the person in the chat as that subagent's row, so name the work rather than restating the prompt. persona is an optional extra system prompt layered ON TOP of your default persona for that task (it adds to, it does not replace, your identity); omit it to run with just your default persona. model is an optional model id for that task; omit it to
|
|
2939
|
+
"Pass tasks: [{ task, title, persona?, model? }]. title is a short 3-6 word name for the task — it is shown to the person in the chat as that subagent's row, so name the work rather than restating the prompt. persona is an optional extra system prompt layered ON TOP of your default persona for that task (it adds to, it does not replace, your identity); omit it to run with just your default persona. model is an optional model id for that task — prefer a lower-cost, faster model for well-scoped subtasks that don't need deep reasoning, and reserve a top-tier model for the ones that do; omit it to inherit your own model.",
|
|
2940
|
+
"Peering: each queued task comes back with its own conversation id. A subagent is a real linked conversation, so to see what one is doing RIGHT NOW while it runs — its reasoning, the tools it has called and their results, its progress — read that conversation with `platform conversations show <conversationId>` (you are already authorized; it is your own delegated run). Check in that way instead of waiting blind for the final result. The read reflects the child's persisted state, which lags a few seconds behind live (tool results land as they complete; in-progress reasoning can be up to ~5s stale), so peek between checkpoints rather than polling in a tight loop.",
|
|
2941
|
+
"Steering: to add context, correct course, or answer a question a subagent needs mid-run, post to its conversation with `platform conversations post <conversationId> --message \"...\"`. If the subagent is still running, your message lands as a live steer picked up in that same turn; if it has gone idle, it queues as its next turn. This is the same primitive as any conversation message — there is no separate steer channel."
|
|
2811
2942
|
].join(" "),
|
|
2812
2943
|
promptSnippet: "subagent — delegate tasks to isolated subagent runs; each rewakes you with its result when done",
|
|
2813
2944
|
parameters: SubagentParams,
|
|
@@ -2825,28 +2956,39 @@ function buildTool(messageId) {
|
|
|
2825
2956
|
task: t.task,
|
|
2826
2957
|
title: t.title ?? null,
|
|
2827
2958
|
persona: t.persona ?? null,
|
|
2828
|
-
model: t.model ?? null
|
|
2959
|
+
model: t.model ?? null,
|
|
2960
|
+
timeoutMs: t.timeoutMinutes != null ? t.timeoutMinutes * 6e4 : DEFAULT_SUBAGENT_TIMEOUT_MS
|
|
2829
2961
|
}));
|
|
2830
2962
|
try {
|
|
2831
|
-
const
|
|
2963
|
+
const spawned = await postSubagentSpawn({
|
|
2832
2964
|
messageId,
|
|
2833
2965
|
tasks: spawnTasks
|
|
2834
2966
|
});
|
|
2835
|
-
|
|
2967
|
+
const { taskIds } = spawned;
|
|
2968
|
+
log$3.info({
|
|
2836
2969
|
event: "subagent_spawned",
|
|
2837
2970
|
count: taskIds.length
|
|
2838
2971
|
}, "subagent tasks queued");
|
|
2839
|
-
const
|
|
2972
|
+
const convByTask = new Map(spawned.tasks.map((t) => [t.taskId, t.conversationId]));
|
|
2973
|
+
const lines = taskIds.map((id, i) => {
|
|
2974
|
+
const label = spawnTasks[i]?.title ?? spawnTasks[i]?.task ?? "";
|
|
2975
|
+
const conv = convByTask.get(id);
|
|
2976
|
+
return `- ${id}: ${label}${conv ? ` — conversation ${conv}` : ""}`;
|
|
2977
|
+
}).join("\n");
|
|
2978
|
+
const peerHint = spawned.tasks.length ? "\nEach subagent runs on its own conversation (id shown per task above). To SEE what one is doing while it runs, read it with `platform conversations show <conversationId>`. To STEER one mid-run — add context, correct course, answer a question — post to its conversation with `platform conversations post <conversationId> --message \"...\"`; it lands as a live steer if the subagent is still running, or as its next turn if it has gone idle." : "";
|
|
2840
2979
|
return {
|
|
2841
2980
|
content: [{
|
|
2842
2981
|
type: "text",
|
|
2843
|
-
text: `Queued ${taskIds.length} subagent ${taskIds.length === 1 ? "run" : "runs"}. Each runs on its own and will send you its result on this thread when it finishes — keep working or end your turn meanwhile.\n${lines}`
|
|
2982
|
+
text: `Queued ${taskIds.length} subagent ${taskIds.length === 1 ? "run" : "runs"}. Each runs on its own and will send you its result on this thread when it finishes — keep working or end your turn meanwhile.\n${lines}${peerHint}`
|
|
2844
2983
|
}],
|
|
2845
|
-
details: {
|
|
2984
|
+
details: {
|
|
2985
|
+
taskIds,
|
|
2986
|
+
tasks: spawned.tasks
|
|
2987
|
+
}
|
|
2846
2988
|
};
|
|
2847
2989
|
} catch (err) {
|
|
2848
2990
|
const message = err instanceof Error ? err.message : String(err);
|
|
2849
|
-
log$
|
|
2991
|
+
log$3.warn({
|
|
2850
2992
|
err,
|
|
2851
2993
|
event: "subagent_spawn_failed"
|
|
2852
2994
|
}, "subagent spawn failed");
|
|
@@ -2863,33 +3005,240 @@ function buildTool(messageId) {
|
|
|
2863
3005
|
};
|
|
2864
3006
|
}
|
|
2865
3007
|
/**
|
|
2866
|
-
* Gated on `harness-subagent-enabled`, read from the shared feature-flag poll.
|
|
2867
3008
|
* The factory takes the session's channel context to resolve the originating
|
|
2868
3009
|
* messageId — the api links each spawned run to the conversation that message
|
|
2869
3010
|
* belongs to and rewakes it on completion (nothing about the parent is piped
|
|
2870
|
-
* from the sandbox beyond that id).
|
|
3011
|
+
* from the sandbox beyond that id). The tool is registered unconditionally at
|
|
3012
|
+
* session_start.
|
|
2871
3013
|
*/
|
|
2872
3014
|
function createSubagentExtension({ channelContext }) {
|
|
2873
3015
|
return (pi) => {
|
|
2874
3016
|
const messageId = extractMessageId(channelContext);
|
|
2875
|
-
startFeatureFlagPoller();
|
|
2876
3017
|
let registered = false;
|
|
2877
3018
|
const registerOnce = () => {
|
|
2878
3019
|
if (registered) return;
|
|
2879
3020
|
registered = true;
|
|
2880
3021
|
pi.registerTool(buildTool(messageId));
|
|
2881
|
-
log$
|
|
3022
|
+
log$3.info({ event: "subagent_enabled" }, "subagent tool registered");
|
|
2882
3023
|
};
|
|
2883
|
-
|
|
2884
|
-
|
|
2885
|
-
});
|
|
2886
|
-
pi.on("session_start", async () => {
|
|
2887
|
-
if (getPolledFlag("subagent") === null) await Promise.race([awaitFirstFlagPoll(), new Promise((resolve) => setTimeout(resolve, COLD_START_FLAG_WAIT_MS).unref?.())]);
|
|
2888
|
-
if (getPolledFlag("subagent") === true) registerOnce();
|
|
3024
|
+
pi.on("session_start", () => {
|
|
3025
|
+
registerOnce();
|
|
2889
3026
|
});
|
|
2890
3027
|
};
|
|
2891
3028
|
}
|
|
2892
3029
|
//#endregion
|
|
3030
|
+
//#region src/extensions/resource-pressure-warning.ts
|
|
3031
|
+
/**
|
|
3032
|
+
* Mid-run resource-pressure warning to the agent.
|
|
3033
|
+
*
|
|
3034
|
+
* The sandbox already detects pressure — the boot scripts cap the
|
|
3035
|
+
* user-workload cgroup (memory.high/memory.max) and watchers log warn/crit
|
|
3036
|
+
* edges for memory and disk — but nothing told the *agent*, so a turn burned
|
|
3037
|
+
* straight to the OOM kill (or a full disk) and only learned about it from
|
|
3038
|
+
* the post-mortem notice. This extension closes that gap in-process: while a
|
|
3039
|
+
* turn is active it polls the agent cgroup and the root filesystem and, the
|
|
3040
|
+
* first time usage crosses a warn threshold, folds a system notification into
|
|
3041
|
+
* the open turn so the agent can checkpoint, shed work (constrain
|
|
3042
|
+
* parallelism, kill a background hog, clean scratch space), or request a
|
|
3043
|
+
* bigger tier BEFORE the kill.
|
|
3044
|
+
*
|
|
3045
|
+
* The notification is triggered by the two conditions that actually kill
|
|
3046
|
+
* work — memory near the cgroup hard cap, disk near full — and reports a
|
|
3047
|
+
* snapshot of all the relevant stats (memory, CPU utilization, disk) so the
|
|
3048
|
+
* agent can tell which resource is the problem and how much headroom the
|
|
3049
|
+
* others have.
|
|
3050
|
+
*
|
|
3051
|
+
* Edge-triggered, once per trigger per turn: the fired flags reset on
|
|
3052
|
+
* agent_start, so a turn that rides a threshold gets one warning per
|
|
3053
|
+
* resource, not a stream. Polling only runs while the agent is active — an
|
|
3054
|
+
* idle sandbox's resource usage is not the agent's problem and there is no
|
|
3055
|
+
* open turn to deliver into anyway.
|
|
3056
|
+
*
|
|
3057
|
+
* Best-effort throughout: any read failure (cgroup absent, controller not
|
|
3058
|
+
* delegated, non-cgroup-v2 host, df missing) reads as "no signal" for that
|
|
3059
|
+
* stat and the extension warns on what it can see — it must never break a
|
|
3060
|
+
* turn over an observability feature.
|
|
3061
|
+
*/
|
|
3062
|
+
const execFileAsync = promisify(execFile);
|
|
3063
|
+
const log$2 = logger.child({ module: "resource-pressure-warning" });
|
|
3064
|
+
const POLL_INTERVAL_MS = 1e4;
|
|
3065
|
+
function envOverride(name) {
|
|
3066
|
+
for (const prefix of ["SKYDIVE_", "ANYONE_"]) {
|
|
3067
|
+
const value = process.env[`${prefix}${name}`];
|
|
3068
|
+
if (value != null && value !== "") return value;
|
|
3069
|
+
}
|
|
3070
|
+
return null;
|
|
3071
|
+
}
|
|
3072
|
+
function cgroupDir() {
|
|
3073
|
+
return envOverride("AGENT_CGROUP") ?? "/sys/fs/cgroup/agent";
|
|
3074
|
+
}
|
|
3075
|
+
function diskRoot() {
|
|
3076
|
+
return envOverride("DISK_ROOT") ?? "/";
|
|
3077
|
+
}
|
|
3078
|
+
/**
|
|
3079
|
+
* Read a cgroup v2 scalar file. Returns a number, or null for "max"
|
|
3080
|
+
* (uncapped), an empty/absent file, or any read/parse error — an uncapped or
|
|
3081
|
+
* unreadable limit means there is nothing meaningful to warn against.
|
|
3082
|
+
*/
|
|
3083
|
+
async function readScalar(file) {
|
|
3084
|
+
try {
|
|
3085
|
+
const raw = (await readFile(`${cgroupDir()}/${file}`, "utf8")).trim();
|
|
3086
|
+
if (raw === "" || raw === "max") return null;
|
|
3087
|
+
const n = Number(raw);
|
|
3088
|
+
return Number.isFinite(n) ? n : null;
|
|
3089
|
+
} catch (_error) {
|
|
3090
|
+
return null;
|
|
3091
|
+
}
|
|
3092
|
+
}
|
|
3093
|
+
/**
|
|
3094
|
+
* Read a cgroup v2 "flat keyed" file (one `key value` pair per line, e.g.
|
|
3095
|
+
* cpu.stat) and return the counter for `key`, or null when absent.
|
|
3096
|
+
*/
|
|
3097
|
+
async function readKeyedCounter(file, key) {
|
|
3098
|
+
try {
|
|
3099
|
+
const raw = await readFile(`${cgroupDir()}/${file}`, "utf8");
|
|
3100
|
+
for (const line of raw.split("\n")) {
|
|
3101
|
+
const [k, v] = line.trim().split(/\s+/);
|
|
3102
|
+
if (k === key) {
|
|
3103
|
+
const n = Number(v);
|
|
3104
|
+
return Number.isFinite(n) ? n : null;
|
|
3105
|
+
}
|
|
3106
|
+
}
|
|
3107
|
+
return null;
|
|
3108
|
+
} catch (_error) {
|
|
3109
|
+
return null;
|
|
3110
|
+
}
|
|
3111
|
+
}
|
|
3112
|
+
/**
|
|
3113
|
+
* Live memory usage as an integer percent of the hard cap, or null when
|
|
3114
|
+
* either side is unreadable/uncapped. Exported for tests.
|
|
3115
|
+
*/
|
|
3116
|
+
async function readMemUsePct() {
|
|
3117
|
+
const [current, max] = await Promise.all([readScalar("memory.current"), readScalar("memory.max")]);
|
|
3118
|
+
if (current === null || max === null || max <= 0) return null;
|
|
3119
|
+
return {
|
|
3120
|
+
pct: Math.floor(current / max * 100),
|
|
3121
|
+
currentBytes: current,
|
|
3122
|
+
maxBytes: max
|
|
3123
|
+
};
|
|
3124
|
+
}
|
|
3125
|
+
/**
|
|
3126
|
+
* Root filesystem used% (df -P Capacity column), or null on any failure.
|
|
3127
|
+
* Exported for tests.
|
|
3128
|
+
*/
|
|
3129
|
+
async function readDiskUsePct() {
|
|
3130
|
+
try {
|
|
3131
|
+
const { stdout } = await execFileAsync("df", ["-P", diskRoot()]);
|
|
3132
|
+
const dataRow = stdout.trim().split("\n")[1];
|
|
3133
|
+
if (dataRow == null) return null;
|
|
3134
|
+
const capacity = dataRow.trim().split(/\s+/)[4];
|
|
3135
|
+
if (capacity == null) return null;
|
|
3136
|
+
const pct = Number(capacity.replace("%", ""));
|
|
3137
|
+
return Number.isFinite(pct) ? pct : null;
|
|
3138
|
+
} catch (_error) {
|
|
3139
|
+
return null;
|
|
3140
|
+
}
|
|
3141
|
+
}
|
|
3142
|
+
/**
|
|
3143
|
+
* CPU utilization sampler. cgroup v2 exposes cumulative CPU time
|
|
3144
|
+
* (cpu.stat usage_usec); utilization is the delta between two samples over
|
|
3145
|
+
* the wall time between them, normalized by core count. The first call after
|
|
3146
|
+
* construction has no previous sample and returns null.
|
|
3147
|
+
*/
|
|
3148
|
+
function createCpuSampler() {
|
|
3149
|
+
let prevUsageUsec = null;
|
|
3150
|
+
let prevAtMs = null;
|
|
3151
|
+
return async () => {
|
|
3152
|
+
const usage = await readKeyedCounter("cpu.stat", "usage_usec");
|
|
3153
|
+
const now = Date.now();
|
|
3154
|
+
const prev = prevUsageUsec;
|
|
3155
|
+
const prevAt = prevAtMs;
|
|
3156
|
+
prevUsageUsec = usage;
|
|
3157
|
+
prevAtMs = now;
|
|
3158
|
+
if (usage === null || prev === null || prevAt === null) return null;
|
|
3159
|
+
const wallUsec = (now - prevAt) * 1e3;
|
|
3160
|
+
if (wallUsec <= 0) return null;
|
|
3161
|
+
const cores = availableParallelism();
|
|
3162
|
+
const pct = Math.round((usage - prev) / (wallUsec * cores) * 100);
|
|
3163
|
+
return Math.max(0, Math.min(100, pct));
|
|
3164
|
+
};
|
|
3165
|
+
}
|
|
3166
|
+
function fmtMb(bytes) {
|
|
3167
|
+
return Math.round(bytes / 1024 / 1024);
|
|
3168
|
+
}
|
|
3169
|
+
/** The model-facing warning text. Exported for tests. */
|
|
3170
|
+
function resourcePressureWarningText(trigger, { mem, cpuPct, diskPct }) {
|
|
3171
|
+
const stats = [];
|
|
3172
|
+
if (mem) stats.push(`memory ${mem.pct}% of cap (${fmtMb(mem.currentBytes)}/${fmtMb(mem.maxBytes)} MB)`);
|
|
3173
|
+
if (cpuPct !== null) stats.push(`CPU ${cpuPct}%`);
|
|
3174
|
+
if (diskPct !== null) stats.push(`disk ${diskPct}% full`);
|
|
3175
|
+
const lead = trigger === "memory" ? `Your sandbox is at ${mem?.pct}% of its memory cap. If usage keeps climbing, the kernel will kill the offending process and this turn may die with it.` : `Your sandbox's disk is ${diskPct}% full. If it fills completely, writes will start failing and this turn may die with them.`;
|
|
3176
|
+
const remedy = trigger === "memory" ? "checkpoint in-flight work (commit and push), then reduce the footprint — constrain parallelism, run heavy steps sequentially, or kill background processes you no longer need." : "checkpoint in-flight work (commit and push), then free space — clean build artifacts, caches, and scratch files you no longer need.";
|
|
3177
|
+
return `<system_notification>${lead} Current usage: ${stats.join(", ")}. Act now: ${remedy} If the workload genuinely needs more resources, request a bigger sandbox with \`platform compute request\`. This is an automated resource warning, not a message from the user; continue the task, adjusted.</system_notification>`;
|
|
3178
|
+
}
|
|
3179
|
+
const resourcePressureWarningExtension = (pi) => {
|
|
3180
|
+
let agentActive = false;
|
|
3181
|
+
let warnedMemThisTurn = false;
|
|
3182
|
+
let warnedDiskThisTurn = false;
|
|
3183
|
+
let timer = null;
|
|
3184
|
+
const sampleCpu = createCpuSampler();
|
|
3185
|
+
async function checkOnce() {
|
|
3186
|
+
if (!agentActive || warnedMemThisTurn && warnedDiskThisTurn) return;
|
|
3187
|
+
const [mem, cpuPct, diskPct] = await Promise.all([
|
|
3188
|
+
readMemUsePct(),
|
|
3189
|
+
sampleCpu(),
|
|
3190
|
+
readDiskUsePct()
|
|
3191
|
+
]);
|
|
3192
|
+
let trigger = null;
|
|
3193
|
+
if (!warnedMemThisTurn && mem !== null && mem.pct >= 80) {
|
|
3194
|
+
trigger = "memory";
|
|
3195
|
+
warnedMemThisTurn = true;
|
|
3196
|
+
} else if (!warnedDiskThisTurn && diskPct !== null && diskPct >= 80) {
|
|
3197
|
+
trigger = "disk";
|
|
3198
|
+
warnedDiskThisTurn = true;
|
|
3199
|
+
}
|
|
3200
|
+
if (trigger === null) return;
|
|
3201
|
+
log$2.warn({
|
|
3202
|
+
trigger,
|
|
3203
|
+
mem,
|
|
3204
|
+
cpuPct,
|
|
3205
|
+
diskPct
|
|
3206
|
+
}, "resource pressure warning delivered to agent");
|
|
3207
|
+
await pi.sendMessage({
|
|
3208
|
+
customType: "anyone-resource-pressure-warning",
|
|
3209
|
+
content: resourcePressureWarningText(trigger, {
|
|
3210
|
+
mem,
|
|
3211
|
+
cpuPct,
|
|
3212
|
+
diskPct
|
|
3213
|
+
}),
|
|
3214
|
+
display: false
|
|
3215
|
+
}, {
|
|
3216
|
+
triggerTurn: true,
|
|
3217
|
+
deliverAs: "followUp"
|
|
3218
|
+
});
|
|
3219
|
+
}
|
|
3220
|
+
pi.on("agent_start", async () => {
|
|
3221
|
+
agentActive = true;
|
|
3222
|
+
warnedMemThisTurn = false;
|
|
3223
|
+
warnedDiskThisTurn = false;
|
|
3224
|
+
if (!timer) {
|
|
3225
|
+
timer = setInterval(() => {
|
|
3226
|
+
checkOnce().catch((err) => {
|
|
3227
|
+
log$2.error({ err }, "resource pressure check failed");
|
|
3228
|
+
});
|
|
3229
|
+
}, POLL_INTERVAL_MS);
|
|
3230
|
+
timer.unref?.();
|
|
3231
|
+
}
|
|
3232
|
+
});
|
|
3233
|
+
pi.on("agent_end", async () => {
|
|
3234
|
+
agentActive = false;
|
|
3235
|
+
if (timer) {
|
|
3236
|
+
clearInterval(timer);
|
|
3237
|
+
timer = null;
|
|
3238
|
+
}
|
|
3239
|
+
});
|
|
3240
|
+
};
|
|
3241
|
+
//#endregion
|
|
2893
3242
|
//#region src/extensions/tool-call-env.ts
|
|
2894
3243
|
const TOOL_CALL_ID_VAR = "TOOL_CALL_ID";
|
|
2895
3244
|
function shellQuoteValue(value) {
|
|
@@ -3663,7 +4012,8 @@ function platformExtensions({ sessionId, channelContext }) {
|
|
|
3663
4012
|
selfTraceExtension,
|
|
3664
4013
|
createBackgroundTasksExtension({ channelContext }),
|
|
3665
4014
|
createSubagentExtension({ channelContext }),
|
|
3666
|
-
createContextManagementExtension()
|
|
4015
|
+
createContextManagementExtension(),
|
|
4016
|
+
resourcePressureWarningExtension
|
|
3667
4017
|
];
|
|
3668
4018
|
}
|
|
3669
4019
|
//#endregion
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@skydiveai/pi-extensions",
|
|
3
|
-
"version": "0.1.0-beta.
|
|
3
|
+
"version": "0.1.0-beta.1210",
|
|
4
4
|
"homepage": "https://skydive.com",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "Create, Inc.",
|
|
@@ -17,12 +17,6 @@
|
|
|
17
17
|
},
|
|
18
18
|
"publishConfig": {
|
|
19
19
|
"access": "public",
|
|
20
|
-
"exports": {
|
|
21
|
-
".": {
|
|
22
|
-
"types": "./dist/index.d.mts",
|
|
23
|
-
"default": "./dist/index.mjs"
|
|
24
|
-
}
|
|
25
|
-
},
|
|
26
20
|
"registry": "https://registry.npmjs.org"
|
|
27
21
|
},
|
|
28
22
|
"scripts": {
|