@skydiveai/pi-extensions 0.1.0-beta.310 → 0.1.0-beta.3103
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -0
- package/dist/index.d.mts +56 -3
- package/dist/index.mjs +2560 -929
- package/package.json +2 -9
package/dist/index.mjs
CHANGED
|
@@ -2,29 +2,34 @@ import { createRequire } from "node:module";
|
|
|
2
2
|
import { DefaultExecutionEventBusManager, DefaultRequestHandler, InMemoryTaskStore } from "@a2a-js/sdk/server";
|
|
3
3
|
import { UserBuilder, restHandler } from "@a2a-js/sdk/server/express";
|
|
4
4
|
import { buildAgentCard, chainMiddleware, composeHandlers, createAgentExecutor, createProtocolHandlers, getCurrentTraceparent, logger, mountAt, requestHeaders, requestUrl, webHandlerToMiddleware } from "@skydiveai/pi-server";
|
|
5
|
-
import { mkdir, open, readFile, readdir, stat, unlink } from "node:fs/promises";
|
|
5
|
+
import { mkdir, open, readFile, readdir, stat, unlink, writeFile } from "node:fs/promises";
|
|
6
6
|
import { basename, dirname, join, relative, resolve } from "node:path";
|
|
7
|
-
import { z } from "zod";
|
|
8
7
|
import { pathToFileURL } from "node:url";
|
|
8
|
+
import { createEditToolDefinition, createLocalBashOperations } from "@earendil-works/pi-coding-agent";
|
|
9
|
+
import { z } from "zod";
|
|
9
10
|
import { CallToolResultSchema } from "@modelcontextprotocol/sdk/types.js";
|
|
10
11
|
import { Type } from "typebox";
|
|
11
12
|
import { Client } from "@modelcontextprotocol/sdk/client/index.js";
|
|
12
|
-
import { StdioClientTransport } from "@modelcontextprotocol/sdk/client/stdio.js";
|
|
13
|
+
import { StdioClientTransport, getDefaultEnvironment } from "@modelcontextprotocol/sdk/client/stdio.js";
|
|
13
14
|
import { StreamableHTTPClientTransport, StreamableHTTPError } from "@modelcontextprotocol/sdk/client/streamableHttp.js";
|
|
14
15
|
import { UnauthorizedError } from "@modelcontextprotocol/sdk/client/auth.js";
|
|
15
16
|
import { Check, Errors } from "typebox/value";
|
|
17
|
+
import { hc } from "hono/client";
|
|
16
18
|
import { ROOT_CONTEXT, SpanStatusCode, propagation, trace } from "@opentelemetry/api";
|
|
17
19
|
import { W3CTraceContextPropagator } from "@opentelemetry/core";
|
|
18
20
|
import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-http";
|
|
19
21
|
import { Resource } from "@opentelemetry/resources";
|
|
20
22
|
import { BatchSpanProcessor, NodeTracerProvider } from "@opentelemetry/sdk-trace-node";
|
|
21
23
|
import { ATTR_SERVICE_NAME } from "@opentelemetry/semantic-conventions";
|
|
22
|
-
import {
|
|
24
|
+
import { execFile } from "node:child_process";
|
|
25
|
+
import { availableParallelism } from "node:os";
|
|
26
|
+
import { promisify } from "node:util";
|
|
23
27
|
import { parse } from "yaml";
|
|
28
|
+
import { getModels } from "@earendil-works/pi-ai";
|
|
24
29
|
import { quote } from "shell-quote";
|
|
30
|
+
import { createHash } from "node:crypto";
|
|
25
31
|
import { createWriteStream } from "node:fs";
|
|
26
32
|
import { finished } from "node:stream/promises";
|
|
27
|
-
import { createLocalBashOperations } from "@earendil-works/pi-coding-agent";
|
|
28
33
|
//#region src/platform-env-middleware.ts
|
|
29
34
|
const ENVD_TOKEN_HEADER = "x-e2b-envd-token";
|
|
30
35
|
const ENVD_URL = "http://localhost:49983/envs";
|
|
@@ -238,167 +243,6 @@ function createHealthHandler({ metadata }) {
|
|
|
238
243
|
};
|
|
239
244
|
}
|
|
240
245
|
//#endregion
|
|
241
|
-
//#region src/extensions/context-management-config.ts
|
|
242
|
-
/**
|
|
243
|
-
* Configuration for the context-management capability (tool-output trimming
|
|
244
|
-
* and client-side microcompact). See the design notes in the extension for
|
|
245
|
-
* what each layer does; this module is purely the env → config surface.
|
|
246
|
-
*
|
|
247
|
-
* The harness is provider-agnostic and runs inside the sandbox, where its
|
|
248
|
-
* only config channel is the environment (loaded by the platform env
|
|
249
|
-
* middleware before any session starts — see platform-env-middleware.ts). So
|
|
250
|
-
* the LaunchDarkly flag `harness-context-management-enabled` is resolved by
|
|
251
|
-
* the platform when it provisions the sandbox and passed through as
|
|
252
|
-
* `SKYDIVE_CONTEXT_MANAGEMENT`; the individual knobs override the defaults
|
|
253
|
-
* below when present.
|
|
254
|
-
*
|
|
255
|
-
* Resolution is defensive: a malformed value never throws (this config is
|
|
256
|
-
* read on the hot path before every LLM call), it falls back to the default
|
|
257
|
-
* for that knob and logs once.
|
|
258
|
-
*/
|
|
259
|
-
const log$13 = logger.child({ module: "context-management-config" });
|
|
260
|
-
const DEFAULT_CONTEXT_MANAGEMENT_CONFIG = {
|
|
261
|
-
enabled: false,
|
|
262
|
-
perResultMaxBytes: 16 * 1024,
|
|
263
|
-
keepRecentToolResults: 3,
|
|
264
|
-
coldCacheGapSeconds: 240,
|
|
265
|
-
warmClearTriggerTokens: 5e4,
|
|
266
|
-
clearAtLeastTokens: 1e4,
|
|
267
|
-
excludeTools: [],
|
|
268
|
-
maxModelCallsPerTurn: 80,
|
|
269
|
-
nativeAnthropicEdits: false
|
|
270
|
-
};
|
|
271
|
-
const boolFromEnv = (value, fallback) => {
|
|
272
|
-
if (value === void 0) return fallback;
|
|
273
|
-
const normalized = value.trim().toLowerCase();
|
|
274
|
-
if (normalized === "1" || normalized === "true") return true;
|
|
275
|
-
if (normalized === "0" || normalized === "false") return false;
|
|
276
|
-
return fallback;
|
|
277
|
-
};
|
|
278
|
-
const positiveInt = (fallback) => z.coerce.number().int().positive().catch(fallback);
|
|
279
|
-
const nonNegativeInt = (fallback) => z.coerce.number().int().nonnegative().catch(fallback);
|
|
280
|
-
const configSchema = z.object({
|
|
281
|
-
perResultMaxBytes: positiveInt(DEFAULT_CONTEXT_MANAGEMENT_CONFIG.perResultMaxBytes),
|
|
282
|
-
keepRecentToolResults: nonNegativeInt(DEFAULT_CONTEXT_MANAGEMENT_CONFIG.keepRecentToolResults),
|
|
283
|
-
coldCacheGapSeconds: positiveInt(DEFAULT_CONTEXT_MANAGEMENT_CONFIG.coldCacheGapSeconds),
|
|
284
|
-
warmClearTriggerTokens: positiveInt(DEFAULT_CONTEXT_MANAGEMENT_CONFIG.warmClearTriggerTokens),
|
|
285
|
-
clearAtLeastTokens: positiveInt(DEFAULT_CONTEXT_MANAGEMENT_CONFIG.clearAtLeastTokens),
|
|
286
|
-
maxModelCallsPerTurn: positiveInt(DEFAULT_CONTEXT_MANAGEMENT_CONFIG.maxModelCallsPerTurn)
|
|
287
|
-
});
|
|
288
|
-
const parseExcludeTools = (value) => {
|
|
289
|
-
if (!value) return DEFAULT_CONTEXT_MANAGEMENT_CONFIG.excludeTools;
|
|
290
|
-
return value.split(",").map((name) => name.trim()).filter((name) => name.length > 0);
|
|
291
|
-
};
|
|
292
|
-
/**
|
|
293
|
-
* Reads the context-management config from `env` (defaults to `process.env`).
|
|
294
|
-
* Never throws — invalid values fall back to defaults.
|
|
295
|
-
*/
|
|
296
|
-
function resolveContextManagementConfig(env = process.env) {
|
|
297
|
-
if (!boolFromEnv(env.SKYDIVE_CONTEXT_MANAGEMENT, false)) return { ...DEFAULT_CONTEXT_MANAGEMENT_CONFIG };
|
|
298
|
-
const parsed = configSchema.safeParse({
|
|
299
|
-
perResultMaxBytes: env.SKYDIVE_CTX_PER_RESULT_MAX_BYTES,
|
|
300
|
-
keepRecentToolResults: env.SKYDIVE_CTX_KEEP_RECENT,
|
|
301
|
-
coldCacheGapSeconds: env.SKYDIVE_CTX_COLD_GAP_SECONDS,
|
|
302
|
-
warmClearTriggerTokens: env.SKYDIVE_CTX_WARM_TRIGGER_TOKENS,
|
|
303
|
-
clearAtLeastTokens: env.SKYDIVE_CTX_CLEAR_AT_LEAST_TOKENS,
|
|
304
|
-
maxModelCallsPerTurn: env.SKYDIVE_CTX_MAX_MODEL_CALLS
|
|
305
|
-
});
|
|
306
|
-
if (!parsed.success) {
|
|
307
|
-
log$13.warn({
|
|
308
|
-
event: "context_management_config_invalid",
|
|
309
|
-
err: parsed.error
|
|
310
|
-
}, "falling back to default context-management config");
|
|
311
|
-
return {
|
|
312
|
-
...DEFAULT_CONTEXT_MANAGEMENT_CONFIG,
|
|
313
|
-
enabled: true
|
|
314
|
-
};
|
|
315
|
-
}
|
|
316
|
-
return {
|
|
317
|
-
enabled: true,
|
|
318
|
-
...parsed.data,
|
|
319
|
-
excludeTools: parseExcludeTools(env.SKYDIVE_CTX_EXCLUDE_TOOLS),
|
|
320
|
-
nativeAnthropicEdits: boolFromEnv(env.SKYDIVE_CTX_NATIVE_ANTHROPIC_EDITS, DEFAULT_CONTEXT_MANAGEMENT_CONFIG.nativeAnthropicEdits)
|
|
321
|
-
};
|
|
322
|
-
}
|
|
323
|
-
//#endregion
|
|
324
|
-
//#region src/extensions/context-management-runtime.ts
|
|
325
|
-
/**
|
|
326
|
-
* Live, mutable view of the context-management config.
|
|
327
|
-
*
|
|
328
|
-
* The env-derived config (context-management-config.ts) is the boot-time
|
|
329
|
-
* default. On top of it, the platform can deliver a global on/off at runtime
|
|
330
|
-
* via the LaunchDarkly flag `harness-context-management-enabled` — resolved
|
|
331
|
-
* server-side and polled by the harness (see the poller in
|
|
332
|
-
* context-management.ts). This holder is where that override lands so a flip
|
|
333
|
-
* (especially a kill-switch) reaches long-lived sandboxes without a restart.
|
|
334
|
-
*
|
|
335
|
-
* Only the master `enabled` toggle is overridable at runtime; the per-knob
|
|
336
|
-
* tunables stay env-derived. `enabled` resolves to the flag override when the
|
|
337
|
-
* platform has reported one, else the env value.
|
|
338
|
-
*/
|
|
339
|
-
let baseConfig = null;
|
|
340
|
-
let flagOverride = null;
|
|
341
|
-
function base() {
|
|
342
|
-
baseConfig ??= resolveContextManagementConfig();
|
|
343
|
-
return baseConfig;
|
|
344
|
-
}
|
|
345
|
-
/** The effective config, with the runtime flag override applied to `enabled`. */
|
|
346
|
-
function getContextManagementConfig() {
|
|
347
|
-
const resolved = base();
|
|
348
|
-
return {
|
|
349
|
-
...resolved,
|
|
350
|
-
enabled: flagOverride ?? resolved.enabled
|
|
351
|
-
};
|
|
352
|
-
}
|
|
353
|
-
/**
|
|
354
|
-
* Apply the platform-reported flag value. `null` clears the override (fall back
|
|
355
|
-
* to the env default) — used when the phone-home result is indeterminate so a
|
|
356
|
-
* transient failure never silently changes behavior.
|
|
357
|
-
*/
|
|
358
|
-
function setContextManagementFlagOverride(enabled) {
|
|
359
|
-
flagOverride = enabled;
|
|
360
|
-
}
|
|
361
|
-
/**
|
|
362
|
-
* Whether a phone-home channel exists to learn the flag at runtime. When false
|
|
363
|
-
* (e.g. a bare local CLI with no platform API), the env value is the only
|
|
364
|
-
* source and there's nothing to poll.
|
|
365
|
-
*/
|
|
366
|
-
function hasFlagSource(env = process.env) {
|
|
367
|
-
return Boolean(env.SKYDIVE_API_URL ?? env.ANYONE_API_URL);
|
|
368
|
-
}
|
|
369
|
-
//#endregion
|
|
370
|
-
//#region src/iteration-cap.ts
|
|
371
|
-
function installIterationCap({ session, log }, configOverride = null) {
|
|
372
|
-
const readConfig = () => configOverride ?? getContextManagementConfig();
|
|
373
|
-
if (!readConfig().enabled && (configOverride !== null || !hasFlagSource())) return;
|
|
374
|
-
const agent = session.agent;
|
|
375
|
-
if (typeof agent.createLoopConfig !== "function") throw new Error("installIterationCap: session.agent.createLoopConfig is missing — pi-agent-core internals changed; update iteration-cap.ts.");
|
|
376
|
-
const original = agent.createLoopConfig.bind(agent);
|
|
377
|
-
agent.createLoopConfig = (options) => {
|
|
378
|
-
const loopConfig = original(options);
|
|
379
|
-
const previousStop = loopConfig.shouldStopAfterTurn;
|
|
380
|
-
let modelCalls = 0;
|
|
381
|
-
return {
|
|
382
|
-
...loopConfig,
|
|
383
|
-
shouldStopAfterTurn: async (ctx) => {
|
|
384
|
-
if (previousStop && await previousStop(ctx)) return true;
|
|
385
|
-
const config = readConfig();
|
|
386
|
-
if (!config.enabled) return false;
|
|
387
|
-
modelCalls += 1;
|
|
388
|
-
if (modelCalls >= config.maxModelCallsPerTurn) {
|
|
389
|
-
log.warn({
|
|
390
|
-
event: "iteration_cap_reached",
|
|
391
|
-
modelCalls,
|
|
392
|
-
max: config.maxModelCallsPerTurn
|
|
393
|
-
}, "reached per-turn model-call cap; stopping turn gracefully");
|
|
394
|
-
return true;
|
|
395
|
-
}
|
|
396
|
-
return false;
|
|
397
|
-
}
|
|
398
|
-
};
|
|
399
|
-
};
|
|
400
|
-
}
|
|
401
|
-
//#endregion
|
|
402
246
|
//#region src/extensions/capability-soul-nudge.ts
|
|
403
247
|
/**
|
|
404
248
|
* One-line reminder appended to a tool-update continuation when the agent
|
|
@@ -415,88 +259,377 @@ function installIterationCap({ session, log }, configOverride = null) {
|
|
|
415
259
|
*/
|
|
416
260
|
const CAPABILITY_SOUL_NUDGE = "New capability gained — once the current task is done, if this changes what you can do for the user, record it in `soul.md` so it carries into future conversations rather than being rediscovered from scratch (then commit and push).";
|
|
417
261
|
//#endregion
|
|
418
|
-
//#region src/extensions/
|
|
262
|
+
//#region src/extensions/tool-call-summary.ts
|
|
263
|
+
const log$15 = logger.child({ module: "tool-call-summary-extension" });
|
|
419
264
|
/**
|
|
420
|
-
*
|
|
421
|
-
*
|
|
422
|
-
*
|
|
423
|
-
*
|
|
424
|
-
* The agent edits files under `tools/` between turns; on every `tool_result`
|
|
425
|
-
* the extension stat()s the directory's children, compares mtimes against
|
|
426
|
-
* what we last reconciled against, and queues a `pendingLocalToolsUpdate`
|
|
427
|
-
* if anything was added/removed/changed. The chat handler drains that queue
|
|
428
|
-
* after the current `session.prompt(...)` returns, calls `session.reload()`,
|
|
429
|
-
* and injects a synthetic continuation message — same dance as MCP.
|
|
430
|
-
*
|
|
431
|
-
* **ESM cache-busting.** `import(url)` in Node's ESM loader keys cached
|
|
432
|
-
* modules by URL. Re-importing the same path after editing the file gets
|
|
433
|
-
* the *original* module back. To force a fresh load on mtime change we
|
|
434
|
-
* append `?v=<mtimeMs>` to the URL — different URL, fresh module
|
|
435
|
-
* evaluation. The old version stays in memory but is unreachable.
|
|
436
|
-
*
|
|
437
|
-
* Each `.ts`/`.mjs`/`.js` file's default export should be a `ToolDefinition`
|
|
438
|
-
* or `ToolDefinition[]`. Files starting with `_` or `.` are skipped, so
|
|
439
|
-
* `tools/_example.ts` documents the shape without registering.
|
|
265
|
+
* The injected parameter name: a namespaced sentinel, so it can never collide
|
|
266
|
+
* with a real tool argument and is unmistakable in transcripts and logs. The
|
|
267
|
+
* frontend renderer (ANY-2723) duplicates this literal — keep the two in sync.
|
|
440
268
|
*/
|
|
441
|
-
const
|
|
442
|
-
|
|
443
|
-
const
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
269
|
+
const TOOL_CALL_SUMMARY_FIELD = "__skydive_summary__";
|
|
270
|
+
/** JSON Schema fragment for the injected parameter. */
|
|
271
|
+
const SUMMARY_PROPERTY = {
|
|
272
|
+
type: "string",
|
|
273
|
+
description: "Required for every tool call. A concise, specific summary (max ~8 words) of what THIS call does and why, written for a person watching the conversation, e.g. \"Searching feedback for billing complaints\" or \"Reading the auth middleware\". Address the user directly in second person: the summary is read by the user, so refer to their things as \"your\", never in third person — \"Reading your emails\", not \"Reading his emails\". Always use the present progressive tense, since it is shown while the call runs: \"Updating your Slack\", never \"Updated your Slack\". Make each summary distinct from your other tool calls; never reuse a generic label like \"Search query\" or \"Running command\"."
|
|
274
|
+
};
|
|
275
|
+
const jsonSchemaObjectSchema = z.object({
|
|
276
|
+
type: z.unknown().optional(),
|
|
277
|
+
properties: z.record(z.string(), z.unknown()).optional(),
|
|
278
|
+
required: z.array(z.string()).optional(),
|
|
279
|
+
additionalProperties: z.unknown().optional()
|
|
280
|
+
}).passthrough();
|
|
281
|
+
const toolEntrySchema = z.object({
|
|
282
|
+
name: z.string().optional(),
|
|
283
|
+
input_schema: jsonSchemaObjectSchema.optional(),
|
|
284
|
+
parameters: jsonSchemaObjectSchema.optional(),
|
|
285
|
+
function: z.object({
|
|
286
|
+
name: z.string().optional(),
|
|
287
|
+
parameters: jsonSchemaObjectSchema.optional()
|
|
288
|
+
}).passthrough().optional()
|
|
289
|
+
}).passthrough();
|
|
290
|
+
const payloadWithToolsSchema = z.object({ tools: z.array(z.unknown()) }).passthrough();
|
|
450
291
|
/**
|
|
451
|
-
*
|
|
452
|
-
*
|
|
453
|
-
*
|
|
454
|
-
* `
|
|
292
|
+
* Add the summary property to one JSON Schema object. Returns the augmented
|
|
293
|
+
* copy, or `null` when the tool should be left untouched: a strict schema
|
|
294
|
+
* (`additionalProperties: false`) whose validation would reject the extra
|
|
295
|
+
* field, or one that already declares a `__skydive_summary__` property of its own.
|
|
455
296
|
*/
|
|
456
|
-
function
|
|
457
|
-
|
|
458
|
-
}
|
|
459
|
-
|
|
460
|
-
const
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
return lines.join("\n");
|
|
471
|
-
}
|
|
472
|
-
function isToolDefinition(x) {
|
|
473
|
-
return !!x && typeof x === "object" && typeof x.name === "string" && typeof x.execute === "function";
|
|
297
|
+
function augmentSchema(schema) {
|
|
298
|
+
if (schema.additionalProperties === false) return null;
|
|
299
|
+
const properties = schema.properties ?? {};
|
|
300
|
+
if ("__skydive_summary__" in properties) return null;
|
|
301
|
+
const required = schema.required ?? [];
|
|
302
|
+
return {
|
|
303
|
+
...schema,
|
|
304
|
+
type: schema.type ?? "object",
|
|
305
|
+
properties: {
|
|
306
|
+
[TOOL_CALL_SUMMARY_FIELD]: SUMMARY_PROPERTY,
|
|
307
|
+
...properties
|
|
308
|
+
},
|
|
309
|
+
required: required.includes("__skydive_summary__") ? required : [...required, TOOL_CALL_SUMMARY_FIELD]
|
|
310
|
+
};
|
|
474
311
|
}
|
|
475
312
|
/**
|
|
476
|
-
*
|
|
477
|
-
*
|
|
478
|
-
*
|
|
479
|
-
* section, which biases the LLM against using it. If the local tool author
|
|
480
|
-
* didn't set a snippet, default to the tool's description so the tool stays
|
|
481
|
-
* visible.
|
|
313
|
+
* Augment a single tool entry, dispatching on which provider shape it is.
|
|
314
|
+
* Returns the (possibly rebuilt) entry and whether anything changed. Skipped
|
|
315
|
+
* tools — wrong shape, strict, or name in `strictToolNames` — return unchanged.
|
|
482
316
|
*/
|
|
483
|
-
function
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
...tool,
|
|
489
|
-
promptSnippet: fallback
|
|
317
|
+
function augmentToolEntry(entry, strictToolNames) {
|
|
318
|
+
const parsed = toolEntrySchema.safeParse(entry);
|
|
319
|
+
if (!parsed.success) return {
|
|
320
|
+
entry,
|
|
321
|
+
changed: false
|
|
490
322
|
};
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
}
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
323
|
+
const tool = parsed.data;
|
|
324
|
+
const name = tool.name ?? tool.function?.name ?? null;
|
|
325
|
+
if (name !== null && strictToolNames.has(name)) return {
|
|
326
|
+
entry,
|
|
327
|
+
changed: false
|
|
328
|
+
};
|
|
329
|
+
if (tool.input_schema) {
|
|
330
|
+
const augmented = augmentSchema(tool.input_schema);
|
|
331
|
+
if (!augmented) return {
|
|
332
|
+
entry,
|
|
333
|
+
changed: false
|
|
334
|
+
};
|
|
335
|
+
return {
|
|
336
|
+
entry: {
|
|
337
|
+
...tool,
|
|
338
|
+
input_schema: augmented
|
|
339
|
+
},
|
|
340
|
+
changed: true
|
|
341
|
+
};
|
|
342
|
+
}
|
|
343
|
+
if (tool.parameters) {
|
|
344
|
+
const augmented = augmentSchema(tool.parameters);
|
|
345
|
+
if (!augmented) return {
|
|
346
|
+
entry,
|
|
347
|
+
changed: false
|
|
348
|
+
};
|
|
349
|
+
return {
|
|
350
|
+
entry: {
|
|
351
|
+
...tool,
|
|
352
|
+
parameters: augmented
|
|
353
|
+
},
|
|
354
|
+
changed: true
|
|
355
|
+
};
|
|
356
|
+
}
|
|
357
|
+
if (tool.function?.parameters) {
|
|
358
|
+
const augmented = augmentSchema(tool.function.parameters);
|
|
359
|
+
if (!augmented) return {
|
|
360
|
+
entry,
|
|
361
|
+
changed: false
|
|
362
|
+
};
|
|
363
|
+
return {
|
|
364
|
+
entry: {
|
|
365
|
+
...tool,
|
|
366
|
+
function: {
|
|
367
|
+
...tool.function,
|
|
368
|
+
parameters: augmented
|
|
369
|
+
}
|
|
370
|
+
},
|
|
371
|
+
changed: true
|
|
372
|
+
};
|
|
373
|
+
}
|
|
374
|
+
return {
|
|
375
|
+
entry,
|
|
376
|
+
changed: false
|
|
377
|
+
};
|
|
378
|
+
}
|
|
379
|
+
/**
|
|
380
|
+
* Inject the summary field into every eligible tool in a provider payload.
|
|
381
|
+
* Returns a new payload when at least one tool was augmented, or `undefined`
|
|
382
|
+
* to signal "no change" (which keeps the original payload, per the
|
|
383
|
+
* `before_provider_request` contract).
|
|
384
|
+
*
|
|
385
|
+
* @param payload The outgoing provider payload (shape varies by provider).
|
|
386
|
+
* @param strictToolNames Names of tools whose registered schema is strict and
|
|
387
|
+
* must be skipped to avoid validation errors.
|
|
388
|
+
*/
|
|
389
|
+
function injectToolCallSummary(payload, strictToolNames) {
|
|
390
|
+
const parsed = payloadWithToolsSchema.safeParse(payload);
|
|
391
|
+
if (!parsed.success || parsed.data.tools.length === 0) return void 0;
|
|
392
|
+
let changed = false;
|
|
393
|
+
const tools = parsed.data.tools.map((entry) => {
|
|
394
|
+
const result = augmentToolEntry(entry, strictToolNames);
|
|
395
|
+
if (result.changed) changed = true;
|
|
396
|
+
return result.entry;
|
|
397
|
+
});
|
|
398
|
+
if (!changed) return void 0;
|
|
399
|
+
return {
|
|
400
|
+
...parsed.data,
|
|
401
|
+
tools
|
|
402
|
+
};
|
|
403
|
+
}
|
|
404
|
+
/**
|
|
405
|
+
* Names of registered tools whose schema sets `additionalProperties: false`.
|
|
406
|
+
* Pi validates the model's tool args against this registered schema, so the
|
|
407
|
+
* injected field would make a strict tool's call fail validation — skip them.
|
|
408
|
+
*/
|
|
409
|
+
function getStrictToolNames(pi) {
|
|
410
|
+
const names = /* @__PURE__ */ new Set();
|
|
411
|
+
for (const tool of pi.getAllTools()) {
|
|
412
|
+
const parsed = jsonSchemaObjectSchema.safeParse(tool.parameters);
|
|
413
|
+
if (parsed.success && parsed.data.additionalProperties === false) names.add(tool.name);
|
|
414
|
+
}
|
|
415
|
+
return names;
|
|
416
|
+
}
|
|
417
|
+
/**
|
|
418
|
+
* Whether a tool's own schema declares a `__skydive_summary__` property. We
|
|
419
|
+
* never inject into such a tool, so any value it carries is a real argument
|
|
420
|
+
* and must be left alone.
|
|
421
|
+
*/
|
|
422
|
+
function schemaDeclaresSummary(parameters) {
|
|
423
|
+
const parsed = jsonSchemaObjectSchema.safeParse(parameters);
|
|
424
|
+
return parsed.success && parsed.data.properties != null && "__skydive_summary__" in parsed.data.properties;
|
|
425
|
+
}
|
|
426
|
+
function toolDeclaresSummaryParam(pi, toolName) {
|
|
427
|
+
const tool = pi.getAllTools().find((candidate) => candidate.name === toolName);
|
|
428
|
+
if (!tool) return false;
|
|
429
|
+
return schemaDeclaresSummary(tool.parameters);
|
|
430
|
+
}
|
|
431
|
+
/** A copy of `args` without the injected summary. Never mutates its input. */
|
|
432
|
+
function withoutInjectedSummary(args) {
|
|
433
|
+
if (args == null || typeof args !== "object" || Array.isArray(args)) return args;
|
|
434
|
+
if (!("__skydive_summary__" in args)) return args;
|
|
435
|
+
const { [TOOL_CALL_SUMMARY_FIELD]: _summary, ...rest } = args;
|
|
436
|
+
return rest;
|
|
437
|
+
}
|
|
438
|
+
/**
|
|
439
|
+
* Register a tool so the injected summary is removed *before* pi validates
|
|
440
|
+
* the model's arguments against the tool's schema.
|
|
441
|
+
*
|
|
442
|
+
* Needed because the `tool_call` strip below runs too late. pi-agent-core's
|
|
443
|
+
* `prepareToolCall` goes: `tool.prepareArguments` → `validateToolArguments` →
|
|
444
|
+
* `beforeToolCall` (which is what dispatches `tool_call`). A tool whose schema
|
|
445
|
+
* sets `additionalProperties: false` therefore rejects the summary and errors
|
|
446
|
+
* the call — client-side, before any request reaches the server — while the
|
|
447
|
+
* strip meant to prevent exactly that sits one step further down.
|
|
448
|
+
*
|
|
449
|
+
* `augmentSchema` already skips strict tools, but that only controls what the
|
|
450
|
+
* model is *told*. It still emits the field on those tools, because every
|
|
451
|
+
* other tool in the list declares it as required for every call. So a server
|
|
452
|
+
* can connect, bind its tools, present a healthy inventory, and have every
|
|
453
|
+
* call fail — and it reads as the server's fault when it is ours.
|
|
454
|
+
*
|
|
455
|
+
* Apply to tools registered from schemas we do not author: MCP servers and
|
|
456
|
+
* local `tools/*.ts`. Tools built here with `Type.Object(...)` do not need it
|
|
457
|
+
* (TypeBox emits no `additionalProperties`, so the field validates fine and
|
|
458
|
+
* the `tool_call` strip removes it in time).
|
|
459
|
+
*
|
|
460
|
+
* Fails open, like the rest of this module: if the strip throws, the original
|
|
461
|
+
* arguments are used rather than failing the call.
|
|
462
|
+
*
|
|
463
|
+
* The type constraint names only the fields this wrapper reads, rather than
|
|
464
|
+
* `ToolDefinition` itself: a concretely-typed definition (e.g. the built-in
|
|
465
|
+
* edit tool's `ToolDefinition<typeof editSchema, EditToolDetails, ...>`) is
|
|
466
|
+
* not assignable to the default `ToolDefinition<TSchema, unknown, any>`
|
|
467
|
+
* because of the render hooks' parameter variance, and this wrapper never
|
|
468
|
+
* touches those hooks anyway.
|
|
469
|
+
*/
|
|
470
|
+
function withSummaryStrippedBeforeValidation(tool) {
|
|
471
|
+
if (schemaDeclaresSummary(tool.parameters)) return tool;
|
|
472
|
+
const toolPrepare = tool.prepareArguments;
|
|
473
|
+
const prepareArguments = ((args) => {
|
|
474
|
+
let stripped = args;
|
|
475
|
+
try {
|
|
476
|
+
stripped = withoutInjectedSummary(args);
|
|
477
|
+
} catch (err) {
|
|
478
|
+
log$15.error({
|
|
479
|
+
err,
|
|
480
|
+
event: "tool_call_summary_prepare_strip_failed",
|
|
481
|
+
toolName: tool.name
|
|
482
|
+
}, "tool_call_summary pre-validation strip failed; leaving arguments untouched");
|
|
483
|
+
}
|
|
484
|
+
return toolPrepare ? toolPrepare(stripped) : stripped;
|
|
485
|
+
});
|
|
486
|
+
return {
|
|
487
|
+
...tool,
|
|
488
|
+
prepareArguments
|
|
489
|
+
};
|
|
490
|
+
}
|
|
491
|
+
/**
|
|
492
|
+
* Remove the injected summary from a tool's execution input. No-op when the
|
|
493
|
+
* field is absent, or when the tool genuinely declares a `__skydive_summary__`
|
|
494
|
+
* parameter of its own (which we never inject into, so its value is real).
|
|
495
|
+
* Mutates `input` in place, matching the `tool_call` contract.
|
|
496
|
+
*
|
|
497
|
+
* Fails open: this runs on the critical path of tool execution, and the
|
|
498
|
+
* `getAllTools()` lookup can throw. On any error we leave `input` untouched
|
|
499
|
+
* (the sentinel may pass through to the tool, but a bug here can never break
|
|
500
|
+
* tool execution).
|
|
501
|
+
*/
|
|
502
|
+
function stripInjectedSummary(pi, toolName, input) {
|
|
503
|
+
try {
|
|
504
|
+
if (!("__skydive_summary__" in input)) return;
|
|
505
|
+
if (toolDeclaresSummaryParam(pi, toolName)) return;
|
|
506
|
+
delete input[TOOL_CALL_SUMMARY_FIELD];
|
|
507
|
+
} catch (err) {
|
|
508
|
+
log$15.error({
|
|
509
|
+
err,
|
|
510
|
+
event: "tool_call_summary_strip_failed",
|
|
511
|
+
toolName
|
|
512
|
+
}, "tool_call_summary strip failed; leaving tool input untouched");
|
|
513
|
+
}
|
|
514
|
+
}
|
|
515
|
+
/**
|
|
516
|
+
* Compute the rewritten payload for a `before_provider_request` event, failing
|
|
517
|
+
* open: on any error the original payload is left untouched so a bug here can
|
|
518
|
+
* never break an LLM call.
|
|
519
|
+
*/
|
|
520
|
+
function buildInjectedPayload(pi, payload) {
|
|
521
|
+
try {
|
|
522
|
+
return injectToolCallSummary(payload, getStrictToolNames(pi));
|
|
523
|
+
} catch (err) {
|
|
524
|
+
log$15.error({
|
|
525
|
+
err,
|
|
526
|
+
event: "tool_call_summary_injection_failed"
|
|
527
|
+
}, "tool_call_summary injection failed; passing payload through unchanged");
|
|
528
|
+
return;
|
|
529
|
+
}
|
|
530
|
+
}
|
|
531
|
+
const toolCallSummaryExtension = (pi) => {
|
|
532
|
+
pi.on("before_provider_request", (event) => buildInjectedPayload(pi, event.payload));
|
|
533
|
+
pi.on("tool_call", (event) => {
|
|
534
|
+
stripInjectedSummary(pi, event.toolName, event.input);
|
|
535
|
+
});
|
|
536
|
+
let editOverrideRegistered = false;
|
|
537
|
+
pi.on("session_start", (_event, ctx) => {
|
|
538
|
+
if (editOverrideRegistered) return;
|
|
539
|
+
try {
|
|
540
|
+
pi.registerTool(withSummaryStrippedBeforeValidation(createEditToolDefinition(ctx.cwd)));
|
|
541
|
+
editOverrideRegistered = true;
|
|
542
|
+
} catch (err) {
|
|
543
|
+
log$15.error({
|
|
544
|
+
err,
|
|
545
|
+
event: "tool_call_summary_edit_override_failed"
|
|
546
|
+
}, "failed to register summary-tolerant edit tool; built-in remains active");
|
|
547
|
+
}
|
|
548
|
+
});
|
|
549
|
+
};
|
|
550
|
+
//#endregion
|
|
551
|
+
//#region src/extensions/local-tools.ts
|
|
552
|
+
/**
|
|
553
|
+
* Local-tools adapter as a pi extension. Mirrors the mcp.ts hot-reload pattern
|
|
554
|
+
* one level over: walks `tools/` (relative to the harness cwd), imports each
|
|
555
|
+
* module, and registers the exported `ToolDefinition`s with pi.
|
|
556
|
+
*
|
|
557
|
+
* The agent edits files under `tools/` between turns; on every `tool_result`
|
|
558
|
+
* the extension stat()s the directory's children, compares mtimes against
|
|
559
|
+
* what we last reconciled against, and queues a `pendingLocalToolsUpdate`
|
|
560
|
+
* if anything was added/removed/changed. The chat handler drains that queue
|
|
561
|
+
* after the current `session.prompt(...)` returns, calls `session.reload()`,
|
|
562
|
+
* and injects a synthetic continuation message — same dance as MCP.
|
|
563
|
+
*
|
|
564
|
+
* **ESM cache-busting.** `import(url)` in Node's ESM loader keys cached
|
|
565
|
+
* modules by URL. Re-importing the same path after editing the file gets
|
|
566
|
+
* the *original* module back. To force a fresh load on mtime change we
|
|
567
|
+
* append `?mtime=<mtimeMs>` to the URL — different URL, fresh module
|
|
568
|
+
* evaluation. The old version stays in memory but is unreachable.
|
|
569
|
+
*
|
|
570
|
+
* Each `.ts`/`.mjs`/`.js` file's default export should be a `ToolDefinition`
|
|
571
|
+
* or `ToolDefinition[]`. Files starting with `_` or `.` are skipped, so
|
|
572
|
+
* `tools/_example.ts` documents the shape without registering.
|
|
573
|
+
*/
|
|
574
|
+
const log$14 = logger.child({ module: "local-tools-extension" });
|
|
575
|
+
const TOOLS_DIRNAME = "tools";
|
|
576
|
+
const fileState = /* @__PURE__ */ new Map();
|
|
577
|
+
let pendingLocalToolsUpdate = null;
|
|
578
|
+
function consumePendingLocalToolsUpdate() {
|
|
579
|
+
const update = pendingLocalToolsUpdate;
|
|
580
|
+
pendingLocalToolsUpdate = null;
|
|
581
|
+
return update;
|
|
582
|
+
}
|
|
583
|
+
/**
|
|
584
|
+
* Non-consuming peek used by the agent loop's `shouldStopAfterTurn` hook to
|
|
585
|
+
* decide whether to end the current `prompt()` after the in-flight turn so a
|
|
586
|
+
* fresh tool snapshot can be taken. The drain still happens in postPrompt via
|
|
587
|
+
* `consumePendingLocalToolsUpdate`.
|
|
588
|
+
*/
|
|
589
|
+
function hasPendingLocalToolsUpdate() {
|
|
590
|
+
return pendingLocalToolsUpdate !== null;
|
|
591
|
+
}
|
|
592
|
+
function formatLocalToolsUpdateMessage(summary) {
|
|
593
|
+
const lines = ["[system] Your local tools/ inventory changed during the previous turn. Your tool list is now updated; act on the new set rather than what was visible before."];
|
|
594
|
+
if (summary.added.length > 0) lines.push(`Newly loaded tool files: ${summary.added.join(", ")}`);
|
|
595
|
+
if (summary.refreshed.length > 0) lines.push(`Refreshed tool files: ${summary.refreshed.join(", ")}`);
|
|
596
|
+
if (summary.removed.length > 0) lines.push(`Removed tool files (and their tools): ${summary.removed.join(", ")}`);
|
|
597
|
+
if (summary.errors.length > 0) {
|
|
598
|
+
lines.push("Errors:");
|
|
599
|
+
for (const e of summary.errors) lines.push(` - ${e.file}: ${e.message}`);
|
|
600
|
+
}
|
|
601
|
+
if (summary.added.length > 0) lines.push(CAPABILITY_SOUL_NUDGE);
|
|
602
|
+
lines.push("Continue from where you left off, using the current tool list. Do not re-do work that already succeeded last turn.");
|
|
603
|
+
return lines.join("\n");
|
|
604
|
+
}
|
|
605
|
+
function isToolDefinition(x) {
|
|
606
|
+
return !!x && typeof x === "object" && typeof x.name === "string" && typeof x.execute === "function";
|
|
607
|
+
}
|
|
608
|
+
/**
|
|
609
|
+
* Pi's system-prompt builder filters its visible-tool list to entries with a
|
|
610
|
+
* non-empty `promptSnippet` (system-prompt.js:49) — a tool with only a
|
|
611
|
+
* `description` won't appear in the system prompt's "available tools"
|
|
612
|
+
* section, which biases the LLM against using it. If the local tool author
|
|
613
|
+
* didn't set a snippet, default to the tool's description so the tool stays
|
|
614
|
+
* visible.
|
|
615
|
+
*/
|
|
616
|
+
function withDefaultPromptSnippet(tool) {
|
|
617
|
+
if (typeof tool.promptSnippet === "string" && tool.promptSnippet.trim()) return tool;
|
|
618
|
+
const fallback = tool.description?.trim();
|
|
619
|
+
if (!fallback) return tool;
|
|
620
|
+
return {
|
|
621
|
+
...tool,
|
|
622
|
+
promptSnippet: fallback
|
|
623
|
+
};
|
|
624
|
+
}
|
|
625
|
+
async function listToolFiles(dir) {
|
|
626
|
+
let entries;
|
|
627
|
+
try {
|
|
628
|
+
entries = await readdir(dir);
|
|
629
|
+
} catch (err) {
|
|
630
|
+
if (err?.code === "ENOENT") return [];
|
|
631
|
+
throw err;
|
|
632
|
+
}
|
|
500
633
|
const out = [];
|
|
501
634
|
for (const file of entries) {
|
|
502
635
|
if (!file.endsWith(".ts") && !file.endsWith(".mjs") && !file.endsWith(".js")) continue;
|
|
@@ -514,7 +647,7 @@ async function listToolFiles(dir) {
|
|
|
514
647
|
return out;
|
|
515
648
|
}
|
|
516
649
|
async function importToolFile({ dir, file, mtimeMs }) {
|
|
517
|
-
const mod = await import(`${pathToFileURL(join(dir, file)).href}?
|
|
650
|
+
const mod = await import(`${pathToFileURL(join(dir, file)).href}?mtime=${mtimeMs}`);
|
|
518
651
|
const exported = mod.default ?? mod.tool ?? mod.tools;
|
|
519
652
|
const tools = [];
|
|
520
653
|
if (Array.isArray(exported)) {
|
|
@@ -565,7 +698,7 @@ async function reconcileLocalTools({ pi, dir }) {
|
|
|
565
698
|
action = existing ? "refreshed" : "added";
|
|
566
699
|
}
|
|
567
700
|
for (const tool of tools) {
|
|
568
|
-
pi.registerTool(withDefaultPromptSnippet(tool));
|
|
701
|
+
pi.registerTool(withSummaryStrippedBeforeValidation(withDefaultPromptSnippet(tool)));
|
|
569
702
|
summary.totalTools++;
|
|
570
703
|
}
|
|
571
704
|
if (action === "added") summary.added.push(file);
|
|
@@ -582,7 +715,7 @@ async function reconcileAndQueue({ pi, dir, reason }) {
|
|
|
582
715
|
dir
|
|
583
716
|
});
|
|
584
717
|
if (reason !== "session_start" && summaryHasChanges$1(summary)) pendingLocalToolsUpdate = summary;
|
|
585
|
-
log$
|
|
718
|
+
log$14.info({
|
|
586
719
|
event: "local_tools_reconcile",
|
|
587
720
|
reason,
|
|
588
721
|
total_tools: summary.totalTools,
|
|
@@ -604,7 +737,7 @@ const localToolsExtension = (pi) => {
|
|
|
604
737
|
reason: "session_start"
|
|
605
738
|
});
|
|
606
739
|
} catch (err) {
|
|
607
|
-
log$
|
|
740
|
+
log$14.error({
|
|
608
741
|
err,
|
|
609
742
|
event: "local_tools_reconcile_failed"
|
|
610
743
|
}, "local tools reconcile failed");
|
|
@@ -616,7 +749,7 @@ const localToolsExtension = (pi) => {
|
|
|
616
749
|
try {
|
|
617
750
|
current = await listToolFiles(dir);
|
|
618
751
|
} catch (err) {
|
|
619
|
-
log$
|
|
752
|
+
log$14.warn({
|
|
620
753
|
err,
|
|
621
754
|
event: "local_tools_listing_failed"
|
|
622
755
|
}, "tools/ listing failed");
|
|
@@ -638,7 +771,7 @@ const localToolsExtension = (pi) => {
|
|
|
638
771
|
reason: "auto_reload"
|
|
639
772
|
});
|
|
640
773
|
} catch (err) {
|
|
641
|
-
log$
|
|
774
|
+
log$14.error({
|
|
642
775
|
err,
|
|
643
776
|
event: "local_tools_auto_reload_failed"
|
|
644
777
|
}, "auto-reload after tools/ change failed");
|
|
@@ -682,8 +815,80 @@ function createStderrBuffer({ maxBytes }) {
|
|
|
682
815
|
* The transport stays alive — caller holds the client so the bridge's
|
|
683
816
|
* callback server keeps listening and a later reconcile re-probes.
|
|
684
817
|
*/
|
|
818
|
+
var SkipOutputSchemaValidation = class {
|
|
819
|
+
getValidator(_schema) {
|
|
820
|
+
return (input) => ({
|
|
821
|
+
valid: true,
|
|
822
|
+
data: input,
|
|
823
|
+
errorMessage: void 0
|
|
824
|
+
});
|
|
825
|
+
}
|
|
826
|
+
};
|
|
827
|
+
const skipOutputSchemaValidation = new SkipOutputSchemaValidation();
|
|
685
828
|
const DEFAULT_CONNECT_TIMEOUT_MS = 5e3;
|
|
686
829
|
const STDERR_BUFFER_BYTES = 4096;
|
|
830
|
+
const SESSION_NOT_FOUND_RETRY_DELAYS_MS = [
|
|
831
|
+
250,
|
|
832
|
+
500,
|
|
833
|
+
1e3,
|
|
834
|
+
2e3,
|
|
835
|
+
4e3,
|
|
836
|
+
8e3
|
|
837
|
+
];
|
|
838
|
+
function delay(ms) {
|
|
839
|
+
return new Promise((resolve) => setTimeout(resolve, ms));
|
|
840
|
+
}
|
|
841
|
+
async function fetchWithSessionNotFoundRetry(input, init) {
|
|
842
|
+
for (let attempt = 0;; attempt += 1) {
|
|
843
|
+
const response = await fetch(input, init);
|
|
844
|
+
const sessionId = new Headers(init?.headers).get("mcp-session-id");
|
|
845
|
+
const retryDelay = SESSION_NOT_FOUND_RETRY_DELAYS_MS[attempt];
|
|
846
|
+
if (!sessionId || response.status !== 404 || retryDelay === void 0 || !(await response.clone().text()).includes("session_not_found")) return response;
|
|
847
|
+
await response.body?.cancel();
|
|
848
|
+
await delay(retryDelay);
|
|
849
|
+
}
|
|
850
|
+
}
|
|
851
|
+
/**
|
|
852
|
+
* Sandbox egress variables a stdio MCP server needs to make outbound HTTPS
|
|
853
|
+
* calls. The SDK's default child environment is a tiny whitelist (HOME,
|
|
854
|
+
* LOGNAME, PATH, SHELL, TERM, USER on Linux), which drops NODE_EXTRA_CA_CERTS
|
|
855
|
+
* — but the sandbox's egress proxy terminates TLS with a private CA, so a
|
|
856
|
+
* child spawned without it fails every HTTPS request with a bare
|
|
857
|
+
* `fetch failed` (cause: UNABLE_TO_VERIFY_LEAF_SIGNATURE) that looks like the
|
|
858
|
+
* remote service is down. Same story for SSL_CERT_FILE (OpenSSL-based
|
|
859
|
+
* runtimes) and the *_PROXY set. Inherit them from this process so a stdio
|
|
860
|
+
* server gets working egress like every other process in the sandbox.
|
|
861
|
+
*/
|
|
862
|
+
const INHERITED_EGRESS_ENV_VARS = [
|
|
863
|
+
"NODE_EXTRA_CA_CERTS",
|
|
864
|
+
"SSL_CERT_FILE",
|
|
865
|
+
"SSL_CERT_DIR",
|
|
866
|
+
"REQUESTS_CA_BUNDLE",
|
|
867
|
+
"CURL_CA_BUNDLE",
|
|
868
|
+
"HTTP_PROXY",
|
|
869
|
+
"HTTPS_PROXY",
|
|
870
|
+
"NO_PROXY",
|
|
871
|
+
"http_proxy",
|
|
872
|
+
"https_proxy",
|
|
873
|
+
"no_proxy"
|
|
874
|
+
];
|
|
875
|
+
/**
|
|
876
|
+
* Environment for a stdio MCP server child: the SDK's safe defaults, plus the
|
|
877
|
+
* sandbox's egress/TLS variables, plus (last, so it wins) the server's own
|
|
878
|
+
* configured `env`. Building the merge here — instead of only when `env` is
|
|
879
|
+
* unset — also fixes the workaround trap where supplying any `env` in
|
|
880
|
+
* mcp.config.json silently replaced the ENTIRE default set, so a config that
|
|
881
|
+
* added one API key lost PATH/HOME and the server failed to spawn at all.
|
|
882
|
+
*/
|
|
883
|
+
function buildStdioEnv(configured, processEnv = process.env) {
|
|
884
|
+
const env = { ...getDefaultEnvironment() };
|
|
885
|
+
for (const key of INHERITED_EGRESS_ENV_VARS) {
|
|
886
|
+
const value = processEnv[key];
|
|
887
|
+
if (value !== void 0) env[key] = value;
|
|
888
|
+
}
|
|
889
|
+
if (configured) Object.assign(env, configured);
|
|
890
|
+
return env;
|
|
891
|
+
}
|
|
687
892
|
/**
|
|
688
893
|
* undici's fetch throws `TypeError: fetch failed` with the actual
|
|
689
894
|
* network error hung off `.cause` (e.g. `getaddrinfo ENOTFOUND ...`,
|
|
@@ -692,6 +897,19 @@ const STDERR_BUFFER_BYTES = 4096;
|
|
|
692
897
|
* indistinguishable from any other transport problem. Walk the cause
|
|
693
898
|
* chain so the agent sees the real underlying error.
|
|
694
899
|
*/
|
|
900
|
+
/**
|
|
901
|
+
* True when an error from an http MCP transport (connect, listTools, or a tool
|
|
902
|
+
* call) is an authentication failure. With no authProvider configured the SDK
|
|
903
|
+
* surfaces a 401 as `StreamableHTTPError(401)`; older paths translate it to
|
|
904
|
+
* `UnauthorizedError`. A dead/expired OAuth token (the proxy can no longer
|
|
905
|
+
* mint one) shows up here on the NEXT request against a previously-connected
|
|
906
|
+
* client — not just at connect — so reconcile must re-classify such a failure
|
|
907
|
+
* as `pending_auth` instead of a generic `failed`, keeping the "waiting on
|
|
908
|
+
* auth" report consistent with `platform auth`.
|
|
909
|
+
*/
|
|
910
|
+
function isUnauthorizedError(err) {
|
|
911
|
+
return err instanceof UnauthorizedError || err instanceof StreamableHTTPError && err.code === 401;
|
|
912
|
+
}
|
|
695
913
|
function formatError(err) {
|
|
696
914
|
if (!(err instanceof Error)) return String(err);
|
|
697
915
|
const parts = [err.message];
|
|
@@ -703,8 +921,20 @@ function formatError(err) {
|
|
|
703
921
|
}
|
|
704
922
|
return parts.join(": ");
|
|
705
923
|
}
|
|
924
|
+
function probeAlive(pid) {
|
|
925
|
+
if (pid === null) return null;
|
|
926
|
+
try {
|
|
927
|
+
process.kill(pid, 0);
|
|
928
|
+
return true;
|
|
929
|
+
} catch {
|
|
930
|
+
return false;
|
|
931
|
+
}
|
|
932
|
+
}
|
|
706
933
|
async function connectHttp(_id, config, client) {
|
|
707
|
-
const transport = new StreamableHTTPClientTransport(new URL(config.url), {
|
|
934
|
+
const transport = new StreamableHTTPClientTransport(new URL(config.url), {
|
|
935
|
+
...config.headers !== null ? { requestInit: { headers: config.headers } } : {},
|
|
936
|
+
fetch: fetchWithSessionNotFoundRetry
|
|
937
|
+
});
|
|
708
938
|
try {
|
|
709
939
|
await client.connect(transport);
|
|
710
940
|
return {
|
|
@@ -713,12 +943,12 @@ async function connectHttp(_id, config, client) {
|
|
|
713
943
|
stderr: null
|
|
714
944
|
};
|
|
715
945
|
} catch (err) {
|
|
716
|
-
if (err
|
|
946
|
+
if (isUnauthorizedError(err)) return {
|
|
717
947
|
status: "pending_auth",
|
|
718
948
|
client,
|
|
719
949
|
stderr: "",
|
|
720
950
|
stderrBuffer: null,
|
|
721
|
-
cliHint: `platform auth mcp ${config.url}`
|
|
951
|
+
cliHint: `platform auth mcp ${config.url} --service "<Product>"`
|
|
722
952
|
};
|
|
723
953
|
return {
|
|
724
954
|
status: "failed",
|
|
@@ -732,14 +962,17 @@ async function connectClient(id, config, opts = {}) {
|
|
|
732
962
|
const client = new Client({
|
|
733
963
|
name: "skydive-harness",
|
|
734
964
|
version: "0.1.0"
|
|
735
|
-
}, {
|
|
965
|
+
}, {
|
|
966
|
+
capabilities: {},
|
|
967
|
+
jsonSchemaValidator: skipOutputSchemaValidation
|
|
968
|
+
});
|
|
736
969
|
if (config.transport === "http") return connectHttp(id, config, client);
|
|
737
970
|
const params = {
|
|
738
971
|
command: config.command,
|
|
739
972
|
args: config.args,
|
|
740
|
-
stderr: "pipe"
|
|
973
|
+
stderr: "pipe",
|
|
974
|
+
env: buildStdioEnv(config.env)
|
|
741
975
|
};
|
|
742
|
-
if (config.env !== null) params.env = config.env;
|
|
743
976
|
if (config.cwd !== null) params.cwd = config.cwd;
|
|
744
977
|
const transport = new StdioClientTransport(params);
|
|
745
978
|
const stderrBuffer = createStderrBuffer({ maxBytes: STDERR_BUFFER_BYTES });
|
|
@@ -751,6 +984,10 @@ async function connectClient(id, config, opts = {}) {
|
|
|
751
984
|
resolve();
|
|
752
985
|
};
|
|
753
986
|
});
|
|
987
|
+
let stdoutMessages = 0;
|
|
988
|
+
transport.onmessage = () => {
|
|
989
|
+
if (stdoutMessages < 1e4) stdoutMessages += 1;
|
|
990
|
+
};
|
|
754
991
|
const connectPromise = client.connect(transport);
|
|
755
992
|
const TIMEOUT_SENTINEL = Symbol("timeout");
|
|
756
993
|
const result = await Promise.race([
|
|
@@ -768,12 +1005,20 @@ async function connectClient(id, config, opts = {}) {
|
|
|
768
1005
|
error: `server "${id}" exited before completing initialize`,
|
|
769
1006
|
stderr: stderrBuffer.read()
|
|
770
1007
|
};
|
|
771
|
-
if (result === TIMEOUT_SENTINEL)
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
|
|
775
|
-
|
|
776
|
-
|
|
1008
|
+
if (result === TIMEOUT_SENTINEL) {
|
|
1009
|
+
const pid = transport.pid;
|
|
1010
|
+
return {
|
|
1011
|
+
status: "timeout",
|
|
1012
|
+
client,
|
|
1013
|
+
stderr: stderrBuffer.read(),
|
|
1014
|
+
stderrBuffer,
|
|
1015
|
+
diagnostics: {
|
|
1016
|
+
pid,
|
|
1017
|
+
alive: probeAlive(pid),
|
|
1018
|
+
stdoutMessages
|
|
1019
|
+
}
|
|
1020
|
+
};
|
|
1021
|
+
}
|
|
777
1022
|
try {
|
|
778
1023
|
await client.close();
|
|
779
1024
|
} catch {}
|
|
@@ -784,6 +1029,126 @@ async function connectClient(id, config, opts = {}) {
|
|
|
784
1029
|
};
|
|
785
1030
|
}
|
|
786
1031
|
//#endregion
|
|
1032
|
+
//#region src/extensions/mcp/limits.ts
|
|
1033
|
+
/**
|
|
1034
|
+
* Tool-budget guardrails for MCP registration.
|
|
1035
|
+
*
|
|
1036
|
+
* Why this exists: the agent loop sends the *full* tool list — every tool's
|
|
1037
|
+
* name, description, and JSON-schema `parameters` — to the model on every
|
|
1038
|
+
* prompt. MCP servers add tools without bound: a handful of chatty servers (or
|
|
1039
|
+
* one server that exposes 100+ tools, or a few tools with enormous schemas) can
|
|
1040
|
+
* push the registered set past what the model can accept, and the request is
|
|
1041
|
+
* rejected before the turn even runs. Because `reconcile` re-registers from
|
|
1042
|
+
* `mcp.config.json` on *every* turn, an over-limit config bricks the harness on
|
|
1043
|
+
* a loop — the agent can't get a turn to run in order to edit the config back
|
|
1044
|
+
* down. Worse, the person can't tell *why*: tools just stop working.
|
|
1045
|
+
*
|
|
1046
|
+
* What actually overflows the request is *tokens*, not tool count — a few tools
|
|
1047
|
+
* with deeply-nested schemas and long descriptions cost more than a hundred
|
|
1048
|
+
* trivial ones. And how many tokens are safe depends on the *model*: a 200K
|
|
1049
|
+
* context window can afford far more tool surface than a 32K one. So the primary
|
|
1050
|
+
* limiter is a **token budget derived from the active model's context window**,
|
|
1051
|
+
* with a fixed tool-count cap as a coarse secondary guard (and the fallback
|
|
1052
|
+
* when the model — hence its window — isn't known at reconcile time).
|
|
1053
|
+
*
|
|
1054
|
+
* The fix is to make registration bounded and fail-soft. We register in
|
|
1055
|
+
* deterministic config order and stop before we blow the budget, recording how
|
|
1056
|
+
* much we dropped so the agent is *told* it hit the limit and which servers
|
|
1057
|
+
* were truncated. A config that would have bricked the harness now degrades to
|
|
1058
|
+
* "a bounded set of tools plus a loud warning", which the agent can act on by
|
|
1059
|
+
* pruning servers.
|
|
1060
|
+
*
|
|
1061
|
+
* Everything is env-overridable so the ceilings can be tuned per deployment
|
|
1062
|
+
* without a release, but ships with conservative defaults. A cap value of 0 (or
|
|
1063
|
+
* a non-finite / negative override) disables that cap — an explicit escape
|
|
1064
|
+
* hatch, not the default.
|
|
1065
|
+
*/
|
|
1066
|
+
const env = process.env;
|
|
1067
|
+
/**
|
|
1068
|
+
* Fraction of the model's context window we're willing to spend on MCP tool
|
|
1069
|
+
* schemas. Tool definitions are sent on every prompt, so they permanently eat
|
|
1070
|
+
* into the window available for the conversation, but a generous tool budget is
|
|
1071
|
+
* worth more than a marginally larger conversation window given how the harness
|
|
1072
|
+
* is used. 0.35 of a 200K window is ~70K tokens of tool schema, comfortably
|
|
1073
|
+
* more than any sane MCP setup; of a 32K window it's ~11.2K, which still forces
|
|
1074
|
+
* truncation before a small model chokes.
|
|
1075
|
+
*/
|
|
1076
|
+
const DEFAULT_MCP_TOOL_TOKEN_BUDGET_FRACTION = .35;
|
|
1077
|
+
/**
|
|
1078
|
+
* Floor for the token budget when the model's context window is unknown at
|
|
1079
|
+
* reconcile time (e.g. the model hasn't been resolved yet). Generous enough not
|
|
1080
|
+
* to truncate an ordinary tool set, low enough to still catch a runaway.
|
|
1081
|
+
*/
|
|
1082
|
+
const DEFAULT_MCP_TOOL_TOKEN_BUDGET_FLOOR = 16e3;
|
|
1083
|
+
/**
|
|
1084
|
+
* Read a positive-integer cap from an env var, falling back to `fallback`.
|
|
1085
|
+
* A `0` override (or any non-finite / negative value) means "no cap" and is
|
|
1086
|
+
* returned as `Infinity`, so callers can compare against it directly.
|
|
1087
|
+
*/
|
|
1088
|
+
function readCap(raw, fallback) {
|
|
1089
|
+
if (raw === void 0 || raw.trim() === "") return fallback;
|
|
1090
|
+
const parsed = Number(raw);
|
|
1091
|
+
if (!Number.isFinite(parsed) || parsed < 0) return Number.POSITIVE_INFINITY;
|
|
1092
|
+
if (parsed === 0) return Number.POSITIVE_INFINITY;
|
|
1093
|
+
return Math.floor(parsed);
|
|
1094
|
+
}
|
|
1095
|
+
/** Read a fraction in (0, 1] from an env var, falling back to `fallback`. */
|
|
1096
|
+
function readFraction(raw, fallback) {
|
|
1097
|
+
if (raw === void 0 || raw.trim() === "") return fallback;
|
|
1098
|
+
const parsed = Number(raw);
|
|
1099
|
+
if (!Number.isFinite(parsed) || parsed <= 0 || parsed > 1) return fallback;
|
|
1100
|
+
return parsed;
|
|
1101
|
+
}
|
|
1102
|
+
/** Read a non-negative integer from an env var, falling back to `fallback`. */
|
|
1103
|
+
function readNonNegativeInt(raw, fallback) {
|
|
1104
|
+
if (raw === void 0 || raw.trim() === "") return fallback;
|
|
1105
|
+
const parsed = Number(raw);
|
|
1106
|
+
if (!Number.isFinite(parsed) || parsed < 0) return fallback;
|
|
1107
|
+
return Math.floor(parsed);
|
|
1108
|
+
}
|
|
1109
|
+
/**
|
|
1110
|
+
* Resolve the active caps from the environment. Read once per reconcile so a
|
|
1111
|
+
* deployment can retune without a restart, cheap enough not to cache.
|
|
1112
|
+
*/
|
|
1113
|
+
function resolveMcpToolLimits() {
|
|
1114
|
+
return {
|
|
1115
|
+
maxTotalTools: readCap(env["SKYDIVE_MCP_MAX_TOTAL_TOOLS"], 128),
|
|
1116
|
+
maxToolsPerServer: readCap(env["SKYDIVE_MCP_MAX_TOOLS_PER_SERVER"], 50),
|
|
1117
|
+
tokenBudgetFraction: readFraction(env["SKYDIVE_MCP_TOOL_TOKEN_BUDGET_FRACTION"], DEFAULT_MCP_TOOL_TOKEN_BUDGET_FRACTION),
|
|
1118
|
+
tokenBudgetFloor: readNonNegativeInt(env["SKYDIVE_MCP_TOOL_TOKEN_BUDGET_FLOOR"], DEFAULT_MCP_TOOL_TOKEN_BUDGET_FLOOR)
|
|
1119
|
+
};
|
|
1120
|
+
}
|
|
1121
|
+
/**
|
|
1122
|
+
* The token budget for MCP tool schemas, given the active model's context
|
|
1123
|
+
* window (or undefined when the model isn't known yet).
|
|
1124
|
+
*
|
|
1125
|
+
* With a known window we spend `tokenBudgetFraction` of it, but never less than
|
|
1126
|
+
* the floor — a tiny window shouldn't collapse the budget to near-zero and
|
|
1127
|
+
* strand every tool. With no window we fall back to the floor outright.
|
|
1128
|
+
*/
|
|
1129
|
+
function resolveTokenBudget(limits, contextWindow) {
|
|
1130
|
+
if (contextWindow === void 0 || !Number.isFinite(contextWindow)) return limits.tokenBudgetFloor;
|
|
1131
|
+
return Math.max(limits.tokenBudgetFloor, Math.floor(contextWindow * limits.tokenBudgetFraction));
|
|
1132
|
+
}
|
|
1133
|
+
/**
|
|
1134
|
+
* Estimate the tokens an MCP tool's *definition* costs in the request. We
|
|
1135
|
+
* serialize what's actually sent to the model — the tool name, its description,
|
|
1136
|
+
* and its JSON-schema parameters — and apply pi's own ~chars/4 heuristic
|
|
1137
|
+
* (`estimateTokens` in the coding agent uses the same convention; there is no
|
|
1138
|
+
* per-provider tokenizer to lean on, and the provider's real usage is only
|
|
1139
|
+
* known after the response). Rough by design, but it tracks the true cost far
|
|
1140
|
+
* better than a flat per-tool count: a fat schema is charged for its fatness.
|
|
1141
|
+
*/
|
|
1142
|
+
function estimateToolTokens(tool) {
|
|
1143
|
+
let chars = tool.name.length + (tool.description?.length ?? 0);
|
|
1144
|
+
if (tool.inputSchema !== void 0 && tool.inputSchema !== null) try {
|
|
1145
|
+
chars += JSON.stringify(tool.inputSchema).length;
|
|
1146
|
+
} catch {
|
|
1147
|
+
chars += 256;
|
|
1148
|
+
}
|
|
1149
|
+
return Math.ceil(chars / 4);
|
|
1150
|
+
}
|
|
1151
|
+
//#endregion
|
|
787
1152
|
//#region src/extensions/mcp/mcp-config.ts
|
|
788
1153
|
/**
|
|
789
1154
|
* mcp.config.json schema + loader. Split out from the extension so it has
|
|
@@ -863,7 +1228,7 @@ async function loadMcpConfig(path) {
|
|
|
863
1228
|
* Clients are keyed by JSON-stringified config and reused across
|
|
864
1229
|
* reloads — only changed configs reconnect.
|
|
865
1230
|
*/
|
|
866
|
-
const log$
|
|
1231
|
+
const log$13 = logger.child({ module: "mcp-extension" });
|
|
867
1232
|
async function closeConnected(connected) {
|
|
868
1233
|
try {
|
|
869
1234
|
await connected.client.close();
|
|
@@ -933,6 +1298,57 @@ function mcpResultToPiContent(result) {
|
|
|
933
1298
|
text: "MCP tool returned no content."
|
|
934
1299
|
}];
|
|
935
1300
|
}
|
|
1301
|
+
const MAX_ENUM_VALUES_LISTED = 12;
|
|
1302
|
+
function formatEnumValues(values) {
|
|
1303
|
+
const shown = values.slice(0, MAX_ENUM_VALUES_LISTED).map((v) => JSON.stringify(v));
|
|
1304
|
+
if (values.length > MAX_ENUM_VALUES_LISTED) shown.push(`… (${values.length} total)`);
|
|
1305
|
+
return shown.join(", ");
|
|
1306
|
+
}
|
|
1307
|
+
function extractEnum(propSchema) {
|
|
1308
|
+
if (Array.isArray(propSchema.enum) && propSchema.enum.length > 0) return propSchema.enum;
|
|
1309
|
+
const items = propSchema.items;
|
|
1310
|
+
if (items !== null && typeof items === "object" && Array.isArray(items.enum) && items.enum.length > 0) return items.enum;
|
|
1311
|
+
return null;
|
|
1312
|
+
}
|
|
1313
|
+
/**
|
|
1314
|
+
* Build a compact, human-readable constraint hint from an MCP tool's raw JSON
|
|
1315
|
+
* schema so the model sees required fields and each param's allowed enum values
|
|
1316
|
+
* inline in the tool description — the schema pass-through hands pi the server's
|
|
1317
|
+
* schema verbatim, but the model attends to the prose far more than the raw
|
|
1318
|
+
* `enum`/`required` keys. Returns '' when there is nothing worth surfacing.
|
|
1319
|
+
*/
|
|
1320
|
+
function describeToolConstraints(inputSchema) {
|
|
1321
|
+
if (inputSchema === null || typeof inputSchema !== "object") return "";
|
|
1322
|
+
const schema = inputSchema;
|
|
1323
|
+
const required = Array.isArray(schema.required) ? schema.required.filter((r) => typeof r === "string") : [];
|
|
1324
|
+
const properties = schema.properties !== null && typeof schema.properties === "object" ? schema.properties : {};
|
|
1325
|
+
const enumLines = [];
|
|
1326
|
+
for (const [propName, rawProp] of Object.entries(properties)) {
|
|
1327
|
+
if (rawProp === null || typeof rawProp !== "object") continue;
|
|
1328
|
+
const values = extractEnum(rawProp);
|
|
1329
|
+
if (values) enumLines.push(`- \`${propName}\` must be one of: ${formatEnumValues(values)}`);
|
|
1330
|
+
}
|
|
1331
|
+
if (required.length === 0 && enumLines.length === 0) return "";
|
|
1332
|
+
const lines = ["Argument constraints:"];
|
|
1333
|
+
if (required.length > 0) lines.push(`- required: ${required.map((r) => `\`${r}\``).join(", ")}`);
|
|
1334
|
+
lines.push(...enumLines);
|
|
1335
|
+
return lines.join("\n");
|
|
1336
|
+
}
|
|
1337
|
+
/**
|
|
1338
|
+
* Build the `promptSnippet` a registered MCP tool exposes to the model: the
|
|
1339
|
+
* server's description with the argument-constraint hint appended, falling back
|
|
1340
|
+
* to a labelled placeholder when the server omitted a description AND the schema
|
|
1341
|
+
* carries no constraints (pi's system-prompt builder hides tools whose snippet
|
|
1342
|
+
* is empty — system-prompt.js:49). This is the layer where `describeToolConstraints`
|
|
1343
|
+
* actually reaches the model, so it is exported and tested directly: the pure
|
|
1344
|
+
* hint being correct is necessary but not sufficient; what the model sees is
|
|
1345
|
+
* this composed snippet.
|
|
1346
|
+
*/
|
|
1347
|
+
function buildToolPromptSnippet({ description, inputSchema, serverId }) {
|
|
1348
|
+
const constraintHint = describeToolConstraints(inputSchema);
|
|
1349
|
+
const enriched = constraintHint.length > 0 ? `${description}${description.length > 0 ? "\n\n" : ""}${constraintHint}` : description;
|
|
1350
|
+
return enriched.length > 0 ? enriched : `MCP tool from server "${serverId}".`;
|
|
1351
|
+
}
|
|
936
1352
|
var McpExtension = class {
|
|
937
1353
|
mcpClients = /* @__PURE__ */ new Map();
|
|
938
1354
|
registeredMcpToolNames = /* @__PURE__ */ new Set();
|
|
@@ -952,8 +1368,12 @@ var McpExtension = class {
|
|
|
952
1368
|
const name = makeToolName(serverId, tool.name);
|
|
953
1369
|
const parameters = Type.Unsafe(tool.inputSchema);
|
|
954
1370
|
const description = tool.description?.trim() ?? "";
|
|
955
|
-
const promptSnippet =
|
|
956
|
-
|
|
1371
|
+
const promptSnippet = buildToolPromptSnippet({
|
|
1372
|
+
description,
|
|
1373
|
+
inputSchema: tool.inputSchema,
|
|
1374
|
+
serverId
|
|
1375
|
+
});
|
|
1376
|
+
pi.registerTool(withSummaryStrippedBeforeValidation({
|
|
957
1377
|
name,
|
|
958
1378
|
label: `MCP: ${serverId}/${tool.name}`,
|
|
959
1379
|
description,
|
|
@@ -987,22 +1407,29 @@ var McpExtension = class {
|
|
|
987
1407
|
};
|
|
988
1408
|
}
|
|
989
1409
|
}
|
|
990
|
-
});
|
|
1410
|
+
}));
|
|
991
1411
|
this.registeredMcpToolNames.add(name);
|
|
992
1412
|
}
|
|
993
|
-
async reconcile({ pi, configPath, connectTimeoutMs }) {
|
|
1413
|
+
async reconcile({ pi, configPath, connectTimeoutMs, contextWindow }) {
|
|
994
1414
|
let config;
|
|
995
1415
|
try {
|
|
996
1416
|
config = await loadMcpConfig(configPath);
|
|
997
1417
|
} catch (err) {
|
|
998
1418
|
throw new Error(`Failed to load MCP config: ${err instanceof Error ? err.message : String(err)}`);
|
|
999
1419
|
}
|
|
1420
|
+
const limits = resolveMcpToolLimits();
|
|
1421
|
+
const tokenBudget = resolveTokenBudget(limits, contextWindow);
|
|
1000
1422
|
const summary = {
|
|
1001
1423
|
added: [],
|
|
1002
1424
|
removed: [],
|
|
1003
1425
|
refreshed: [],
|
|
1004
1426
|
errors: [],
|
|
1005
1427
|
totalTools: 0,
|
|
1428
|
+
droppedTools: 0,
|
|
1429
|
+
limits,
|
|
1430
|
+
tokenBudget,
|
|
1431
|
+
tokensUsed: 0,
|
|
1432
|
+
toolCounts: {},
|
|
1006
1433
|
servers: {}
|
|
1007
1434
|
};
|
|
1008
1435
|
const desiredIds = new Set(Object.keys(config.servers));
|
|
@@ -1026,15 +1453,58 @@ var McpExtension = class {
|
|
|
1026
1453
|
if (outcome.error) summary.errors.push(outcome.error);
|
|
1027
1454
|
if (outcome.change === "added") summary.added.push(id);
|
|
1028
1455
|
else if (outcome.change === "refreshed") summary.refreshed.push(id);
|
|
1029
|
-
|
|
1030
|
-
|
|
1031
|
-
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
|
|
1035
|
-
|
|
1036
|
-
|
|
1456
|
+
}
|
|
1457
|
+
const alloc = outcomes.filter((o) => Boolean(o.tools)).map((o) => {
|
|
1458
|
+
const advertised = o.tools.list.length;
|
|
1459
|
+
return {
|
|
1460
|
+
id: o.id,
|
|
1461
|
+
client: o.tools.client,
|
|
1462
|
+
list: o.tools.list,
|
|
1463
|
+
advertised,
|
|
1464
|
+
room: Math.min(advertised, limits.maxToolsPerServer),
|
|
1465
|
+
registered: 0,
|
|
1466
|
+
serverTokens: 0,
|
|
1467
|
+
droppedReason: null
|
|
1468
|
+
};
|
|
1469
|
+
});
|
|
1470
|
+
const maxAdvertised = alloc.reduce((m, s) => Math.max(m, s.advertised), 0);
|
|
1471
|
+
outer: for (let i = 0; i < maxAdvertised; i++) for (const s of alloc) {
|
|
1472
|
+
if (summary.totalTools >= limits.maxTotalTools) {
|
|
1473
|
+
for (const t of alloc) if (t.registered < t.advertised && t.droppedReason === null) t.droppedReason = "total";
|
|
1474
|
+
break outer;
|
|
1037
1475
|
}
|
|
1476
|
+
if (i >= s.room) {
|
|
1477
|
+
if (s.registered < s.advertised && s.droppedReason === null) s.droppedReason = "per_server";
|
|
1478
|
+
continue;
|
|
1479
|
+
}
|
|
1480
|
+
const tool = s.list[i];
|
|
1481
|
+
if (tool === void 0) continue;
|
|
1482
|
+
const cost = estimateToolTokens(tool);
|
|
1483
|
+
if (summary.tokensUsed + cost > tokenBudget && summary.totalTools > 0) {
|
|
1484
|
+
for (const t of alloc) if (t.registered < t.advertised && t.droppedReason === null) t.droppedReason = "tokens";
|
|
1485
|
+
break outer;
|
|
1486
|
+
}
|
|
1487
|
+
this.registerMcpTool({
|
|
1488
|
+
pi,
|
|
1489
|
+
serverId: s.id,
|
|
1490
|
+
client: s.client,
|
|
1491
|
+
tool
|
|
1492
|
+
});
|
|
1493
|
+
s.registered++;
|
|
1494
|
+
s.serverTokens += cost;
|
|
1495
|
+
summary.totalTools++;
|
|
1496
|
+
summary.tokensUsed += cost;
|
|
1497
|
+
}
|
|
1498
|
+
for (const s of alloc) {
|
|
1499
|
+
const dropped = s.advertised - s.registered;
|
|
1500
|
+
const droppedReason = dropped > 0 ? s.droppedReason : null;
|
|
1501
|
+
if (dropped > 0) summary.droppedTools += dropped;
|
|
1502
|
+
summary.toolCounts[s.id] = {
|
|
1503
|
+
advertised: s.advertised,
|
|
1504
|
+
registered: s.registered,
|
|
1505
|
+
tokens: s.serverTokens,
|
|
1506
|
+
droppedReason
|
|
1507
|
+
};
|
|
1038
1508
|
}
|
|
1039
1509
|
return summary;
|
|
1040
1510
|
}
|
|
@@ -1099,7 +1569,8 @@ var McpExtension = class {
|
|
|
1099
1569
|
},
|
|
1100
1570
|
serverStatus: {
|
|
1101
1571
|
status: "timeout",
|
|
1102
|
-
stderr: result.stderr
|
|
1572
|
+
stderr: result.stderr,
|
|
1573
|
+
diagnostics: result.diagnostics
|
|
1103
1574
|
},
|
|
1104
1575
|
change,
|
|
1105
1576
|
error: null,
|
|
@@ -1146,7 +1617,8 @@ var McpExtension = class {
|
|
|
1146
1617
|
},
|
|
1147
1618
|
serverStatus: {
|
|
1148
1619
|
status: "timeout",
|
|
1149
|
-
stderr: retry.stderr
|
|
1620
|
+
stderr: retry.stderr,
|
|
1621
|
+
diagnostics: retry.diagnostics
|
|
1150
1622
|
},
|
|
1151
1623
|
change: null,
|
|
1152
1624
|
error: null,
|
|
@@ -1179,7 +1651,8 @@ var McpExtension = class {
|
|
|
1179
1651
|
store: connected,
|
|
1180
1652
|
serverStatus: {
|
|
1181
1653
|
status: "timeout",
|
|
1182
|
-
stderr
|
|
1654
|
+
stderr,
|
|
1655
|
+
diagnostics: null
|
|
1183
1656
|
},
|
|
1184
1657
|
change: null,
|
|
1185
1658
|
error: null,
|
|
@@ -1190,6 +1663,64 @@ var McpExtension = class {
|
|
|
1190
1663
|
try {
|
|
1191
1664
|
mcpTools = (await connected.client.listTools()).tools;
|
|
1192
1665
|
} catch (err) {
|
|
1666
|
+
if (isUnauthorizedError(err) && serverConfig.transport === "http") {
|
|
1667
|
+
await closeConnected(connected);
|
|
1668
|
+
const retry = await connectClient(id, serverConfig, { connectTimeoutMs });
|
|
1669
|
+
if (retry.status === "pending_auth") return {
|
|
1670
|
+
id,
|
|
1671
|
+
store: {
|
|
1672
|
+
client: retry.client,
|
|
1673
|
+
configKey,
|
|
1674
|
+
status: "pending_auth",
|
|
1675
|
+
stderrBuffer: null,
|
|
1676
|
+
cliHint: retry.cliHint
|
|
1677
|
+
},
|
|
1678
|
+
serverStatus: {
|
|
1679
|
+
status: "pending_auth",
|
|
1680
|
+
stderr: "",
|
|
1681
|
+
cliHint: retry.cliHint
|
|
1682
|
+
},
|
|
1683
|
+
change: null,
|
|
1684
|
+
error: null,
|
|
1685
|
+
tools: null
|
|
1686
|
+
};
|
|
1687
|
+
if (retry.status === "failed") return {
|
|
1688
|
+
id,
|
|
1689
|
+
store: null,
|
|
1690
|
+
serverStatus: {
|
|
1691
|
+
status: "failed",
|
|
1692
|
+
error: retry.error,
|
|
1693
|
+
stderr: retry.stderr
|
|
1694
|
+
},
|
|
1695
|
+
change: null,
|
|
1696
|
+
error: {
|
|
1697
|
+
serverId: id,
|
|
1698
|
+
message: retry.error
|
|
1699
|
+
},
|
|
1700
|
+
tools: null
|
|
1701
|
+
};
|
|
1702
|
+
if (retry.status === "connected") {
|
|
1703
|
+
connected = {
|
|
1704
|
+
client: retry.client,
|
|
1705
|
+
configKey,
|
|
1706
|
+
status: "connected",
|
|
1707
|
+
stderrBuffer: retry.stderr,
|
|
1708
|
+
cliHint: null
|
|
1709
|
+
};
|
|
1710
|
+
mcpTools = (await connected.client.listTools()).tools;
|
|
1711
|
+
return {
|
|
1712
|
+
id,
|
|
1713
|
+
store: connected,
|
|
1714
|
+
serverStatus: { status: "connected" },
|
|
1715
|
+
change: action === "reused" ? "refreshed" : action,
|
|
1716
|
+
error: null,
|
|
1717
|
+
tools: {
|
|
1718
|
+
client: connected.client,
|
|
1719
|
+
list: mcpTools
|
|
1720
|
+
}
|
|
1721
|
+
};
|
|
1722
|
+
}
|
|
1723
|
+
}
|
|
1193
1724
|
const message = err instanceof Error ? err.message : String(err);
|
|
1194
1725
|
const stderr = connected.stderrBuffer?.read() ?? "";
|
|
1195
1726
|
return {
|
|
@@ -1220,17 +1751,21 @@ var McpExtension = class {
|
|
|
1220
1751
|
}
|
|
1221
1752
|
};
|
|
1222
1753
|
}
|
|
1223
|
-
async reconcileAndRecordMtime({ pi, configPath, reason }) {
|
|
1754
|
+
async reconcileAndRecordMtime({ pi, configPath, reason, contextWindow }) {
|
|
1224
1755
|
const summary = await this.reconcile({
|
|
1225
1756
|
pi,
|
|
1226
|
-
configPath
|
|
1757
|
+
configPath,
|
|
1758
|
+
contextWindow
|
|
1227
1759
|
});
|
|
1228
1760
|
this.lastConfigMtimeMs = await readConfigMtimeMs(configPath);
|
|
1229
1761
|
if (reason !== "session_start" && summaryHasChanges(summary)) this.pendingMcpUpdate = summary;
|
|
1230
|
-
log$
|
|
1762
|
+
log$13.info({
|
|
1231
1763
|
event: "mcp_reconcile",
|
|
1232
1764
|
reason,
|
|
1233
1765
|
total_tools: summary.totalTools,
|
|
1766
|
+
tokens_used: summary.tokensUsed,
|
|
1767
|
+
token_budget: summary.tokenBudget,
|
|
1768
|
+
dropped_tools: summary.droppedTools,
|
|
1234
1769
|
added: summary.added,
|
|
1235
1770
|
removed: summary.removed,
|
|
1236
1771
|
refreshed: summary.refreshed,
|
|
@@ -1247,10 +1782,11 @@ var McpExtension = class {
|
|
|
1247
1782
|
await this.reconcileAndRecordMtime({
|
|
1248
1783
|
pi,
|
|
1249
1784
|
configPath,
|
|
1250
|
-
reason: "session_start"
|
|
1785
|
+
reason: "session_start",
|
|
1786
|
+
contextWindow: ctx.model?.contextWindow
|
|
1251
1787
|
});
|
|
1252
1788
|
} catch (err) {
|
|
1253
|
-
log$
|
|
1789
|
+
log$13.error({
|
|
1254
1790
|
err,
|
|
1255
1791
|
event: "mcp_reconcile_failed"
|
|
1256
1792
|
}, "MCP reconcile failed");
|
|
@@ -1262,7 +1798,7 @@ var McpExtension = class {
|
|
|
1262
1798
|
try {
|
|
1263
1799
|
mtime = await readConfigMtimeMs(configPath);
|
|
1264
1800
|
} catch (err) {
|
|
1265
|
-
log$
|
|
1801
|
+
log$13.warn({
|
|
1266
1802
|
err,
|
|
1267
1803
|
event: "mcp_mtime_check_failed"
|
|
1268
1804
|
}, "mtime check on mcp.config.json failed");
|
|
@@ -1273,10 +1809,11 @@ var McpExtension = class {
|
|
|
1273
1809
|
await this.reconcileAndRecordMtime({
|
|
1274
1810
|
pi,
|
|
1275
1811
|
configPath,
|
|
1276
|
-
reason: "auto_reload"
|
|
1812
|
+
reason: "auto_reload",
|
|
1813
|
+
contextWindow: ctx.model?.contextWindow
|
|
1277
1814
|
});
|
|
1278
1815
|
} catch (err) {
|
|
1279
|
-
log$
|
|
1816
|
+
log$13.error({
|
|
1280
1817
|
err,
|
|
1281
1818
|
event: "mcp_auto_reload_failed"
|
|
1282
1819
|
}, "auto-reload after mcp.config.json change failed");
|
|
@@ -1293,7 +1830,8 @@ var McpExtension = class {
|
|
|
1293
1830
|
const summary = await this.reconcileAndRecordMtime({
|
|
1294
1831
|
pi,
|
|
1295
1832
|
configPath,
|
|
1296
|
-
reason: "tool"
|
|
1833
|
+
reason: "tool",
|
|
1834
|
+
contextWindow: ctx.model?.contextWindow
|
|
1297
1835
|
});
|
|
1298
1836
|
return {
|
|
1299
1837
|
content: [{
|
|
@@ -1335,9 +1873,21 @@ function pendingAuthEntries(summary) {
|
|
|
1335
1873
|
function timeoutEntries(summary) {
|
|
1336
1874
|
return Object.entries(summary.servers).flatMap(([id, status]) => status.status === "timeout" ? [{
|
|
1337
1875
|
id,
|
|
1338
|
-
stderr: status.stderr
|
|
1876
|
+
stderr: status.stderr,
|
|
1877
|
+
diagnostics: status.diagnostics
|
|
1339
1878
|
}] : []);
|
|
1340
1879
|
}
|
|
1880
|
+
/**
|
|
1881
|
+
* One line describing what the connect probe learned about a timed-out
|
|
1882
|
+
* stdio child, so an empty stderr is no longer a dead end. Empty array
|
|
1883
|
+
* when the entry was re-surfaced without a fresh probe.
|
|
1884
|
+
*/
|
|
1885
|
+
function timeoutDiagnosticLines(diagnostics) {
|
|
1886
|
+
if (diagnostics === null) return [];
|
|
1887
|
+
const alive = diagnostics.alive === null ? "unknown (no pid)" : diagnostics.alive ? "alive" : "DEAD";
|
|
1888
|
+
const spoke = diagnostics.stdoutMessages === 0 ? "no JSON-RPC received on stdout (initialize never answered — slow startup, or blocked e.g. on OAuth)" : `${diagnostics.stdoutMessages} JSON-RPC message(s) received on stdout (server spoke but the handshake stalled)`;
|
|
1889
|
+
return [` child process: ${alive}${diagnostics.pid !== null ? ` (pid ${diagnostics.pid})` : ""}; ${spoke}`];
|
|
1890
|
+
}
|
|
1341
1891
|
function failedEntries(summary) {
|
|
1342
1892
|
return Object.entries(summary.servers).flatMap(([id, status]) => status.status === "failed" ? [{
|
|
1343
1893
|
id,
|
|
@@ -1345,6 +1895,13 @@ function failedEntries(summary) {
|
|
|
1345
1895
|
stderr: status.stderr
|
|
1346
1896
|
}] : []);
|
|
1347
1897
|
}
|
|
1898
|
+
function appendTimeoutStderrBlock(lines, stderr) {
|
|
1899
|
+
if (!stderr) {
|
|
1900
|
+
lines.push(" stderr: (empty — the child wrote nothing to stderr)");
|
|
1901
|
+
return;
|
|
1902
|
+
}
|
|
1903
|
+
appendStderrBlock(lines, stderr);
|
|
1904
|
+
}
|
|
1348
1905
|
function appendStderrBlock(lines, stderr) {
|
|
1349
1906
|
if (!stderr) return;
|
|
1350
1907
|
lines.push(" stderr:");
|
|
@@ -1352,9 +1909,29 @@ function appendStderrBlock(lines, stderr) {
|
|
|
1352
1909
|
for (const line of stderr.trimEnd().split("\n")) lines.push(` ${line}`);
|
|
1353
1910
|
lines.push(" ---");
|
|
1354
1911
|
}
|
|
1912
|
+
/**
|
|
1913
|
+
* Human-readable lines describing any budget-driven truncation, shared by
|
|
1914
|
+
* `summaryText` (the reload_mcp tool output) and `formatMcpUpdateMessage` (the
|
|
1915
|
+
* synthetic continuation). Empty when nothing was dropped.
|
|
1916
|
+
*/
|
|
1917
|
+
function truncationLines(summary) {
|
|
1918
|
+
if (summary.droppedTools <= 0) return [];
|
|
1919
|
+
const lines = [];
|
|
1920
|
+
const { maxTotalTools, maxToolsPerServer } = summary.limits;
|
|
1921
|
+
const totalCapHint = Number.isFinite(maxTotalTools) ? `${maxTotalTools}` : "unlimited";
|
|
1922
|
+
lines.push(` WARNING: ${summary.droppedTools} MCP tool(s) were NOT registered because a tool budget was hit (token budget: ~${summary.tokenBudget} tokens, used ~${summary.tokensUsed}; total-tool cap: ${totalCapHint}, per-server cap: ${Number.isFinite(maxToolsPerServer) ? maxToolsPerServer : "unlimited"}).`);
|
|
1923
|
+
lines.push(" Tool definitions are sent to the model on every prompt; too many (or too-large) tool schemas push the request over the limit, so they are budgeted against the model context window. Prune servers from mcp.config.json to bring the tool surface down.");
|
|
1924
|
+
for (const [id, count] of Object.entries(summary.toolCounts)) {
|
|
1925
|
+
if (count.droppedReason === null) continue;
|
|
1926
|
+
const why = count.droppedReason === "tokens" ? "token budget exhausted" : count.droppedReason === "total" ? "total-tool cap reached" : "per-server cap";
|
|
1927
|
+
lines.push(` - ${id}: registered ${count.registered}/${count.advertised} tools (~${count.tokens} tokens, ${why}).`);
|
|
1928
|
+
}
|
|
1929
|
+
return lines;
|
|
1930
|
+
}
|
|
1355
1931
|
function summaryText(summary) {
|
|
1356
1932
|
const lines = [];
|
|
1357
1933
|
lines.push(`MCP reconcile complete: ${summary.totalTools} tool(s) live.`);
|
|
1934
|
+
lines.push(...truncationLines(summary));
|
|
1358
1935
|
if (summary.added.length > 0) lines.push(` Added: ${summary.added.join(", ")}`);
|
|
1359
1936
|
if (summary.refreshed.length > 0) lines.push(` Refreshed: ${summary.refreshed.join(", ")}`);
|
|
1360
1937
|
if (summary.removed.length > 0) lines.push(` Removed: ${summary.removed.join(", ")}`);
|
|
@@ -1363,11 +1940,12 @@ function summaryText(summary) {
|
|
|
1363
1940
|
lines.push(` run \`${cliHint}\` in the shell to authenticate.`);
|
|
1364
1941
|
appendStderrBlock(lines, stderr);
|
|
1365
1942
|
}
|
|
1366
|
-
for (const { id, stderr } of timeoutEntries(summary)) {
|
|
1943
|
+
for (const { id, stderr, diagnostics } of timeoutEntries(summary)) {
|
|
1367
1944
|
lines.push(` TIMED OUT ${id}:`);
|
|
1368
1945
|
lines.push(` bridge spawned but did not complete initialize in time`);
|
|
1369
1946
|
lines.push(` (usually mcp-remote mid-OAuth — see captured stderr).`);
|
|
1370
|
-
|
|
1947
|
+
lines.push(...timeoutDiagnosticLines(diagnostics));
|
|
1948
|
+
appendTimeoutStderrBlock(lines, stderr);
|
|
1371
1949
|
}
|
|
1372
1950
|
for (const { id, error, stderr } of failedEntries(summary)) {
|
|
1373
1951
|
lines.push(` FAILED ${id}: ${error}`);
|
|
@@ -1380,7 +1958,7 @@ function summaryText(summary) {
|
|
|
1380
1958
|
return lines.join("\n");
|
|
1381
1959
|
}
|
|
1382
1960
|
function summaryHasChanges(summary) {
|
|
1383
|
-
return summary.added.length > 0 || summary.removed.length > 0 || summary.refreshed.length > 0 || summary.errors.length > 0;
|
|
1961
|
+
return summary.added.length > 0 || summary.removed.length > 0 || summary.refreshed.length > 0 || summary.errors.length > 0 || summary.droppedTools > 0;
|
|
1384
1962
|
}
|
|
1385
1963
|
/**
|
|
1386
1964
|
* Format a queued tool-update as a synthetic system-style message for
|
|
@@ -1393,6 +1971,10 @@ function formatMcpUpdateMessage(summary) {
|
|
|
1393
1971
|
if (summary.added.length > 0) lines.push(`Newly available servers: ${summary.added.join(", ")}`);
|
|
1394
1972
|
if (summary.refreshed.length > 0) lines.push(`Refreshed servers: ${summary.refreshed.join(", ")}`);
|
|
1395
1973
|
if (summary.removed.length > 0) lines.push(`Removed servers (and their tools): ${summary.removed.join(", ")}`);
|
|
1974
|
+
if (summary.droppedTools > 0) {
|
|
1975
|
+
lines.push("");
|
|
1976
|
+
lines.push(...truncationLines(summary));
|
|
1977
|
+
}
|
|
1396
1978
|
const pending = pendingAuthEntries(summary);
|
|
1397
1979
|
if (pending.length > 0) {
|
|
1398
1980
|
lines.push("");
|
|
@@ -1408,9 +1990,10 @@ function formatMcpUpdateMessage(summary) {
|
|
|
1408
1990
|
if (timedOut.length > 0) {
|
|
1409
1991
|
lines.push("");
|
|
1410
1992
|
lines.push("Servers that timed out during initialize (bridge is still alive in the background; usually mcp-remote-style stdio bridges mid-OAuth):");
|
|
1411
|
-
for (const { id, stderr } of timedOut) {
|
|
1993
|
+
for (const { id, stderr, diagnostics } of timedOut) {
|
|
1412
1994
|
lines.push(` - ${id}:`);
|
|
1413
|
-
|
|
1995
|
+
lines.push(...timeoutDiagnosticLines(diagnostics));
|
|
1996
|
+
appendTimeoutStderrBlock(lines, stderr);
|
|
1414
1997
|
}
|
|
1415
1998
|
lines.push("If the bridge prints an auth URL in its stderr, share it with the user. Then call `reload_mcp` once they've finished.");
|
|
1416
1999
|
}
|
|
@@ -1499,156 +2082,16 @@ async function runToolUpdateLoop({ session, log }) {
|
|
|
1499
2082
|
}, "reached tool-update continuation cap; further updates will land on next request");
|
|
1500
2083
|
}
|
|
1501
2084
|
//#endregion
|
|
1502
|
-
//#region src/tracing.ts
|
|
1503
|
-
const SERVICE_NAME = "skydive-agent-harness";
|
|
1504
|
-
const DAEMON_TRACES_URL = "http://localhost:38994/v1/traces";
|
|
1505
|
-
let provider = null;
|
|
1506
|
-
function initTracing() {
|
|
1507
|
-
if (provider) return;
|
|
1508
|
-
propagation.setGlobalPropagator(new W3CTraceContextPropagator());
|
|
1509
|
-
provider = new NodeTracerProvider({ resource: new Resource({ [ATTR_SERVICE_NAME]: SERVICE_NAME }) });
|
|
1510
|
-
const exporter = new OTLPTraceExporter({ url: DAEMON_TRACES_URL });
|
|
1511
|
-
provider.addSpanProcessor(new BatchSpanProcessor(exporter, { scheduledDelayMillis: 1e3 }));
|
|
1512
|
-
provider.register();
|
|
1513
|
-
logger.info({
|
|
1514
|
-
event: "tracing_enabled",
|
|
1515
|
-
endpoint: DAEMON_TRACES_URL
|
|
1516
|
-
}, "OTel tracing initialized — exporting via daemon");
|
|
1517
|
-
const shutdown = async () => {
|
|
1518
|
-
await shutdownTracing();
|
|
1519
|
-
process.exit(0);
|
|
1520
|
-
};
|
|
1521
|
-
process.on("SIGTERM", shutdown);
|
|
1522
|
-
process.on("SIGINT", shutdown);
|
|
1523
|
-
}
|
|
1524
|
-
function getTracer() {
|
|
1525
|
-
return trace.getTracer(SERVICE_NAME);
|
|
1526
|
-
}
|
|
1527
|
-
function extractRemoteContext() {
|
|
1528
|
-
const traceparent = getCurrentTraceparent();
|
|
1529
|
-
if (!traceparent) return ROOT_CONTEXT;
|
|
1530
|
-
const carrier = { traceparent };
|
|
1531
|
-
return propagation.extract(ROOT_CONTEXT, carrier, {
|
|
1532
|
-
get: (c, key) => c[key],
|
|
1533
|
-
keys: (c) => Object.keys(c)
|
|
1534
|
-
});
|
|
1535
|
-
}
|
|
1536
|
-
async function shutdownTracing() {
|
|
1537
|
-
if (provider) await provider.shutdown();
|
|
1538
|
-
}
|
|
1539
|
-
//#endregion
|
|
1540
|
-
//#region src/harness.ts
|
|
1541
|
-
/**
|
|
1542
|
-
* Skydive composition over @skydiveai/pi-server: wires the platform
|
|
1543
|
-
* defaults (tracing to the daemon, tool-update hot-reload hooks, prewarm
|
|
1544
|
-
* paths, header passthrough for proxy routing, Skydive agent-card
|
|
1545
|
-
* branding) into the generic protocol server, and returns mountable
|
|
1546
|
-
* handlers. The agent supplies the pi session factory (their session.ts)
|
|
1547
|
-
* and owns the express app:
|
|
1548
|
-
*
|
|
1549
|
-
* const { platform, protocols } = createHarness({
|
|
1550
|
-
* cwd: process.cwd(),
|
|
1551
|
-
* createSession,
|
|
1552
|
-
* });
|
|
1553
|
-
* app.get('/health', platform.handlers.health);
|
|
1554
|
-
* app.use(platform.handlers.injectEnv);
|
|
1555
|
-
* app.use(platform.handlers.prewarm); // matches /_skydive/prewarm + legacy alias
|
|
1556
|
-
* app.use(protocols.handlers.all);
|
|
1557
|
-
*/
|
|
1558
|
-
const A2A_PATH = "/a2a";
|
|
1559
|
-
const AGENT_CARD_PATH = "/.well-known/agent-card.json";
|
|
1560
|
-
const PREWARM_PATHS = ["/_skydive/prewarm", "/_anyone/prewarm"];
|
|
1561
|
-
const PASSTHROUGH_HEADER_PREFIXES = ["x-anyone-", "x-skydive-"];
|
|
1562
|
-
function createHarness(options) {
|
|
1563
|
-
initTracing();
|
|
1564
|
-
readPlatformVersions().then((runtimeVersions) => logger.info({
|
|
1565
|
-
event: "runtime_versions",
|
|
1566
|
-
runtimeVersions
|
|
1567
|
-
}, "platform runtime versions"));
|
|
1568
|
-
const serverOptions = {
|
|
1569
|
-
...options,
|
|
1570
|
-
onSessionSetup: options.onSessionSetup ?? ((args) => {
|
|
1571
|
-
installToolUpdateAutoStop(args);
|
|
1572
|
-
installIterationCap(args);
|
|
1573
|
-
}),
|
|
1574
|
-
postPrompt: options.postPrompt ?? runToolUpdateLoop,
|
|
1575
|
-
passthroughHeaderPrefixes: options.passthroughHeaderPrefixes ?? PASSTHROUGH_HEADER_PREFIXES,
|
|
1576
|
-
prewarmPaths: options.prewarmPaths ?? PREWARM_PATHS
|
|
1577
|
-
};
|
|
1578
|
-
const webHandlers = createProtocolHandlers(serverOptions);
|
|
1579
|
-
const cardOverrides = {
|
|
1580
|
-
name: "Skydive Agent",
|
|
1581
|
-
description: "An AI coding agent powered by the Skydive platform.",
|
|
1582
|
-
...options.agentCard
|
|
1583
|
-
};
|
|
1584
|
-
const agentCardWeb = async (request) => {
|
|
1585
|
-
const url = new URL(request.url);
|
|
1586
|
-
if (request.method !== "GET" || url.pathname !== AGENT_CARD_PATH) return null;
|
|
1587
|
-
return Response.json(buildAgentCard(cardOverrides.url ?? `${url.origin}${A2A_PATH}`, cardOverrides));
|
|
1588
|
-
};
|
|
1589
|
-
const a2a = restHandler({
|
|
1590
|
-
requestHandler: new DefaultRequestHandler(buildAgentCard(cardOverrides.url ?? A2A_PATH, cardOverrides), new InMemoryTaskStore(), createAgentExecutor(serverOptions), new DefaultExecutionEventBusManager()),
|
|
1591
|
-
userBuilder: UserBuilder.noAuthentication
|
|
1592
|
-
});
|
|
1593
|
-
return {
|
|
1594
|
-
platform: { handlers: {
|
|
1595
|
-
/**
|
|
1596
|
-
* Platform health plus the agent's `healthMetadata`. Mount above
|
|
1597
|
-
* injectEnv — health must respond immediately for prewarm
|
|
1598
|
-
* stashing and readiness probes, and injectEnv can wait up to
|
|
1599
|
-
* 10s for env vars during boot.
|
|
1600
|
-
*/
|
|
1601
|
-
health: createHealthHandler({ metadata: options.healthMetadata ?? null }),
|
|
1602
|
-
/** Loads platform env (e2b envd / daemon long-poll). */
|
|
1603
|
-
injectEnv: createPlatformEnvMiddleware(),
|
|
1604
|
-
prewarm: webHandlerToMiddleware(webHandlers.prewarm)
|
|
1605
|
-
} },
|
|
1606
|
-
protocols: { handlers: {
|
|
1607
|
-
/** Express-style; mount at /a2a. */
|
|
1608
|
-
a2a,
|
|
1609
|
-
/** GET /.well-known/agent-card.json. */
|
|
1610
|
-
agentCard: webHandlerToMiddleware(agentCardWeb),
|
|
1611
|
-
/** Mirrors each vendor's API shape. */
|
|
1612
|
-
openai: { v1: {
|
|
1613
|
-
chat: { completions: webHandlerToMiddleware(webHandlers.chatCompletions) },
|
|
1614
|
-
responses: webHandlerToMiddleware(webHandlers.responses)
|
|
1615
|
-
} },
|
|
1616
|
-
anthropic: { v1: { messages: webHandlerToMiddleware(webHandlers.messages) } },
|
|
1617
|
-
/**
|
|
1618
|
-
* Everything in one mount: a2a (+ agent card) at their well-known
|
|
1619
|
-
* paths, then chat-completions / anthropic-messages / responses.
|
|
1620
|
-
* Calls next() when nothing matches.
|
|
1621
|
-
*/
|
|
1622
|
-
all: chainMiddleware([mountAt(A2A_PATH, a2a), webHandlerToMiddleware(composeHandlers([
|
|
1623
|
-
agentCardWeb,
|
|
1624
|
-
webHandlers.chatCompletions,
|
|
1625
|
-
webHandlers.messages,
|
|
1626
|
-
webHandlers.responses
|
|
1627
|
-
]))])
|
|
1628
|
-
} }
|
|
1629
|
-
};
|
|
1630
|
-
}
|
|
1631
|
-
/** Effective bash timeout: the model's value when it gave a positive number, else the default. */
|
|
1632
|
-
function resolveBashTimeout(provided) {
|
|
1633
|
-
return typeof provided === "number" && provided > 0 ? provided : 600;
|
|
1634
|
-
}
|
|
1635
|
-
const bashDefaultTimeoutExtension = (pi) => {
|
|
1636
|
-
pi.on("tool_call", async (event) => {
|
|
1637
|
-
if (event.toolName !== "bash") return;
|
|
1638
|
-
event.input.timeout = resolveBashTimeout(event.input.timeout);
|
|
1639
|
-
});
|
|
1640
|
-
};
|
|
1641
|
-
//#endregion
|
|
1642
2085
|
//#region src/channel-context-ref.ts
|
|
1643
2086
|
/**
|
|
1644
|
-
* The worker injects
|
|
1645
|
-
* sandbox env (`SKYDIVE_CHANNEL_CONTEXT`) rather than the full context.
|
|
2087
|
+
* The worker injects a small reference — `{ channel, messageId }` —
|
|
2088
|
+
* into the sandbox env (`SKYDIVE_CHANNEL_CONTEXT`) rather than the full context.
|
|
1646
2089
|
*
|
|
1647
2090
|
* The canonical `ChannelContextRef` type + `parseChannelContextRef` live in
|
|
1648
2091
|
* `@createinc/anyone-channels`, but the harness (`@skydiveai/*`) keeps zero
|
|
1649
2092
|
* `@createinc/*` dependencies — importing that package would pull the whole
|
|
1650
|
-
* platform channel stack (Slack/email/Linq SDKs, messaging)
|
|
1651
|
-
*
|
|
2093
|
+
* platform channel stack (Slack/email/Linq SDKs, messaging) just to read one
|
|
2094
|
+
* field. So we validate the field this consumer needs locally instead.
|
|
1652
2095
|
*/
|
|
1653
2096
|
const channelContextRefSchema = z.object({ messageId: z.string().nullable() });
|
|
1654
2097
|
/**
|
|
@@ -1680,39 +2123,108 @@ function apiBaseUrl() {
|
|
|
1680
2123
|
}
|
|
1681
2124
|
//#endregion
|
|
1682
2125
|
//#region src/extensions/platform.ts
|
|
2126
|
+
/**
|
|
2127
|
+
* Platform extension — bridges the agent harness to the Skydive platform daemon.
|
|
2128
|
+
*
|
|
2129
|
+
* Responsibilities:
|
|
2130
|
+
* - Heartbeat: periodic POST to the API so the sandbox manager knows the
|
|
2131
|
+
* agent is alive. Throttled to once per minute, triggered by tool events.
|
|
2132
|
+
* - Session tracking: registers the session with the daemon on start,
|
|
2133
|
+
* streams tool_call / tool_result events so the daemon can track which
|
|
2134
|
+
* session is actively executing, and signals session end on agent_end.
|
|
2135
|
+
* - Channel context: passes the SKYDIVE_CHANNEL_CONTEXT (containing the
|
|
2136
|
+
* messageId) to the daemon so file writes can be attributed to the
|
|
2137
|
+
* correct conversation.
|
|
2138
|
+
*
|
|
2139
|
+
* All daemon POSTs are fire-and-forget — failures are logged but never
|
|
2140
|
+
* block the agent. The daemon may not be running (e.g. local dev without
|
|
2141
|
+
* a sandbox), and that's fine.
|
|
2142
|
+
*/
|
|
1683
2143
|
const HEARTBEAT_THROTTLE_MS = 6e4;
|
|
1684
2144
|
const TOOL_HEARTBEAT_INTERVAL_MS = 5e3;
|
|
1685
2145
|
const MAX_TOOL_HEARTBEATS = 1440 * 60 * 1e3 / TOOL_HEARTBEAT_INTERVAL_MS;
|
|
1686
2146
|
const DAEMON_URL = "http://localhost:38994";
|
|
1687
|
-
const log$
|
|
2147
|
+
const log$12 = logger.child({ module: "platform-ext" });
|
|
1688
2148
|
function sandboxClient() {
|
|
1689
2149
|
const apiUrl = apiBaseUrl();
|
|
1690
2150
|
if (!apiUrl) return null;
|
|
1691
2151
|
return hc(`${apiUrl}/api/v1/sandbox`);
|
|
1692
2152
|
}
|
|
1693
2153
|
/**
|
|
1694
|
-
*
|
|
1695
|
-
*
|
|
1696
|
-
*
|
|
1697
|
-
*
|
|
1698
|
-
*
|
|
1699
|
-
*
|
|
2154
|
+
* Is this box still an unclaimed warm-pool sandbox? (ANY-6000, the
|
|
2155
|
+
* feature-flags half of the ANY-5184 pool 403 wave.)
|
|
2156
|
+
*
|
|
2157
|
+
* `GET /sandbox/feature-flags` is agent-only, so the shared poller's request
|
|
2158
|
+
* from a pool box can only 403 — a guaranteed-failing GET every 60s for the
|
|
2159
|
+
* life of the pool phase. The discriminator is the sandbox token's `type`
|
|
2160
|
+
* claim, read UNVERIFIED (this box never holds the signing secret): not an
|
|
2161
|
+
* authorization decision, only "should I bother calling?", and the api still
|
|
2162
|
+
* authorizes every request.
|
|
2163
|
+
*
|
|
2164
|
+
* Read per call from the daemon's persisted env file, NOT process.env:
|
|
2165
|
+
* claiming a pool box rebinds the token in place (the daemon rewrites this
|
|
2166
|
+
* file) while the harness's process.env keeps the boot snapshot, so a
|
|
2167
|
+
* process-env gate would leave a claimed box permanently skipping — trading a
|
|
2168
|
+
* wasted request for silently frozen flags, which is strictly worse. "Cannot
|
|
2169
|
+
* tell" (no file, no token, unparseable payload) reports false so the poll
|
|
2170
|
+
* proceeds.
|
|
2171
|
+
*/
|
|
2172
|
+
const daemonEnvIdentitySchema = z.object({
|
|
2173
|
+
ANYONE_SANDBOX_TOKEN: z.string().optional(),
|
|
2174
|
+
SKYDIVE_SANDBOX_TOKEN: z.string().optional()
|
|
2175
|
+
}).passthrough();
|
|
2176
|
+
const tokenTypeSchema = z.object({ type: z.string() }).passthrough();
|
|
2177
|
+
async function isPoolIdentity() {
|
|
2178
|
+
try {
|
|
2179
|
+
const override = process.env.ANYONE_DAEMON_ENV_CACHE;
|
|
2180
|
+
const candidates = override ? [override] : ["/run/anyone-system/daemon-env.json", "/tmp/.anyone/daemon-env.json"];
|
|
2181
|
+
let raw = null;
|
|
2182
|
+
for (const file of candidates) {
|
|
2183
|
+
raw = await readFile(file, "utf8").catch(() => null);
|
|
2184
|
+
if (raw !== null) break;
|
|
2185
|
+
}
|
|
2186
|
+
if (raw === null) return false;
|
|
2187
|
+
const env = daemonEnvIdentitySchema.safeParse(JSON.parse(raw));
|
|
2188
|
+
if (!env.success) return false;
|
|
2189
|
+
const token = env.data.ANYONE_SANDBOX_TOKEN ?? env.data.SKYDIVE_SANDBOX_TOKEN;
|
|
2190
|
+
if (typeof token !== "string" || token === "") return false;
|
|
2191
|
+
const payload = token.split(".")[1];
|
|
2192
|
+
if (!payload) return false;
|
|
2193
|
+
const claims = tokenTypeSchema.safeParse(JSON.parse(Buffer.from(payload, "base64url").toString("utf8")));
|
|
2194
|
+
return claims.success && claims.data.type === "onboarding-pool";
|
|
2195
|
+
} catch (_err) {
|
|
2196
|
+
return false;
|
|
2197
|
+
}
|
|
2198
|
+
}
|
|
2199
|
+
/**
|
|
2200
|
+
* Fetch every harness feature flag in one GET (`{ contextManagement, ... }`
|
|
2201
|
+
* — see apps/anyone/api/src/routes/sandbox-feature-flags.ts). Returns
|
|
2202
|
+
* null when indeterminate (no api url, the request failed, or the box is an
|
|
2203
|
+
* unclaimed pool sandbox whose token the route would 403) so the shared
|
|
2204
|
+
* poller keeps the last-known values rather than flipping on a transient error.
|
|
2205
|
+
* This is the single fetch behind `feature-flags-poll.ts`; extensions read the
|
|
2206
|
+
* polled values there instead of issuing their own GET.
|
|
1700
2207
|
*/
|
|
1701
2208
|
async function fetchHarnessFlags() {
|
|
1702
2209
|
const client = sandboxClient();
|
|
1703
2210
|
if (!client) return null;
|
|
2211
|
+
if (await isPoolIdentity()) return null;
|
|
1704
2212
|
try {
|
|
1705
2213
|
const res = await client["feature-flags"].$get();
|
|
1706
2214
|
if (!res.ok) {
|
|
1707
|
-
log$
|
|
2215
|
+
log$12.debug({
|
|
1708
2216
|
status: res.status,
|
|
1709
2217
|
event: "feature_flags_fetch_failed"
|
|
1710
2218
|
}, "feature-flags fetch failed");
|
|
1711
2219
|
return null;
|
|
1712
2220
|
}
|
|
1713
|
-
|
|
2221
|
+
const body = await res.json();
|
|
2222
|
+
return {
|
|
2223
|
+
contextManagement: body.contextManagement ?? null,
|
|
2224
|
+
closingText: body.closingText ?? null
|
|
2225
|
+
};
|
|
1714
2226
|
} catch (err) {
|
|
1715
|
-
log$
|
|
2227
|
+
log$12.debug({
|
|
1716
2228
|
err,
|
|
1717
2229
|
event: "feature_flags_fetch_error"
|
|
1718
2230
|
}, "feature-flags request errored");
|
|
@@ -1723,7 +2235,7 @@ function postHeartbeat({ messageId }) {
|
|
|
1723
2235
|
const client = sandboxClient();
|
|
1724
2236
|
if (!client) return;
|
|
1725
2237
|
client.heartbeat.$post({ json: { messageId } }).catch((err) => {
|
|
1726
|
-
log$
|
|
2238
|
+
log$12.debug({
|
|
1727
2239
|
err,
|
|
1728
2240
|
event: "heartbeat_failed"
|
|
1729
2241
|
}, "heartbeat failed");
|
|
@@ -1735,7 +2247,7 @@ async function resolveConversationFromApi(messageId) {
|
|
|
1735
2247
|
try {
|
|
1736
2248
|
const res = await client["message-conversation"].$get({ query: { messageId } });
|
|
1737
2249
|
if (!res.ok) {
|
|
1738
|
-
log$
|
|
2250
|
+
log$12.warn({
|
|
1739
2251
|
status: res.status,
|
|
1740
2252
|
messageId,
|
|
1741
2253
|
event: "resolve_conversation_failed"
|
|
@@ -1744,7 +2256,7 @@ async function resolveConversationFromApi(messageId) {
|
|
|
1744
2256
|
}
|
|
1745
2257
|
return (await res.json()).conversationId ?? null;
|
|
1746
2258
|
} catch (err) {
|
|
1747
|
-
log$
|
|
2259
|
+
log$12.warn({
|
|
1748
2260
|
err,
|
|
1749
2261
|
messageId,
|
|
1750
2262
|
event: "resolve_conversation_error"
|
|
@@ -1761,6 +2273,71 @@ async function postBackgroundTaskDone({ messageId, content }) {
|
|
|
1761
2273
|
} });
|
|
1762
2274
|
if (!res.ok) throw new Error(`bg-task-done POST failed: ${res.status}`);
|
|
1763
2275
|
}
|
|
2276
|
+
async function putBackgroundTaskJournalSpec({ messageId, spec }) {
|
|
2277
|
+
const client = sandboxClient();
|
|
2278
|
+
if (!client) return;
|
|
2279
|
+
try {
|
|
2280
|
+
const res = await client["bg-task-journal"].$put({ json: {
|
|
2281
|
+
messageId,
|
|
2282
|
+
spec
|
|
2283
|
+
} });
|
|
2284
|
+
if (!res.ok) log$12.warn({
|
|
2285
|
+
status: res.status,
|
|
2286
|
+
taskId: spec.id,
|
|
2287
|
+
event: "bg_journal_put_failed"
|
|
2288
|
+
}, "bg-task journal PUT failed");
|
|
2289
|
+
} catch (err) {
|
|
2290
|
+
log$12.warn({
|
|
2291
|
+
err,
|
|
2292
|
+
taskId: spec.id,
|
|
2293
|
+
event: "bg_journal_put_failed"
|
|
2294
|
+
}, "bg-task journal PUT threw");
|
|
2295
|
+
}
|
|
2296
|
+
}
|
|
2297
|
+
async function deleteBackgroundTaskJournalSpec({ messageId, taskId }) {
|
|
2298
|
+
const client = sandboxClient();
|
|
2299
|
+
if (!client) return;
|
|
2300
|
+
try {
|
|
2301
|
+
await client["bg-task-journal"].$delete({ json: {
|
|
2302
|
+
messageId,
|
|
2303
|
+
taskId
|
|
2304
|
+
} });
|
|
2305
|
+
} catch (err) {
|
|
2306
|
+
log$12.debug({
|
|
2307
|
+
err,
|
|
2308
|
+
taskId,
|
|
2309
|
+
event: "bg_journal_delete_failed"
|
|
2310
|
+
}, "bg-task journal DELETE failed");
|
|
2311
|
+
}
|
|
2312
|
+
}
|
|
2313
|
+
async function listBackgroundTaskJournalSpecs({ messageId }) {
|
|
2314
|
+
const client = sandboxClient();
|
|
2315
|
+
if (!client) return [];
|
|
2316
|
+
try {
|
|
2317
|
+
const res = await client["bg-task-journal"].$get({ query: { messageId } });
|
|
2318
|
+
if (!res.ok) return [];
|
|
2319
|
+
return (await res.json()).specs ?? [];
|
|
2320
|
+
} catch (err) {
|
|
2321
|
+
log$12.debug({
|
|
2322
|
+
err,
|
|
2323
|
+
event: "bg_journal_list_failed"
|
|
2324
|
+
}, "bg-task journal GET failed");
|
|
2325
|
+
return [];
|
|
2326
|
+
}
|
|
2327
|
+
}
|
|
2328
|
+
function postBackgroundTasksSnapshot({ messageId, tasks }) {
|
|
2329
|
+
const client = sandboxClient();
|
|
2330
|
+
if (!client || !messageId) return;
|
|
2331
|
+
client["bg-tasks"].$post({ json: {
|
|
2332
|
+
messageId,
|
|
2333
|
+
tasks
|
|
2334
|
+
} }).catch((err) => {
|
|
2335
|
+
log$12.debug({
|
|
2336
|
+
err,
|
|
2337
|
+
event: "bg_tasks_snapshot_failed"
|
|
2338
|
+
}, "bg-tasks snapshot publish failed");
|
|
2339
|
+
});
|
|
2340
|
+
}
|
|
1764
2341
|
async function postSubagentSpawn({ messageId, tasks }) {
|
|
1765
2342
|
const client = sandboxClient();
|
|
1766
2343
|
if (!client) throw new Error("no api url for subagent-spawn");
|
|
@@ -1768,8 +2345,43 @@ async function postSubagentSpawn({ messageId, tasks }) {
|
|
|
1768
2345
|
messageId,
|
|
1769
2346
|
tasks
|
|
1770
2347
|
} });
|
|
1771
|
-
if (!res.ok)
|
|
1772
|
-
|
|
2348
|
+
if (!res.ok) {
|
|
2349
|
+
let detail = "";
|
|
2350
|
+
try {
|
|
2351
|
+
const errBody = await res.json();
|
|
2352
|
+
if (errBody && typeof errBody.error === "string") detail = `: ${errBody.error}`;
|
|
2353
|
+
} catch {}
|
|
2354
|
+
throw new Error(`subagent-spawn POST failed (${res.status})${detail}`);
|
|
2355
|
+
}
|
|
2356
|
+
const body = await res.json();
|
|
2357
|
+
return {
|
|
2358
|
+
taskIds: body.taskIds,
|
|
2359
|
+
tasks: body.tasks ?? []
|
|
2360
|
+
};
|
|
2361
|
+
}
|
|
2362
|
+
async function postSubagentRearm({ messageId, taskId, timeoutMinutes }) {
|
|
2363
|
+
const client = sandboxClient();
|
|
2364
|
+
if (!client) throw new Error("no api url for subagent re-arm");
|
|
2365
|
+
const res = await client.a2a.tasks[":id"].rearm.$post({
|
|
2366
|
+
param: { id: taskId },
|
|
2367
|
+
json: {
|
|
2368
|
+
messageId,
|
|
2369
|
+
timeoutMinutes
|
|
2370
|
+
}
|
|
2371
|
+
});
|
|
2372
|
+
if (!res.ok) {
|
|
2373
|
+
let detail = "";
|
|
2374
|
+
try {
|
|
2375
|
+
const body = await res.json();
|
|
2376
|
+
if (typeof body.error === "string") detail = `: ${body.error}`;
|
|
2377
|
+
} catch {}
|
|
2378
|
+
throw new Error(`subagent re-arm POST failed (${res.status})${detail}`);
|
|
2379
|
+
}
|
|
2380
|
+
const body = await res.json();
|
|
2381
|
+
return {
|
|
2382
|
+
taskId: body.id,
|
|
2383
|
+
deadlineAt: body.deadlineAt
|
|
2384
|
+
};
|
|
1773
2385
|
}
|
|
1774
2386
|
function createHeartbeatThrottle({ messageId }) {
|
|
1775
2387
|
let lastAt = 0;
|
|
@@ -1818,7 +2430,7 @@ function createToolHeartbeat({ messageId }) {
|
|
|
1818
2430
|
}
|
|
1819
2431
|
heartbeatCount++;
|
|
1820
2432
|
if (heartbeatCount > MAX_TOOL_HEARTBEATS) {
|
|
1821
|
-
log$
|
|
2433
|
+
log$12.warn({
|
|
1822
2434
|
heartbeatCount,
|
|
1823
2435
|
activeToolCalls: [...activeToolCalls]
|
|
1824
2436
|
}, "tool heartbeat max reached, stopping");
|
|
@@ -1849,7 +2461,7 @@ function postToDaemon(path, body) {
|
|
|
1849
2461
|
headers: { "content-type": "application/json" },
|
|
1850
2462
|
body: JSON.stringify(body)
|
|
1851
2463
|
}).catch((err) => {
|
|
1852
|
-
log$
|
|
2464
|
+
log$12.debug({
|
|
1853
2465
|
err,
|
|
1854
2466
|
path,
|
|
1855
2467
|
event: "daemon_post_failed"
|
|
@@ -1858,7 +2470,7 @@ function postToDaemon(path, body) {
|
|
|
1858
2470
|
}
|
|
1859
2471
|
function createPlatformExtensions({ sessionId, channelContext }) {
|
|
1860
2472
|
return (pi) => {
|
|
1861
|
-
log$
|
|
2473
|
+
log$12.info({
|
|
1862
2474
|
sessionId,
|
|
1863
2475
|
hasChannelContext: Boolean(channelContext)
|
|
1864
2476
|
}, "platform extension initialized");
|
|
@@ -1907,94 +2519,793 @@ function createPlatformExtensions({ sessionId, channelContext }) {
|
|
|
1907
2519
|
});
|
|
1908
2520
|
});
|
|
1909
2521
|
pi.on("agent_end", () => {
|
|
1910
|
-
log$
|
|
2522
|
+
log$12.info({ sessionId }, "session ending");
|
|
1911
2523
|
postToDaemon("/session/end", { sessionId });
|
|
1912
2524
|
});
|
|
1913
2525
|
};
|
|
1914
2526
|
}
|
|
1915
2527
|
//#endregion
|
|
2528
|
+
//#region src/extensions/context-management-config.ts
|
|
2529
|
+
/**
|
|
2530
|
+
* Configuration for the context-management capability (tool-output trimming
|
|
2531
|
+
* and client-side microcompact). See the design notes in the extension for
|
|
2532
|
+
* what each layer does; this module is purely the env → config surface.
|
|
2533
|
+
*
|
|
2534
|
+
* The harness is provider-agnostic and runs inside the sandbox, where its
|
|
2535
|
+
* only config channel is the environment (loaded by the platform env
|
|
2536
|
+
* middleware before any session starts — see platform-env-middleware.ts).
|
|
2537
|
+
* `SKYDIVE_CONTEXT_MANAGEMENT` is the env-side master toggle; the individual
|
|
2538
|
+
* `SKYDIVE_CTX_*` knobs override the defaults below only when that toggle is
|
|
2539
|
+
* set. Prod never sets it: the LaunchDarkly flag
|
|
2540
|
+
* `harness-context-management-enabled` reaches running sandboxes through the
|
|
2541
|
+
* feature-flag poller (context-management-runtime.ts), which overlays only
|
|
2542
|
+
* `enabled`. So in prod every knob resolves to its default regardless of what
|
|
2543
|
+
* the platform forwards — see the knob-gating follow-up before relying on one.
|
|
2544
|
+
*
|
|
2545
|
+
* Resolution is defensive: a malformed value never throws (this config is
|
|
2546
|
+
* read on the hot path before every LLM call), it falls back to the default
|
|
2547
|
+
* for that knob and logs once.
|
|
2548
|
+
*/
|
|
2549
|
+
const log$11 = logger.child({ module: "context-management-config" });
|
|
2550
|
+
const DEFAULT_CONTEXT_MANAGEMENT_CONFIG = {
|
|
2551
|
+
enabled: false,
|
|
2552
|
+
perResultMaxBytes: 16 * 1024,
|
|
2553
|
+
keepRecentToolResults: 3,
|
|
2554
|
+
coldCacheGapSeconds: 240,
|
|
2555
|
+
warmClearTriggerTokens: 5e4,
|
|
2556
|
+
clearAtLeastTokens: 1e4,
|
|
2557
|
+
excludeTools: [],
|
|
2558
|
+
nativeAnthropicEdits: false
|
|
2559
|
+
};
|
|
2560
|
+
const boolFromEnv = (value, fallback) => {
|
|
2561
|
+
if (value === void 0) return fallback;
|
|
2562
|
+
const normalized = value.trim().toLowerCase();
|
|
2563
|
+
if (normalized === "1" || normalized === "true") return true;
|
|
2564
|
+
if (normalized === "0" || normalized === "false") return false;
|
|
2565
|
+
return fallback;
|
|
2566
|
+
};
|
|
2567
|
+
const positiveInt = (fallback) => z.coerce.number().int().positive().catch(fallback);
|
|
2568
|
+
const nonNegativeInt = (fallback) => z.coerce.number().int().nonnegative().catch(fallback);
|
|
2569
|
+
const configSchema = z.object({
|
|
2570
|
+
perResultMaxBytes: positiveInt(DEFAULT_CONTEXT_MANAGEMENT_CONFIG.perResultMaxBytes),
|
|
2571
|
+
keepRecentToolResults: nonNegativeInt(DEFAULT_CONTEXT_MANAGEMENT_CONFIG.keepRecentToolResults),
|
|
2572
|
+
coldCacheGapSeconds: positiveInt(DEFAULT_CONTEXT_MANAGEMENT_CONFIG.coldCacheGapSeconds),
|
|
2573
|
+
warmClearTriggerTokens: positiveInt(DEFAULT_CONTEXT_MANAGEMENT_CONFIG.warmClearTriggerTokens),
|
|
2574
|
+
clearAtLeastTokens: positiveInt(DEFAULT_CONTEXT_MANAGEMENT_CONFIG.clearAtLeastTokens)
|
|
2575
|
+
});
|
|
2576
|
+
const parseExcludeTools = (value) => {
|
|
2577
|
+
if (!value) return DEFAULT_CONTEXT_MANAGEMENT_CONFIG.excludeTools;
|
|
2578
|
+
return value.split(",").map((name) => name.trim()).filter((name) => name.length > 0);
|
|
2579
|
+
};
|
|
2580
|
+
/**
|
|
2581
|
+
* Reads the context-management config from `env` (defaults to `process.env`).
|
|
2582
|
+
* Never throws — invalid values fall back to defaults.
|
|
2583
|
+
*/
|
|
2584
|
+
function resolveContextManagementConfig(env = process.env) {
|
|
2585
|
+
if (!boolFromEnv(env.SKYDIVE_CONTEXT_MANAGEMENT, false)) return { ...DEFAULT_CONTEXT_MANAGEMENT_CONFIG };
|
|
2586
|
+
const parsed = configSchema.safeParse({
|
|
2587
|
+
perResultMaxBytes: env.SKYDIVE_CTX_PER_RESULT_MAX_BYTES,
|
|
2588
|
+
keepRecentToolResults: env.SKYDIVE_CTX_KEEP_RECENT,
|
|
2589
|
+
coldCacheGapSeconds: env.SKYDIVE_CTX_COLD_GAP_SECONDS,
|
|
2590
|
+
warmClearTriggerTokens: env.SKYDIVE_CTX_WARM_TRIGGER_TOKENS,
|
|
2591
|
+
clearAtLeastTokens: env.SKYDIVE_CTX_CLEAR_AT_LEAST_TOKENS
|
|
2592
|
+
});
|
|
2593
|
+
if (!parsed.success) {
|
|
2594
|
+
log$11.warn({
|
|
2595
|
+
event: "context_management_config_invalid",
|
|
2596
|
+
err: parsed.error
|
|
2597
|
+
}, "falling back to default context-management config");
|
|
2598
|
+
return {
|
|
2599
|
+
...DEFAULT_CONTEXT_MANAGEMENT_CONFIG,
|
|
2600
|
+
enabled: true
|
|
2601
|
+
};
|
|
2602
|
+
}
|
|
2603
|
+
return {
|
|
2604
|
+
enabled: true,
|
|
2605
|
+
...parsed.data,
|
|
2606
|
+
excludeTools: parseExcludeTools(env.SKYDIVE_CTX_EXCLUDE_TOOLS),
|
|
2607
|
+
nativeAnthropicEdits: boolFromEnv(env.SKYDIVE_CTX_NATIVE_ANTHROPIC_EDITS, DEFAULT_CONTEXT_MANAGEMENT_CONFIG.nativeAnthropicEdits)
|
|
2608
|
+
};
|
|
2609
|
+
}
|
|
2610
|
+
//#endregion
|
|
2611
|
+
//#region src/extensions/context-management-runtime.ts
|
|
2612
|
+
/**
|
|
2613
|
+
* Live, mutable view of the context-management config.
|
|
2614
|
+
*
|
|
2615
|
+
* The env-derived config (context-management-config.ts) is the boot-time
|
|
2616
|
+
* default. On top of it, the platform can deliver a global on/off at runtime
|
|
2617
|
+
* via the LaunchDarkly flag `harness-context-management-enabled` — resolved
|
|
2618
|
+
* server-side and polled by the harness (see the poller in
|
|
2619
|
+
* context-management.ts). This holder is where that override lands so a flip
|
|
2620
|
+
* (especially a kill-switch) reaches long-lived sandboxes without a restart.
|
|
2621
|
+
*
|
|
2622
|
+
* Only the master `enabled` toggle is overridable at runtime; the per-knob
|
|
2623
|
+
* tunables stay env-derived. `enabled` resolves to the flag override when the
|
|
2624
|
+
* platform has reported one, else the env value.
|
|
2625
|
+
*/
|
|
2626
|
+
let baseConfig = null;
|
|
2627
|
+
let flagOverride = null;
|
|
2628
|
+
function base() {
|
|
2629
|
+
baseConfig ??= resolveContextManagementConfig();
|
|
2630
|
+
return baseConfig;
|
|
2631
|
+
}
|
|
2632
|
+
/** The effective config, with the runtime flag override applied to `enabled`. */
|
|
2633
|
+
function getContextManagementConfig() {
|
|
2634
|
+
const resolved = base();
|
|
2635
|
+
return {
|
|
2636
|
+
...resolved,
|
|
2637
|
+
enabled: flagOverride ?? resolved.enabled
|
|
2638
|
+
};
|
|
2639
|
+
}
|
|
2640
|
+
/**
|
|
2641
|
+
* Apply the platform-reported flag value. `null` clears the override (fall back
|
|
2642
|
+
* to the env default) — used when the phone-home result is indeterminate so a
|
|
2643
|
+
* transient failure never silently changes behavior.
|
|
2644
|
+
*/
|
|
2645
|
+
function setContextManagementFlagOverride(enabled) {
|
|
2646
|
+
flagOverride = enabled;
|
|
2647
|
+
}
|
|
2648
|
+
/**
|
|
2649
|
+
* Whether a phone-home channel exists to learn the flag at runtime. When false
|
|
2650
|
+
* (e.g. a bare local CLI with no platform API), the env value is the only
|
|
2651
|
+
* source and there's nothing to poll.
|
|
2652
|
+
*/
|
|
2653
|
+
function hasFlagSource(env = process.env) {
|
|
2654
|
+
return Boolean(env.SKYDIVE_API_URL ?? env.ANYONE_API_URL);
|
|
2655
|
+
}
|
|
2656
|
+
//#endregion
|
|
1916
2657
|
//#region src/extensions/feature-flags-poll.ts
|
|
1917
2658
|
/**
|
|
1918
|
-
* Shared harness feature-flag poll.
|
|
2659
|
+
* Shared harness feature-flag poll.
|
|
2660
|
+
*
|
|
2661
|
+
* The api exposes one `/feature-flags` GET that returns every harness flag in a
|
|
2662
|
+
* single response (`{ contextManagement, closingText, commandFlags }` — see
|
|
2663
|
+
* apps/anyone/api/src/routes/sandbox-feature-flags.ts). Rather than each
|
|
2664
|
+
* extension issuing its own GET — and, worse, a *blocking* GET on the
|
|
2665
|
+
* pre-first-token `session_start` path — a single background poller fetches
|
|
2666
|
+
* that response once per interval and fans the values out to every subscriber.
|
|
2667
|
+
*
|
|
2668
|
+
* Why one poller: context-management consumes the `contextManagement` flag and
|
|
2669
|
+
* the closing-text loop consumes `closingText`, both without a blocking GET on
|
|
2670
|
+
* the pre-first-token `session_start` path. Reading the last-polled value keeps
|
|
2671
|
+
* the hot path allocation-only; a cold cache reads as `null` and a
|
|
2672
|
+
* newly-flipped flag takes effect on the next poll.
|
|
2673
|
+
*
|
|
2674
|
+
* The poll is fire-and-forget and self-unref'd — it never keeps the process
|
|
2675
|
+
* alive and an indeterminate result (no api url / transient failure) leaves the
|
|
2676
|
+
* last-known values untouched so a blip can't silently flip behavior.
|
|
2677
|
+
*/
|
|
2678
|
+
const log$10 = logger.child({ module: "feature-flags-poll" });
|
|
2679
|
+
const FLAG_POLL_INTERVAL_MS = 6e4;
|
|
2680
|
+
let contextManagement = null;
|
|
2681
|
+
let closingText = null;
|
|
2682
|
+
const subscribers = {
|
|
2683
|
+
contextManagement: /* @__PURE__ */ new Set(),
|
|
2684
|
+
closingText: /* @__PURE__ */ new Set()
|
|
2685
|
+
};
|
|
2686
|
+
let pollerStarted = false;
|
|
2687
|
+
let firstPollSettled = false;
|
|
2688
|
+
let resolveFirstPoll = null;
|
|
2689
|
+
new Promise((resolve) => {
|
|
2690
|
+
resolveFirstPoll = resolve;
|
|
2691
|
+
});
|
|
2692
|
+
function markFirstPollSettled() {
|
|
2693
|
+
if (firstPollSettled) return;
|
|
2694
|
+
firstPollSettled = true;
|
|
2695
|
+
resolveFirstPoll?.();
|
|
2696
|
+
}
|
|
2697
|
+
/** Last-polled value of a flag, or `null` if not yet resolved. */
|
|
2698
|
+
function getPolledFlag(name) {
|
|
2699
|
+
return name === "closingText" ? closingText : contextManagement;
|
|
2700
|
+
}
|
|
2701
|
+
/**
|
|
2702
|
+
* Subscribe to changes of a flag. The callback fires only on a *transition*
|
|
2703
|
+
* (skipped while the value is unchanged), so a subscriber registered before the
|
|
2704
|
+
* first poll still learns the initial value. Returns an unsubscribe fn.
|
|
2705
|
+
*/
|
|
2706
|
+
function onFlagChange(name, cb) {
|
|
2707
|
+
subscribers[name].add(cb);
|
|
2708
|
+
return () => subscribers[name].delete(cb);
|
|
2709
|
+
}
|
|
2710
|
+
function apply(name, next) {
|
|
2711
|
+
if (next === null) return;
|
|
2712
|
+
const prev = name === "closingText" ? closingText : contextManagement;
|
|
2713
|
+
if (name === "closingText") closingText = next;
|
|
2714
|
+
else contextManagement = next;
|
|
2715
|
+
if (next !== prev) for (const cb of subscribers[name]) try {
|
|
2716
|
+
cb(next);
|
|
2717
|
+
} catch (err) {
|
|
2718
|
+
log$10.warn({
|
|
2719
|
+
err,
|
|
2720
|
+
flag: name
|
|
2721
|
+
}, "flag subscriber threw");
|
|
2722
|
+
}
|
|
2723
|
+
}
|
|
2724
|
+
async function pollOnce() {
|
|
2725
|
+
try {
|
|
2726
|
+
const flags = await fetchHarnessFlags();
|
|
2727
|
+
if (!flags) return;
|
|
2728
|
+
apply("contextManagement", flags.contextManagement ?? null);
|
|
2729
|
+
apply("closingText", flags.closingText ?? null);
|
|
2730
|
+
} catch (err) {
|
|
2731
|
+
log$10.debug({ err }, "feature-flag poll threw");
|
|
2732
|
+
}
|
|
2733
|
+
}
|
|
2734
|
+
/**
|
|
2735
|
+
* Start the shared background poll (idempotent). No-op when there's no
|
|
2736
|
+
* phone-home channel (bare CLI): there's nothing to poll and callers keep their
|
|
2737
|
+
* env/boot default. Kicks an immediate poll, then repeats on an interval that
|
|
2738
|
+
* does not keep the process alive.
|
|
2739
|
+
*/
|
|
2740
|
+
function startFeatureFlagPoller() {
|
|
2741
|
+
if (pollerStarted || !hasFlagSource()) return;
|
|
2742
|
+
pollerStarted = true;
|
|
2743
|
+
pollOnce().finally(markFirstPollSettled);
|
|
2744
|
+
setInterval(() => void pollOnce(), FLAG_POLL_INTERVAL_MS).unref?.();
|
|
2745
|
+
}
|
|
2746
|
+
//#endregion
|
|
2747
|
+
//#region src/closing-text-loop.ts
|
|
2748
|
+
const CLOSING_TEXT_NUDGE = "<system_notification>You ended that turn without writing any reply to the user. Write a brief reply now: what you did and the outcome, or the answer to their question. Do not call any more tools unless you genuinely still need to.</system_notification>";
|
|
2749
|
+
function isNonEmptyText(block) {
|
|
2750
|
+
return block.type === "text" && typeof block.text === "string" && block.text.trim().length > 0;
|
|
2751
|
+
}
|
|
2752
|
+
/**
|
|
2753
|
+
* True when the assistant produced no visible text since the human's last
|
|
2754
|
+
* message this turn — i.e. the turn ended on tool calls / thinking only.
|
|
2755
|
+
*
|
|
2756
|
+
* We scan from the end back to the most recent `user` message (the human turn
|
|
2757
|
+
* that triggered this run) and inspect every `assistant` message after it.
|
|
2758
|
+
* A string-content assistant message (rare, but pi allows it) counts as text
|
|
2759
|
+
* when non-blank. `toolResult` and custom messages are ignored.
|
|
2760
|
+
*/
|
|
2761
|
+
function turnEndedWithoutVisibleText(messages) {
|
|
2762
|
+
let sawAssistant = false;
|
|
2763
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
2764
|
+
const m = messages[i];
|
|
2765
|
+
if (!m || m.role === "user") break;
|
|
2766
|
+
if (m.role !== "assistant") continue;
|
|
2767
|
+
sawAssistant = true;
|
|
2768
|
+
if (typeof m.content === "string") {
|
|
2769
|
+
if (m.content.trim().length > 0) return false;
|
|
2770
|
+
continue;
|
|
2771
|
+
}
|
|
2772
|
+
if (Array.isArray(m.content) && m.content.some(isNonEmptyText)) return false;
|
|
2773
|
+
}
|
|
2774
|
+
return sawAssistant;
|
|
2775
|
+
}
|
|
2776
|
+
/**
|
|
2777
|
+
* True when the last assistant message this turn ended in a way we must not
|
|
2778
|
+
* nudge: a user abort or a provider error. Both already surface their own
|
|
2779
|
+
* terminal signal to the client; a closing-text prompt on top would be noise.
|
|
2780
|
+
*/
|
|
2781
|
+
function turnEndedAbnormally(messages) {
|
|
2782
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
2783
|
+
const m = messages[i];
|
|
2784
|
+
if (!m || m.role !== "assistant") continue;
|
|
2785
|
+
return m.stopReason === "aborted" || m.stopReason === "error";
|
|
2786
|
+
}
|
|
2787
|
+
return false;
|
|
2788
|
+
}
|
|
2789
|
+
async function runClosingTextLoop({ session, log, enabled = getPolledFlag("closingText") === true }) {
|
|
2790
|
+
if (!enabled) return;
|
|
2791
|
+
if (turnEndedAbnormally(session.messages)) return;
|
|
2792
|
+
if (!turnEndedWithoutVisibleText(session.messages)) return;
|
|
2793
|
+
log.info({ event: "closing_text_continuation_injected" }, "turn produced no visible reply; injecting a closing-text continuation");
|
|
2794
|
+
await session.sendCustomMessage({
|
|
2795
|
+
customType: "closing-text-nudge",
|
|
2796
|
+
content: CLOSING_TEXT_NUDGE,
|
|
2797
|
+
display: false,
|
|
2798
|
+
details: {}
|
|
2799
|
+
}, { triggerTurn: true });
|
|
2800
|
+
}
|
|
2801
|
+
//#endregion
|
|
2802
|
+
//#region src/tracing.ts
|
|
2803
|
+
const SERVICE_NAME = "skydive-agent-harness";
|
|
2804
|
+
const DAEMON_TRACES_URL = "http://localhost:38994/v1/traces";
|
|
2805
|
+
let provider = null;
|
|
2806
|
+
function initTracing() {
|
|
2807
|
+
if (provider) return;
|
|
2808
|
+
propagation.setGlobalPropagator(new W3CTraceContextPropagator());
|
|
2809
|
+
provider = new NodeTracerProvider({ resource: new Resource({ [ATTR_SERVICE_NAME]: SERVICE_NAME }) });
|
|
2810
|
+
const exporter = new OTLPTraceExporter({ url: DAEMON_TRACES_URL });
|
|
2811
|
+
provider.addSpanProcessor(new BatchSpanProcessor(exporter, { scheduledDelayMillis: 1e3 }));
|
|
2812
|
+
provider.register();
|
|
2813
|
+
logger.info({
|
|
2814
|
+
event: "tracing_enabled",
|
|
2815
|
+
endpoint: DAEMON_TRACES_URL
|
|
2816
|
+
}, "OTel tracing initialized — exporting via daemon");
|
|
2817
|
+
const shutdown = async () => {
|
|
2818
|
+
await shutdownTracing();
|
|
2819
|
+
process.exit(0);
|
|
2820
|
+
};
|
|
2821
|
+
process.on("SIGTERM", shutdown);
|
|
2822
|
+
process.on("SIGINT", shutdown);
|
|
2823
|
+
}
|
|
2824
|
+
function getTracer() {
|
|
2825
|
+
return trace.getTracer(SERVICE_NAME);
|
|
2826
|
+
}
|
|
2827
|
+
function extractRemoteContext() {
|
|
2828
|
+
const traceparent = getCurrentTraceparent();
|
|
2829
|
+
if (!traceparent) return ROOT_CONTEXT;
|
|
2830
|
+
const carrier = { traceparent };
|
|
2831
|
+
return propagation.extract(ROOT_CONTEXT, carrier, {
|
|
2832
|
+
get: (c, key) => c[key],
|
|
2833
|
+
keys: (c) => Object.keys(c)
|
|
2834
|
+
});
|
|
2835
|
+
}
|
|
2836
|
+
async function shutdownTracing() {
|
|
2837
|
+
if (provider) await provider.shutdown();
|
|
2838
|
+
}
|
|
2839
|
+
//#endregion
|
|
2840
|
+
//#region src/harness.ts
|
|
2841
|
+
/**
|
|
2842
|
+
* Skydive composition over @skydiveai/pi-server: wires the platform
|
|
2843
|
+
* defaults (tracing to the daemon, tool-update hot-reload hooks, prewarm
|
|
2844
|
+
* paths, header passthrough for proxy routing, Skydive agent-card
|
|
2845
|
+
* branding) into the generic protocol server, and returns mountable
|
|
2846
|
+
* handlers. The agent supplies the pi session factory (their session.ts)
|
|
2847
|
+
* and owns the express app:
|
|
2848
|
+
*
|
|
2849
|
+
* const { platform, protocols } = createHarness({
|
|
2850
|
+
* cwd: process.cwd(),
|
|
2851
|
+
* createSession,
|
|
2852
|
+
* });
|
|
2853
|
+
* app.get('/health', platform.handlers.health);
|
|
2854
|
+
* app.use(platform.handlers.injectEnv);
|
|
2855
|
+
* app.use(platform.handlers.prewarm); // matches /_skydive/prewarm + legacy alias
|
|
2856
|
+
* app.use(protocols.handlers.all);
|
|
2857
|
+
*/
|
|
2858
|
+
const A2A_PATH = "/a2a";
|
|
2859
|
+
const AGENT_CARD_PATH = "/.well-known/agent-card.json";
|
|
2860
|
+
const PREWARM_PATHS = ["/_skydive/prewarm", "/_anyone/prewarm"];
|
|
2861
|
+
const PASSTHROUGH_HEADER_PREFIXES = ["x-anyone-", "x-skydive-"];
|
|
2862
|
+
function createHarness(options) {
|
|
2863
|
+
initTracing();
|
|
2864
|
+
startFeatureFlagPoller();
|
|
2865
|
+
readPlatformVersions().then((runtimeVersions) => logger.info({
|
|
2866
|
+
event: "runtime_versions",
|
|
2867
|
+
runtimeVersions
|
|
2868
|
+
}, "platform runtime versions"));
|
|
2869
|
+
const serverOptions = {
|
|
2870
|
+
...options,
|
|
2871
|
+
onSessionSetup: options.onSessionSetup ?? installToolUpdateAutoStop,
|
|
2872
|
+
postPrompt: options.postPrompt ?? (async (args) => {
|
|
2873
|
+
await runToolUpdateLoop(args);
|
|
2874
|
+
await runClosingTextLoop(args);
|
|
2875
|
+
}),
|
|
2876
|
+
passthroughHeaderPrefixes: options.passthroughHeaderPrefixes ?? PASSTHROUGH_HEADER_PREFIXES,
|
|
2877
|
+
prewarmPaths: options.prewarmPaths ?? PREWARM_PATHS
|
|
2878
|
+
};
|
|
2879
|
+
const webHandlers = createProtocolHandlers(serverOptions);
|
|
2880
|
+
const cardOverrides = {
|
|
2881
|
+
name: "Skydive Agent",
|
|
2882
|
+
description: "An AI coding agent powered by the Skydive platform.",
|
|
2883
|
+
...options.agentCard
|
|
2884
|
+
};
|
|
2885
|
+
const agentCardWeb = async (request) => {
|
|
2886
|
+
const url = new URL(request.url);
|
|
2887
|
+
if (request.method !== "GET" || url.pathname !== AGENT_CARD_PATH) return null;
|
|
2888
|
+
return Response.json(buildAgentCard(cardOverrides.url ?? `${url.origin}${A2A_PATH}`, cardOverrides));
|
|
2889
|
+
};
|
|
2890
|
+
const a2a = restHandler({
|
|
2891
|
+
requestHandler: new DefaultRequestHandler(buildAgentCard(cardOverrides.url ?? A2A_PATH, cardOverrides), new InMemoryTaskStore(), createAgentExecutor(serverOptions), new DefaultExecutionEventBusManager()),
|
|
2892
|
+
userBuilder: UserBuilder.noAuthentication
|
|
2893
|
+
});
|
|
2894
|
+
return {
|
|
2895
|
+
platform: { handlers: {
|
|
2896
|
+
/**
|
|
2897
|
+
* Platform health plus the agent's `healthMetadata`. Mount above
|
|
2898
|
+
* injectEnv — health must respond immediately for prewarm
|
|
2899
|
+
* stashing and readiness probes, and injectEnv can wait up to
|
|
2900
|
+
* 10s for env vars during boot.
|
|
2901
|
+
*/
|
|
2902
|
+
health: createHealthHandler({ metadata: options.healthMetadata ?? null }),
|
|
2903
|
+
/** Loads platform env (e2b envd / daemon long-poll). */
|
|
2904
|
+
injectEnv: createPlatformEnvMiddleware(),
|
|
2905
|
+
prewarm: webHandlerToMiddleware(webHandlers.prewarm)
|
|
2906
|
+
} },
|
|
2907
|
+
protocols: { handlers: {
|
|
2908
|
+
/** Express-style; mount at /a2a. */
|
|
2909
|
+
a2a,
|
|
2910
|
+
/** GET /.well-known/agent-card.json. */
|
|
2911
|
+
agentCard: webHandlerToMiddleware(agentCardWeb),
|
|
2912
|
+
/** Mirrors each vendor's API shape. */
|
|
2913
|
+
openai: { v1: {
|
|
2914
|
+
chat: { completions: webHandlerToMiddleware(webHandlers.chatCompletions) },
|
|
2915
|
+
responses: webHandlerToMiddleware(webHandlers.responses)
|
|
2916
|
+
} },
|
|
2917
|
+
anthropic: { v1: { messages: webHandlerToMiddleware(webHandlers.messages) } },
|
|
2918
|
+
/**
|
|
2919
|
+
* Everything in one mount: a2a (+ agent card) at their well-known
|
|
2920
|
+
* paths, then chat-completions / anthropic-messages / responses.
|
|
2921
|
+
* Calls next() when nothing matches.
|
|
2922
|
+
*/
|
|
2923
|
+
all: chainMiddleware([mountAt(A2A_PATH, a2a), webHandlerToMiddleware(composeHandlers([
|
|
2924
|
+
agentCardWeb,
|
|
2925
|
+
webHandlers.chatCompletions,
|
|
2926
|
+
webHandlers.messages,
|
|
2927
|
+
webHandlers.responses
|
|
2928
|
+
]))])
|
|
2929
|
+
} }
|
|
2930
|
+
};
|
|
2931
|
+
}
|
|
2932
|
+
/** Effective bash timeout: the model's value when it gave a positive number, else the default. */
|
|
2933
|
+
function resolveBashTimeout(provided) {
|
|
2934
|
+
return typeof provided === "number" && provided > 0 ? provided : 600;
|
|
2935
|
+
}
|
|
2936
|
+
const bashDefaultTimeoutExtension = (pi) => {
|
|
2937
|
+
pi.on("tool_call", async (event) => {
|
|
2938
|
+
if (event.toolName !== "bash") return;
|
|
2939
|
+
event.input.timeout = resolveBashTimeout(event.input.timeout);
|
|
2940
|
+
});
|
|
2941
|
+
};
|
|
2942
|
+
//#endregion
|
|
2943
|
+
//#region src/extensions/resource-pressure-warning.ts
|
|
2944
|
+
/**
|
|
2945
|
+
* Mid-run resource-pressure warning to the agent.
|
|
1919
2946
|
*
|
|
1920
|
-
* The
|
|
1921
|
-
*
|
|
1922
|
-
*
|
|
1923
|
-
*
|
|
1924
|
-
*
|
|
1925
|
-
*
|
|
2947
|
+
* The sandbox already detects pressure — the boot scripts cap the
|
|
2948
|
+
* user-workload cgroup (memory.high/memory.max) and watchers log warn/crit
|
|
2949
|
+
* edges for memory and disk — but nothing told the *agent*, so a turn burned
|
|
2950
|
+
* straight to the OOM kill (or a full disk) and only learned about it from
|
|
2951
|
+
* the post-mortem notice. This extension closes that gap in-process: while a
|
|
2952
|
+
* turn is active it polls the agent cgroup and the root filesystem and, the
|
|
2953
|
+
* first time usage crosses a warn threshold, folds a system notification into
|
|
2954
|
+
* the open turn so the agent can checkpoint, shed work (constrain
|
|
2955
|
+
* parallelism, kill a background hog, clean scratch space), or request a
|
|
2956
|
+
* bigger tier BEFORE the kill. The warning is model-facing only; the
|
|
2957
|
+
* agent must not narrate it to the user unless asked.
|
|
1926
2958
|
*
|
|
1927
|
-
*
|
|
1928
|
-
*
|
|
1929
|
-
*
|
|
1930
|
-
*
|
|
2959
|
+
* The notification is triggered by the two conditions that actually kill
|
|
2960
|
+
* work — memory near the cgroup hard cap, disk near full — and reports a
|
|
2961
|
+
* snapshot of all the relevant stats (memory, CPU utilization, disk) so the
|
|
2962
|
+
* agent can tell which resource is the problem and how much headroom the
|
|
2963
|
+
* others have.
|
|
1931
2964
|
*
|
|
1932
|
-
*
|
|
1933
|
-
*
|
|
1934
|
-
*
|
|
2965
|
+
* Edge-triggered, once per trigger per turn: the fired flags reset on
|
|
2966
|
+
* agent_start, so a turn that rides a threshold gets one warning per
|
|
2967
|
+
* resource, not a stream. Polling only runs while the agent is active — an
|
|
2968
|
+
* idle sandbox's resource usage is not the agent's problem and there is no
|
|
2969
|
+
* open turn to deliver into anyway.
|
|
2970
|
+
*
|
|
2971
|
+
* Best-effort throughout: any read failure (cgroup absent, controller not
|
|
2972
|
+
* delegated, non-cgroup-v2 host, df missing) reads as "no signal" for that
|
|
2973
|
+
* stat and the extension warns on what it can see — it must never break a
|
|
2974
|
+
* turn over an observability feature.
|
|
1935
2975
|
*/
|
|
1936
|
-
const
|
|
1937
|
-
const
|
|
1938
|
-
|
|
1939
|
-
|
|
1940
|
-
|
|
1941
|
-
|
|
1942
|
-
|
|
1943
|
-
|
|
1944
|
-
|
|
1945
|
-
});
|
|
1946
|
-
function markFirstPollSettled() {
|
|
1947
|
-
if (firstPollSettled) return;
|
|
1948
|
-
firstPollSettled = true;
|
|
1949
|
-
resolveFirstPoll?.();
|
|
2976
|
+
const execFileAsync = promisify(execFile);
|
|
2977
|
+
const log$9 = logger.child({ module: "resource-pressure-warning" });
|
|
2978
|
+
const POLL_INTERVAL_MS = 1e4;
|
|
2979
|
+
function envOverride(name) {
|
|
2980
|
+
for (const prefix of ["SKYDIVE_", "ANYONE_"]) {
|
|
2981
|
+
const value = process.env[`${prefix}${name}`];
|
|
2982
|
+
if (value != null && value !== "") return value;
|
|
2983
|
+
}
|
|
2984
|
+
return null;
|
|
1950
2985
|
}
|
|
1951
|
-
|
|
1952
|
-
|
|
1953
|
-
|
|
2986
|
+
function cgroupDir() {
|
|
2987
|
+
return envOverride("AGENT_CGROUP") ?? "/sys/fs/cgroup/agent";
|
|
2988
|
+
}
|
|
2989
|
+
function diskRoot() {
|
|
2990
|
+
return envOverride("DISK_ROOT") ?? "/";
|
|
1954
2991
|
}
|
|
1955
2992
|
/**
|
|
1956
|
-
*
|
|
1957
|
-
* (
|
|
1958
|
-
*
|
|
2993
|
+
* Read a cgroup v2 scalar file. Returns a number, or null for "max"
|
|
2994
|
+
* (uncapped), an empty/absent file, or any read/parse error — an uncapped or
|
|
2995
|
+
* unreadable limit means there is nothing meaningful to warn against.
|
|
1959
2996
|
*/
|
|
1960
|
-
function
|
|
1961
|
-
|
|
1962
|
-
|
|
2997
|
+
async function readScalar(file) {
|
|
2998
|
+
try {
|
|
2999
|
+
const raw = (await readFile(`${cgroupDir()}/${file}`, "utf8")).trim();
|
|
3000
|
+
if (raw === "" || raw === "max") return null;
|
|
3001
|
+
const n = Number(raw);
|
|
3002
|
+
return Number.isFinite(n) ? n : null;
|
|
3003
|
+
} catch (_error) {
|
|
3004
|
+
return null;
|
|
3005
|
+
}
|
|
1963
3006
|
}
|
|
1964
|
-
|
|
1965
|
-
|
|
1966
|
-
|
|
1967
|
-
|
|
1968
|
-
|
|
1969
|
-
|
|
1970
|
-
|
|
1971
|
-
|
|
1972
|
-
|
|
1973
|
-
|
|
1974
|
-
|
|
3007
|
+
/**
|
|
3008
|
+
* Read a cgroup v2 "flat keyed" file (one `key value` pair per line, e.g.
|
|
3009
|
+
* cpu.stat) and return the counter for `key`, or null when absent.
|
|
3010
|
+
*/
|
|
3011
|
+
async function readKeyedCounter(file, key) {
|
|
3012
|
+
try {
|
|
3013
|
+
const raw = await readFile(`${cgroupDir()}/${file}`, "utf8");
|
|
3014
|
+
for (const line of raw.split("\n")) {
|
|
3015
|
+
const [k, v] = line.trim().split(/\s+/);
|
|
3016
|
+
if (k === key) {
|
|
3017
|
+
const n = Number(v);
|
|
3018
|
+
return Number.isFinite(n) ? n : null;
|
|
3019
|
+
}
|
|
3020
|
+
}
|
|
3021
|
+
return null;
|
|
3022
|
+
} catch (_error) {
|
|
3023
|
+
return null;
|
|
1975
3024
|
}
|
|
1976
3025
|
}
|
|
1977
|
-
|
|
3026
|
+
/**
|
|
3027
|
+
* Live memory usage as an integer percent of the hard cap, or null when
|
|
3028
|
+
* either side is unreadable/uncapped. Exported for tests.
|
|
3029
|
+
*/
|
|
3030
|
+
async function readMemUsePct() {
|
|
3031
|
+
const [current, max] = await Promise.all([readScalar("memory.current"), readScalar("memory.max")]);
|
|
3032
|
+
if (current === null || max === null || max <= 0) return null;
|
|
3033
|
+
return {
|
|
3034
|
+
pct: Math.floor(current / max * 100),
|
|
3035
|
+
currentBytes: current,
|
|
3036
|
+
maxBytes: max
|
|
3037
|
+
};
|
|
3038
|
+
}
|
|
3039
|
+
/**
|
|
3040
|
+
* Root filesystem used% (df -P Capacity column), or null on any failure.
|
|
3041
|
+
* Exported for tests.
|
|
3042
|
+
*/
|
|
3043
|
+
async function readDiskUsePct() {
|
|
1978
3044
|
try {
|
|
1979
|
-
const
|
|
1980
|
-
|
|
1981
|
-
|
|
1982
|
-
|
|
1983
|
-
|
|
3045
|
+
const { stdout } = await execFileAsync("df", ["-P", diskRoot()]);
|
|
3046
|
+
const dataRow = stdout.trim().split("\n")[1];
|
|
3047
|
+
if (dataRow == null) return null;
|
|
3048
|
+
const capacity = dataRow.trim().split(/\s+/)[4];
|
|
3049
|
+
if (capacity == null) return null;
|
|
3050
|
+
const pct = Number(capacity.replace("%", ""));
|
|
3051
|
+
return Number.isFinite(pct) ? pct : null;
|
|
3052
|
+
} catch (_error) {
|
|
3053
|
+
return null;
|
|
1984
3054
|
}
|
|
1985
3055
|
}
|
|
1986
3056
|
/**
|
|
1987
|
-
*
|
|
1988
|
-
*
|
|
1989
|
-
*
|
|
1990
|
-
*
|
|
3057
|
+
* CPU utilization sampler. cgroup v2 exposes cumulative CPU time
|
|
3058
|
+
* (cpu.stat usage_usec); utilization is the delta between two samples over
|
|
3059
|
+
* the wall time between them, normalized by core count. The first call after
|
|
3060
|
+
* construction has no previous sample and returns null.
|
|
1991
3061
|
*/
|
|
1992
|
-
function
|
|
1993
|
-
|
|
1994
|
-
|
|
1995
|
-
|
|
1996
|
-
|
|
3062
|
+
function createCpuSampler() {
|
|
3063
|
+
let prevUsageUsec = null;
|
|
3064
|
+
let prevAtMs = null;
|
|
3065
|
+
return async () => {
|
|
3066
|
+
const usage = await readKeyedCounter("cpu.stat", "usage_usec");
|
|
3067
|
+
const now = Date.now();
|
|
3068
|
+
const prev = prevUsageUsec;
|
|
3069
|
+
const prevAt = prevAtMs;
|
|
3070
|
+
prevUsageUsec = usage;
|
|
3071
|
+
prevAtMs = now;
|
|
3072
|
+
if (usage === null || prev === null || prevAt === null) return null;
|
|
3073
|
+
const wallUsec = (now - prevAt) * 1e3;
|
|
3074
|
+
if (wallUsec <= 0) return null;
|
|
3075
|
+
const cores = availableParallelism();
|
|
3076
|
+
const pct = Math.round((usage - prev) / (wallUsec * cores) * 100);
|
|
3077
|
+
return Math.max(0, Math.min(100, pct));
|
|
3078
|
+
};
|
|
3079
|
+
}
|
|
3080
|
+
function fmtMb(bytes) {
|
|
3081
|
+
return Math.round(bytes / 1024 / 1024);
|
|
3082
|
+
}
|
|
3083
|
+
/** The model-facing warning text. Exported for tests. */
|
|
3084
|
+
function resourcePressureWarningText(trigger, { mem, cpuPct, diskPct }) {
|
|
3085
|
+
const stats = [];
|
|
3086
|
+
if (mem) stats.push(`memory ${mem.pct}% of cap (${fmtMb(mem.currentBytes)}/${fmtMb(mem.maxBytes)} MB)`);
|
|
3087
|
+
if (cpuPct !== null) stats.push(`CPU ${cpuPct}%`);
|
|
3088
|
+
if (diskPct !== null) stats.push(`disk ${diskPct}% full`);
|
|
3089
|
+
const lead = trigger === "memory" ? `Your sandbox is at ${mem?.pct}% of its memory cap. If usage keeps climbing, the kernel will kill the offending process and this turn may die with it.` : `Your sandbox's disk is ${diskPct}% full. If it fills completely, writes will start failing and this turn may die with them.`;
|
|
3090
|
+
const remedy = trigger === "memory" ? "checkpoint in-flight work (commit and push), then reduce the footprint — constrain parallelism, run heavy steps sequentially, or kill background processes you no longer need." : "checkpoint in-flight work (commit and push), then free space — clean build artifacts, caches, and scratch files you no longer need.";
|
|
3091
|
+
const closer = trigger === "memory" ? "If the workload genuinely needs more memory or CPU, request a bigger sandbox with `platform compute request`." : "A bigger compute tier will not add disk. Free space on the current computer instead, and do not request a compute upgrade for disk pressure.";
|
|
3092
|
+
return `<system_notification>${lead} Current usage: ${stats.join(", ")}. Act now: ${remedy} ${closer} Do not mention this constraint as you work unless the user asks. This warning does not change whether the user is owed a reply. Do not call \`platform channel suppress-reply\` because of this warning. Complete any reply already owed after checkpointing. This is an automated resource warning, not a message from the user; continue the task, adjusted.</system_notification>`;
|
|
3093
|
+
}
|
|
3094
|
+
const resourcePressureWarningExtension = (pi) => {
|
|
3095
|
+
let agentActive = false;
|
|
3096
|
+
let warnedMemThisTurn = false;
|
|
3097
|
+
let warnedDiskThisTurn = false;
|
|
3098
|
+
let timer = null;
|
|
3099
|
+
const sampleCpu = createCpuSampler();
|
|
3100
|
+
async function checkOnce() {
|
|
3101
|
+
if (!agentActive || warnedMemThisTurn && warnedDiskThisTurn) return;
|
|
3102
|
+
const [mem, cpuPct, diskPct] = await Promise.all([
|
|
3103
|
+
readMemUsePct(),
|
|
3104
|
+
sampleCpu(),
|
|
3105
|
+
readDiskUsePct()
|
|
3106
|
+
]);
|
|
3107
|
+
if (!agentActive) return;
|
|
3108
|
+
let trigger = null;
|
|
3109
|
+
if (!warnedMemThisTurn && mem !== null && mem.pct >= 80) {
|
|
3110
|
+
trigger = "memory";
|
|
3111
|
+
warnedMemThisTurn = true;
|
|
3112
|
+
} else if (!warnedDiskThisTurn && diskPct !== null && diskPct >= 80) {
|
|
3113
|
+
trigger = "disk";
|
|
3114
|
+
warnedDiskThisTurn = true;
|
|
3115
|
+
}
|
|
3116
|
+
if (trigger === null) return;
|
|
3117
|
+
log$9.warn({
|
|
3118
|
+
trigger,
|
|
3119
|
+
mem,
|
|
3120
|
+
cpuPct,
|
|
3121
|
+
diskPct
|
|
3122
|
+
}, "resource pressure warning delivered to agent");
|
|
3123
|
+
await pi.sendMessage({
|
|
3124
|
+
customType: "anyone-resource-pressure-warning",
|
|
3125
|
+
content: resourcePressureWarningText(trigger, {
|
|
3126
|
+
mem,
|
|
3127
|
+
cpuPct,
|
|
3128
|
+
diskPct
|
|
3129
|
+
}),
|
|
3130
|
+
display: false
|
|
3131
|
+
}, {
|
|
3132
|
+
triggerTurn: true,
|
|
3133
|
+
deliverAs: "followUp"
|
|
3134
|
+
});
|
|
3135
|
+
}
|
|
3136
|
+
pi.on("agent_start", async () => {
|
|
3137
|
+
agentActive = true;
|
|
3138
|
+
warnedMemThisTurn = false;
|
|
3139
|
+
warnedDiskThisTurn = false;
|
|
3140
|
+
if (!timer) {
|
|
3141
|
+
timer = setInterval(() => {
|
|
3142
|
+
checkOnce().catch((err) => {
|
|
3143
|
+
log$9.error({ err }, "resource pressure check failed");
|
|
3144
|
+
});
|
|
3145
|
+
}, POLL_INTERVAL_MS);
|
|
3146
|
+
timer.unref?.();
|
|
3147
|
+
}
|
|
3148
|
+
});
|
|
3149
|
+
pi.on("agent_end", async () => {
|
|
3150
|
+
agentActive = false;
|
|
3151
|
+
if (timer) {
|
|
3152
|
+
clearInterval(timer);
|
|
3153
|
+
timer = null;
|
|
3154
|
+
}
|
|
3155
|
+
});
|
|
3156
|
+
};
|
|
3157
|
+
//#endregion
|
|
3158
|
+
//#region src/extensions/disk-guard.ts
|
|
3159
|
+
const log$8 = logger.child({ module: "disk-guard" });
|
|
3160
|
+
/**
|
|
3161
|
+
* In-band bypass. The guard is a safety net, not a jail: when the agent knows
|
|
3162
|
+
* a flagged command is genuinely safe (writing to a different mount, a tiny
|
|
3163
|
+
* bounded download, a delete-then-clone one-liner, an emergency it accepts the
|
|
3164
|
+
* risk on) it can force the command through by appending this marker as a
|
|
3165
|
+
* trailing shell comment. Kept as a comment so it never changes what the
|
|
3166
|
+
* command does, and matched case-insensitively with flexible spacing so the
|
|
3167
|
+
* agent doesn't have to reproduce it byte-for-byte.
|
|
3168
|
+
*/
|
|
3169
|
+
const BYPASS_MARKER = /#\s*disk-guard:\s*allow\b/i;
|
|
3170
|
+
/** The exact marker text the block message tells the agent to append. */
|
|
3171
|
+
const BYPASS_HINT = "# disk-guard: allow";
|
|
3172
|
+
/**
|
|
3173
|
+
* Harness-level kill switch: set DISK_GUARD_DISABLE=1 to turn the guard off
|
|
3174
|
+
* entirely. This is the "I own my harness, let me opt out" knob — an agent
|
|
3175
|
+
* that boots its own harness can disable the guard for its whole process
|
|
3176
|
+
* without a code roll, and it's also the fleet-wide escape hatch if the
|
|
3177
|
+
* classifier ever misfires and blocks real work. The bare name is honored
|
|
3178
|
+
* first; the SKYDIVE_/ANYONE_ prefixes are accepted too for consistency with
|
|
3179
|
+
* the other env overrides. Empty/unset/"0"/"false" leave the guard on.
|
|
3180
|
+
*/
|
|
3181
|
+
function guardDisabledByEnv() {
|
|
3182
|
+
for (const name of [
|
|
3183
|
+
"DISK_GUARD_DISABLE",
|
|
3184
|
+
"SKYDIVE_DISK_GUARD_DISABLE",
|
|
3185
|
+
"ANYONE_DISK_GUARD_DISABLE"
|
|
3186
|
+
]) {
|
|
3187
|
+
const value = process.env[name];
|
|
3188
|
+
if (value != null && value !== "" && value !== "0" && value !== "false") return true;
|
|
3189
|
+
}
|
|
3190
|
+
return false;
|
|
3191
|
+
}
|
|
3192
|
+
/** True when the command carries the in-band bypass marker. */
|
|
3193
|
+
function hasBypassMarker(command) {
|
|
3194
|
+
return BYPASS_MARKER.test(command);
|
|
3195
|
+
}
|
|
3196
|
+
/**
|
|
3197
|
+
* Commands that reclaim space or merely inspect it. If any of these verbs
|
|
3198
|
+
* appears in the command line, we never block — otherwise the guard would trap
|
|
3199
|
+
* the agent by blocking the exact command it needs to dig out. Matched as
|
|
3200
|
+
* whole words so `remove-item` etc. don't accidentally match `rm`.
|
|
3201
|
+
*/
|
|
3202
|
+
const RECLAIM_PATTERNS = [
|
|
3203
|
+
/\brm\b/,
|
|
3204
|
+
/\brmdir\b/,
|
|
3205
|
+
/\bdf\b/,
|
|
3206
|
+
/\bdu\b/,
|
|
3207
|
+
/\bncdu\b/,
|
|
3208
|
+
/\bfind\b[^|]*\s-delete\b/,
|
|
3209
|
+
/\btruncate\b/,
|
|
3210
|
+
/\bgit\s+(gc|prune|clean|worktree\s+remove|worktree\s+prune)\b/,
|
|
3211
|
+
/\b(yarn|npm|pnpm|bun)\s+.*\b(cache\s+clean|cache\s+clear|store\s+prune)\b/,
|
|
3212
|
+
/\bcache\s+(clean|clear|prune)\b/,
|
|
3213
|
+
/\b(docker|podman)\s+.*\bprune\b/,
|
|
3214
|
+
/\bapt(-get)?\s+clean\b/,
|
|
3215
|
+
/\bjournalctl\b[^|]*--vacuum/
|
|
3216
|
+
];
|
|
3217
|
+
/**
|
|
3218
|
+
* File extensions that mean a download is actually LARGE — archives, disk
|
|
3219
|
+
* images, compiled/binary artifacts, model weights, media. A curl/wget is only
|
|
3220
|
+
* gated when it writes one of these; an API/page fetch to a `.json`/`.html`/
|
|
3221
|
+
* `.txt` file is tiny and must not be blocked. Derived from 4,144 real
|
|
3222
|
+
* commands: ~64% of `curl -o` uses were tiny fetches, only ~4% large.
|
|
3223
|
+
*/
|
|
3224
|
+
const BIG_DOWNLOAD_EXT = "(?:tar\\.gz|tgz|tar|zip|iso|gz|bz2|xz|zst|deb|rpm|pkg|dmg|whl|jar|7z|img|mp4|mov|avi|mkv|onnx|gguf|safetensors|bin|node)";
|
|
3225
|
+
/**
|
|
3226
|
+
* Commands that consume a meaningful amount of disk. Kept deliberately tight
|
|
3227
|
+
* and high-precision: validated against 4,144 real commands from the last 7
|
|
3228
|
+
* days, the earlier "writes a file" heuristic flagged 82% of everything (a
|
|
3229
|
+
* `curl -o /tmp/x.json` API call is not a disk event). This set flags ~33%,
|
|
3230
|
+
* almost all genuinely large — real installs, clones, big-archive downloads,
|
|
3231
|
+
* extractions. What was DROPPED and why:
|
|
3232
|
+
* - `git fetch` / `git pull` — incremental on an existing clone, usually tiny.
|
|
3233
|
+
* - `git checkout` — overwhelmingly `git checkout <ref> -- <file>` or a
|
|
3234
|
+
* branch switch, ~zero net growth; the rare full materialization isn't
|
|
3235
|
+
* worth the false-positive rate.
|
|
3236
|
+
* - bare `curl -o` / `wget -o` — see BIG_DOWNLOAD_EXT above.
|
|
3237
|
+
* - loose `… build` — matched `--mode=skip-build`, `oxfmt … build`, prose.
|
|
3238
|
+
* The remaining big-disk op in escher is `git clone` and `git worktree add`
|
|
3239
|
+
* (which is really a checkout), both kept.
|
|
3240
|
+
*/
|
|
3241
|
+
const SPACE_HUNGRY_PATTERNS = [
|
|
3242
|
+
/\bgit\s+clone\b/,
|
|
3243
|
+
/\bgit\s+worktree\s+add\b/,
|
|
3244
|
+
/\b(yarn|npm|pnpm|bun)\s+(install|add|ci)\b/,
|
|
3245
|
+
/\byarn\s*$/,
|
|
3246
|
+
/\byarn\s+--(?!version|help)\S/,
|
|
3247
|
+
/\bpip3?\s+install\b/,
|
|
3248
|
+
/\bapt(-get)?\s+install\b/,
|
|
3249
|
+
/\bnpm\s+pack\b/,
|
|
3250
|
+
/\bdocker\s+(build|pull)\b/,
|
|
3251
|
+
new RegExp(`\\b(?:curl|wget)\\b[^\\n]*\\s-[a-zA-Z]*[oO]\\b[^\\n]*\\.${BIG_DOWNLOAD_EXT}\\b`, "i"),
|
|
3252
|
+
new RegExp(`\\b(?:curl|wget)\\b[^\\n]*\\.${BIG_DOWNLOAD_EXT}\\b[^\\n]*\\s-[a-zA-Z]*[oO]\\b`, "i"),
|
|
3253
|
+
/\btar\s+[^\n|]*x[^\n|]*f/,
|
|
3254
|
+
/\bunzip\b/,
|
|
3255
|
+
/\bdd\b[^\n|]*\bof=/
|
|
3256
|
+
];
|
|
3257
|
+
/**
|
|
3258
|
+
* True when the command reclaims or inspects space — these are always allowed,
|
|
3259
|
+
* even on a 100%-full box, so the agent can dig itself out.
|
|
3260
|
+
*/
|
|
3261
|
+
function isReclaimCommand(command) {
|
|
3262
|
+
return RECLAIM_PATTERNS.some((re) => re.test(command));
|
|
1997
3263
|
}
|
|
3264
|
+
/**
|
|
3265
|
+
* True when the command is likely to consume a meaningful amount of disk.
|
|
3266
|
+
* A reclaim/inspect command is never space-hungry — the reclaim check wins so a
|
|
3267
|
+
* `git worktree remove` or a `yarn cache clean` is never mistaken for growth.
|
|
3268
|
+
*/
|
|
3269
|
+
function isSpaceHungryCommand(command) {
|
|
3270
|
+
if (isReclaimCommand(command)) return false;
|
|
3271
|
+
return SPACE_HUNGRY_PATTERNS.some((re) => re.test(command));
|
|
3272
|
+
}
|
|
3273
|
+
/**
|
|
3274
|
+
* The decision, factored out and pure so it's exhaustively testable without a
|
|
3275
|
+
* real filesystem. Block only when we have a disk reading, it's at/above the
|
|
3276
|
+
* critical threshold, the command is space-hungry (and not a reclaim), and the
|
|
3277
|
+
* agent hasn't explicitly opted out with the bypass marker.
|
|
3278
|
+
*/
|
|
3279
|
+
function shouldBlockForDisk(command, diskPct) {
|
|
3280
|
+
if (diskPct === null) return false;
|
|
3281
|
+
if (diskPct < 95) return false;
|
|
3282
|
+
if (hasBypassMarker(command)) return false;
|
|
3283
|
+
return isSpaceHungryCommand(command);
|
|
3284
|
+
}
|
|
3285
|
+
/** The agent-facing explanation returned as the blocked tool result. */
|
|
3286
|
+
function diskBlockReason(command, diskPct) {
|
|
3287
|
+
return `Blocked: the sandbox disk is ${diskPct}% full and this command (\`${command.trim().slice(0, 120)}\`) writes a large amount, so it would fail partway with ENOSPC and leave a corrupt result. Reclaim space FIRST, then retry. Free ONLY what THIS conversation created — scratch/build output you wrote this run, downloads you're done with, and worktrees/branches whose work you've already committed and pushed (\`git worktree remove\`, \`yarn cache clean\`, delete your own scratch). Do NOT blindly wipe /tmp or delete a clone/worktree you don't recognize — other conversations share this box. Check headroom with \`df -h /\` and \`du -sh ~/workspace/* 2>/dev/null\`. If you genuinely can't free enough, stop and tell the user you're blocked on disk rather than retrying the write. If you're certain this command is safe anyway (writes elsewhere, tiny bounded size, delete-then-write), force it through by appending \` ${BYPASS_HINT}\` to the command.`;
|
|
3288
|
+
}
|
|
3289
|
+
const diskGuardExtension = (pi) => {
|
|
3290
|
+
pi.on("tool_call", async (event) => {
|
|
3291
|
+
if (event.toolName !== "bash") return;
|
|
3292
|
+
if (guardDisabledByEnv()) return;
|
|
3293
|
+
const command = event.input.command;
|
|
3294
|
+
if (typeof command !== "string" || command.length === 0) return;
|
|
3295
|
+
if (hasBypassMarker(command)) return;
|
|
3296
|
+
if (!isSpaceHungryCommand(command)) return;
|
|
3297
|
+
const diskPct = await readDiskUsePct();
|
|
3298
|
+
if (!shouldBlockForDisk(command, diskPct)) return;
|
|
3299
|
+
log$8.warn({
|
|
3300
|
+
diskPct,
|
|
3301
|
+
command: command.slice(0, 200)
|
|
3302
|
+
}, "blocked space-hungry bash command on near-full disk");
|
|
3303
|
+
return {
|
|
3304
|
+
block: true,
|
|
3305
|
+
reason: diskBlockReason(command, diskPct)
|
|
3306
|
+
};
|
|
3307
|
+
});
|
|
3308
|
+
};
|
|
1998
3309
|
//#endregion
|
|
1999
3310
|
//#region src/extensions/context-management-trim.ts
|
|
2000
3311
|
const CLEARED_PLACEHOLDER = "[old tool result cleared to save context — re-run the tool or re-read the source to recover it]";
|
|
@@ -2087,7 +3398,7 @@ function transformContextMessages(messages, config, now) {
|
|
|
2087
3398
|
}
|
|
2088
3399
|
//#endregion
|
|
2089
3400
|
//#region src/extensions/context-management.ts
|
|
2090
|
-
const log$
|
|
3401
|
+
const log$7 = logger.child({ module: "context-management-extension" });
|
|
2091
3402
|
function isAnthropicMessagesPayload(payload) {
|
|
2092
3403
|
if (typeof payload !== "object" || payload === null) return false;
|
|
2093
3404
|
const candidate = payload;
|
|
@@ -2152,13 +3463,13 @@ function createContextManagementExtension() {
|
|
|
2152
3463
|
setContextManagementFlagOverride(getPolledFlag("contextManagement"));
|
|
2153
3464
|
onFlagChange("contextManagement", (enabled) => {
|
|
2154
3465
|
setContextManagementFlagOverride(enabled);
|
|
2155
|
-
log$
|
|
3466
|
+
log$7.info({
|
|
2156
3467
|
event: "context_management_flag_update",
|
|
2157
3468
|
enabled
|
|
2158
3469
|
}, "context-management flag updated from platform");
|
|
2159
3470
|
});
|
|
2160
3471
|
startFeatureFlagPoller();
|
|
2161
|
-
log$
|
|
3472
|
+
log$7.info({
|
|
2162
3473
|
event: "context_management_registered",
|
|
2163
3474
|
enabled: initial.enabled,
|
|
2164
3475
|
flagSource: hasFlagSource(),
|
|
@@ -2170,13 +3481,13 @@ function createContextManagementExtension() {
|
|
|
2170
3481
|
const { messages } = event;
|
|
2171
3482
|
try {
|
|
2172
3483
|
const result = transformContextIfEnabled(messages, getContextManagementConfig(), Date.now());
|
|
2173
|
-
if (result.stats && (result.stats.clearedResults > 0 || result.stats.trimmedResults > 0)) log$
|
|
3484
|
+
if (result.stats && (result.stats.clearedResults > 0 || result.stats.trimmedResults > 0)) log$7.info({
|
|
2174
3485
|
event: "context_management_applied",
|
|
2175
3486
|
...result.stats
|
|
2176
3487
|
}, "trimmed/cleared tool output before LLM call");
|
|
2177
3488
|
return { messages: result.messages };
|
|
2178
3489
|
} catch (err) {
|
|
2179
|
-
log$
|
|
3490
|
+
log$7.error({
|
|
2180
3491
|
err,
|
|
2181
3492
|
event: "context_management_transform_failed"
|
|
2182
3493
|
}, "context transform failed; passing messages through unchanged");
|
|
@@ -2188,7 +3499,7 @@ function createContextManagementExtension() {
|
|
|
2188
3499
|
}
|
|
2189
3500
|
//#endregion
|
|
2190
3501
|
//#region src/extensions/current-time.ts
|
|
2191
|
-
const log$
|
|
3502
|
+
const log$6 = logger.child({ module: "current-time-extension" });
|
|
2192
3503
|
const PI_DATE_LINE = /^Current date:.*$/m;
|
|
2193
3504
|
function formatCurrentTimeLine(now) {
|
|
2194
3505
|
return `Current date: ${now.getUTCFullYear()}-${String(now.getUTCMonth() + 1).padStart(2, "0")}-${String(now.getUTCDate()).padStart(2, "0")} (${new Intl.DateTimeFormat("en-US", {
|
|
@@ -2201,7 +3512,7 @@ const currentTimeExtension = (pi) => {
|
|
|
2201
3512
|
const line = formatCurrentTimeLine(/* @__PURE__ */ new Date());
|
|
2202
3513
|
const base = event.systemPrompt;
|
|
2203
3514
|
if (PI_DATE_LINE.test(base)) {
|
|
2204
|
-
log$
|
|
3515
|
+
log$6.info({ event: "pi_date_line_present" }, "pi base prompt carries its own 'Current date:' line again; replacing it in place (pi prompt format may have changed)");
|
|
2205
3516
|
return { systemPrompt: base.replace(PI_DATE_LINE, line) };
|
|
2206
3517
|
}
|
|
2207
3518
|
return { systemPrompt: `${base}\n${line}` };
|
|
@@ -2210,37 +3521,35 @@ const currentTimeExtension = (pi) => {
|
|
|
2210
3521
|
//#endregion
|
|
2211
3522
|
//#region src/memory.ts
|
|
2212
3523
|
/**
|
|
2213
|
-
* In-harness memory
|
|
2214
|
-
*
|
|
2215
|
-
* The agent has an agent-level file-based memory at `<cwd>/.memory/`,
|
|
2216
|
-
* organized by directory:
|
|
3524
|
+
* In-harness memory readers.
|
|
2217
3525
|
*
|
|
2218
|
-
*
|
|
2219
|
-
*
|
|
2220
|
-
* .memory/feedback/<topic>.md
|
|
2221
|
-
* .memory/reference/<topic>.md
|
|
3526
|
+
* The agent has an agent-level file-based memory at `<cwd>/.memory/`, split
|
|
3527
|
+
* into two halves that are surfaced differently:
|
|
2222
3528
|
*
|
|
2223
|
-
*
|
|
2224
|
-
*
|
|
2225
|
-
*
|
|
3529
|
+
* 1. Shared knowledge (projects, lessons, external systems) — indexed by a
|
|
3530
|
+
* single hand-maintained `<cwd>/.memory/MEMORY.md` that the *agent* writes
|
|
3531
|
+
* and curates, Claude-Code style: one line per fact pointing at the file
|
|
3532
|
+
* that holds it. `readMemoryIndexFile` just reads that file; the agent owns
|
|
3533
|
+
* its contents. This is the whole index for the shared half — there is no
|
|
3534
|
+
* derived walk and no per-directory `MEMORY.md`.
|
|
2226
3535
|
*
|
|
2227
|
-
*
|
|
2228
|
-
*
|
|
2229
|
-
*
|
|
2230
|
-
*
|
|
2231
|
-
*
|
|
2232
|
-
*
|
|
2233
|
-
*
|
|
3536
|
+
* 2. Per-person memory — `.memory/users/<id>-<name>/<topic>.md`. This half is
|
|
3537
|
+
* *derived*, not hand-maintained, because it has to be filtered to the one
|
|
3538
|
+
* person on the current turn (a single hand-written index couldn't be
|
|
3539
|
+
* scoped per-user without leaking one person's notes into another's
|
|
3540
|
+
* conversation). `buildMemoryIndex` walks a single user's directory, reads
|
|
3541
|
+
* only the frontmatter of each `.md` (open fd → read first ~4KB → close, in
|
|
3542
|
+
* parallel), and renders an index. Each `.md`'s frontmatter carries `name`
|
|
3543
|
+
* and `description`; bodies are never read — the agent loads a specific
|
|
3544
|
+
* memory's body on demand via the `read` tool.
|
|
2234
3545
|
*
|
|
2235
|
-
* Mtime cache keyed by cwd — within the
|
|
2236
|
-
* is fixed, so this is effectively a single-entry
|
|
2237
|
-
* when any `.md`
|
|
2238
|
-
* memory didn't change reuse the cached
|
|
3546
|
+
* Mtime cache (for the derived per-user half) keyed by cwd — within the
|
|
3547
|
+
* lifetime of a sandbox the cwd is fixed, so this is effectively a single-entry
|
|
3548
|
+
* cache. It invalidates when any `.md` under `users/` is added/modified/
|
|
3549
|
+
* deleted; turns where memory didn't change reuse the cached entries.
|
|
2239
3550
|
*
|
|
2240
|
-
* Frontmatter is parsed as YAML (`yaml` package) and validated with a
|
|
2241
|
-
*
|
|
2242
|
-
* index. The same schema can be reused at write time if we want to
|
|
2243
|
-
* validate before commit.
|
|
3551
|
+
* Frontmatter is parsed as YAML (`yaml` package) and validated with a zod
|
|
3552
|
+
* schema — files that don't match the shape are dropped from the index.
|
|
2244
3553
|
*/
|
|
2245
3554
|
const FRONTMATTER_READ_BYTES = 4096;
|
|
2246
3555
|
const FrontmatterSchema = z.object({
|
|
@@ -2248,33 +3557,118 @@ const FrontmatterSchema = z.object({
|
|
|
2248
3557
|
description: z.string().min(1)
|
|
2249
3558
|
}).passthrough();
|
|
2250
3559
|
const MEMORY_DIRNAME = ".memory";
|
|
2251
|
-
const
|
|
2252
|
-
|
|
2253
|
-
|
|
2254
|
-
|
|
2255
|
-
|
|
3560
|
+
const MEMORY_INDEX_FILENAME = "MEMORY.md";
|
|
3561
|
+
const USERS_DIRNAME = "users";
|
|
3562
|
+
/**
|
|
3563
|
+
* The shared-knowledge type dirs from the old frontmatter-indexed layout, used
|
|
3564
|
+
* only to seed a `MEMORY.md` for agents created before it existed (see
|
|
3565
|
+
* `seedMemoryIndexFile`). `users/` is deliberately excluded — per-person memory
|
|
3566
|
+
* stays derived and never lands in the shared, un-scoped `MEMORY.md`.
|
|
3567
|
+
*/
|
|
3568
|
+
const LEGACY_SHARED_TYPES = [
|
|
3569
|
+
{
|
|
3570
|
+
dir: "projects",
|
|
3571
|
+
label: "Projects"
|
|
3572
|
+
},
|
|
3573
|
+
{
|
|
3574
|
+
dir: "feedback",
|
|
3575
|
+
label: "Feedback"
|
|
3576
|
+
},
|
|
3577
|
+
{
|
|
3578
|
+
dir: "reference",
|
|
3579
|
+
label: "Reference"
|
|
3580
|
+
}
|
|
2256
3581
|
];
|
|
2257
|
-
|
|
2258
|
-
|
|
2259
|
-
|
|
2260
|
-
|
|
2261
|
-
|
|
2262
|
-
|
|
2263
|
-
|
|
3582
|
+
/**
|
|
3583
|
+
* Soft budget for an injected index block. The index is read and injected into
|
|
3584
|
+
* the system prompt on every turn, so every entry costs context for the rest of
|
|
3585
|
+
* the conversation. Past this size we nudge the agent to consolidate and prune
|
|
3586
|
+
* rather than keep appending. Not a hard cap — nothing is truncated.
|
|
3587
|
+
*/
|
|
3588
|
+
const MEMORY_INDEX_SOFT_BUDGET_CHARS = 2e4;
|
|
3589
|
+
/**
|
|
3590
|
+
* Hard cap on the injected index — double the soft budget. The soft budget only
|
|
3591
|
+
* warns; this actually bounds what we inject so a runaway index can't consume
|
|
3592
|
+
* unbounded context on every turn. Past this, the index is truncated (on a line
|
|
3593
|
+
* boundary) before injection. It's a backstop, not a normal operating point.
|
|
3594
|
+
*/
|
|
3595
|
+
const MEMORY_INDEX_HARD_BUDGET_CHARS = MEMORY_INDEX_SOFT_BUDGET_CHARS * 2;
|
|
3596
|
+
/**
|
|
3597
|
+
* Cheap size summary of a rendered index block, used to surface how much of the
|
|
3598
|
+
* every-turn context budget the index is spending so the agent keeps it lean.
|
|
3599
|
+
* Counts pointer/file lines — both the derived `` - `path` — desc`` form and
|
|
3600
|
+
* the hand-maintained `- [Title](path) — hook` form — not the group headers.
|
|
3601
|
+
*/
|
|
3602
|
+
function summarizeIndex(index) {
|
|
3603
|
+
const entryCount = index.split("\n").filter((line) => /^\s*- (?:`|\[)/.test(line)).length;
|
|
3604
|
+
const charCount = index.length;
|
|
3605
|
+
return {
|
|
3606
|
+
entryCount,
|
|
3607
|
+
charCount,
|
|
3608
|
+
overBudget: charCount > MEMORY_INDEX_SOFT_BUDGET_CHARS
|
|
3609
|
+
};
|
|
3610
|
+
}
|
|
3611
|
+
/**
|
|
3612
|
+
* One-line size note for an index block header, e.g. `12 entries, 3187 chars`.
|
|
3613
|
+
*/
|
|
3614
|
+
function indexSizeNote(index) {
|
|
3615
|
+
const { entryCount, charCount } = summarizeIndex(index);
|
|
3616
|
+
return `${entryCount} ${entryCount === 1 ? "entry" : "entries"}, ${charCount} chars`;
|
|
3617
|
+
}
|
|
3618
|
+
/**
|
|
3619
|
+
* An explicit warning to surface to the agent when an index has grown past its
|
|
3620
|
+
* budget, or `null` when it's within budget. Extensions render this prominently
|
|
3621
|
+
* above the index so the agent prunes before it keeps appending.
|
|
3622
|
+
*/
|
|
3623
|
+
function indexBudgetWarning(index) {
|
|
3624
|
+
const { charCount, overBudget } = summarizeIndex(index);
|
|
3625
|
+
if (!overBudget) return null;
|
|
3626
|
+
return `⚠️ This memory index is ${charCount} chars, over its ${MEMORY_INDEX_SOFT_BUDGET_CHARS}-char budget. It's costing you context on every turn — consolidate duplicate entries and delete stale ones to bring it back under budget before adding anything new.`;
|
|
3627
|
+
}
|
|
3628
|
+
/**
|
|
3629
|
+
* Enforce the hard cap on an index before injection. Under the cap the index is
|
|
3630
|
+
* returned unchanged; over it, the index is truncated on a line boundary and a
|
|
3631
|
+
* notice is appended naming the true size so the agent knows entries are hidden
|
|
3632
|
+
* and must be pruned. This is the actual bound on injected context — callers
|
|
3633
|
+
* still report the true size via {@link indexSizeNote} so nothing is masked.
|
|
3634
|
+
*/
|
|
3635
|
+
function enforceMemoryIndexHardBudget(index) {
|
|
3636
|
+
if (index.length <= 4e4) return index;
|
|
3637
|
+
const clipped = index.slice(0, MEMORY_INDEX_HARD_BUDGET_CHARS);
|
|
3638
|
+
const lastNewline = clipped.lastIndexOf("\n");
|
|
3639
|
+
return `${lastNewline > 0 ? clipped.slice(0, lastNewline) : clipped}\n\n⚠️ Memory index truncated at ${MEMORY_INDEX_HARD_BUDGET_CHARS} chars (it is ${index.length}). Entries past this point are NOT shown. Prune the index now — delete stale entries and consolidate duplicates.`;
|
|
3640
|
+
}
|
|
3641
|
+
/**
|
|
3642
|
+
* Read the agent's hand-maintained shared index at `.memory/MEMORY.md`.
|
|
3643
|
+
* Returns the trimmed contents, or `null` when the file is absent or empty —
|
|
3644
|
+
* the agent owns this file, so we surface exactly what it wrote.
|
|
3645
|
+
*/
|
|
3646
|
+
async function readMemoryIndexFile({ cwd }) {
|
|
3647
|
+
const path = join(cwd, MEMORY_DIRNAME, MEMORY_INDEX_FILENAME);
|
|
3648
|
+
try {
|
|
3649
|
+
const trimmed = (await readFile(path, "utf-8")).trim();
|
|
3650
|
+
return trimmed.length > 0 ? trimmed : null;
|
|
3651
|
+
} catch {
|
|
3652
|
+
return null;
|
|
3653
|
+
}
|
|
3654
|
+
}
|
|
2264
3655
|
const cache = /* @__PURE__ */ new Map();
|
|
2265
3656
|
/**
|
|
3657
|
+
* Build the derived per-person index for a single user.
|
|
3658
|
+
*
|
|
2266
3659
|
* Returns:
|
|
2267
|
-
* - `null` if `.memory/` doesn't exist
|
|
2268
|
-
* - `""` if
|
|
3660
|
+
* - `null` if `.memory/users/` doesn't exist
|
|
3661
|
+
* - `""` if it exists but this user has no memory
|
|
2269
3662
|
* - rendered markdown body (no surrounding header — caller wraps)
|
|
2270
3663
|
*
|
|
2271
|
-
*
|
|
2272
|
-
*
|
|
2273
|
-
*
|
|
3664
|
+
* Scoping by id keeps one person's memory from bleeding into another's
|
|
3665
|
+
* conversation. The mtime-keyed cache stores the raw walked entries (the cost
|
|
3666
|
+
* is the FS walk); filtering by user is cheap and runs per call, so two turns
|
|
3667
|
+
* with different users on the same cwd render correctly from one cached walk.
|
|
2274
3668
|
*/
|
|
2275
|
-
async function buildMemoryIndex({ cwd,
|
|
2276
|
-
const
|
|
2277
|
-
const maxMtimeMs = await maxMtimeAcrossDir(
|
|
3669
|
+
async function buildMemoryIndex({ cwd, userId }) {
|
|
3670
|
+
const usersDirAbs = join(cwd, MEMORY_DIRNAME, USERS_DIRNAME);
|
|
3671
|
+
const maxMtimeMs = await maxMtimeAcrossDir(usersDirAbs);
|
|
2278
3672
|
if (maxMtimeMs === null) {
|
|
2279
3673
|
cache.delete(cwd);
|
|
2280
3674
|
return null;
|
|
@@ -2282,13 +3676,13 @@ async function buildMemoryIndex({ cwd, scope }) {
|
|
|
2282
3676
|
let cached = cache.get(cwd);
|
|
2283
3677
|
if (!cached || cached.builtAtMs < maxMtimeMs) {
|
|
2284
3678
|
cached = {
|
|
2285
|
-
entries: await
|
|
3679
|
+
entries: await collectUserEntries(usersDirAbs, cwd),
|
|
2286
3680
|
builtAtMs: Date.now()
|
|
2287
3681
|
};
|
|
2288
3682
|
cache.set(cwd, cached);
|
|
2289
3683
|
}
|
|
2290
|
-
const visible = cached.entries.filter((entry) =>
|
|
2291
|
-
return visible.length === 0 ? "" :
|
|
3684
|
+
const visible = cached.entries.filter((entry) => entry.subject.startsWith(userId));
|
|
3685
|
+
return visible.length === 0 ? "" : renderUserIndex(visible);
|
|
2292
3686
|
}
|
|
2293
3687
|
async function maxMtimeAcrossDir(dir) {
|
|
2294
3688
|
let dirStat;
|
|
@@ -2336,45 +3730,22 @@ async function listSubdirs(dir) {
|
|
|
2336
3730
|
}
|
|
2337
3731
|
return entries.filter((e) => e.isDirectory()).map((e) => join(dir, e.name));
|
|
2338
3732
|
}
|
|
2339
|
-
async function
|
|
2340
|
-
const
|
|
2341
|
-
await Promise.all(
|
|
2342
|
-
const
|
|
2343
|
-
|
|
2344
|
-
|
|
2345
|
-
await
|
|
2346
|
-
|
|
2347
|
-
|
|
2348
|
-
|
|
2349
|
-
|
|
2350
|
-
|
|
2351
|
-
|
|
2352
|
-
|
|
2353
|
-
|
|
2354
|
-
|
|
2355
|
-
subject,
|
|
2356
|
-
relPath: relative(cwd, file)
|
|
2357
|
-
};
|
|
2358
|
-
}));
|
|
2359
|
-
for (const e of parsed) if (e) collected.push(e);
|
|
2360
|
-
}));
|
|
2361
|
-
} else {
|
|
2362
|
-
const files = await listMdFilesShallow(typeDirAbs);
|
|
2363
|
-
const parsed = await Promise.all(files.map(async (file) => {
|
|
2364
|
-
const fm = await readFrontmatterOnly(file);
|
|
2365
|
-
if (!fm?.name || !fm?.description) return null;
|
|
2366
|
-
return {
|
|
2367
|
-
name: fm.name,
|
|
2368
|
-
description: fm.description,
|
|
2369
|
-
type,
|
|
2370
|
-
subject: null,
|
|
2371
|
-
relPath: relative(cwd, file)
|
|
2372
|
-
};
|
|
2373
|
-
}));
|
|
2374
|
-
for (const e of parsed) if (e) collected.push(e);
|
|
2375
|
-
}
|
|
2376
|
-
}));
|
|
2377
|
-
return collected;
|
|
3733
|
+
async function collectUserEntries(usersDirAbs, cwd) {
|
|
3734
|
+
const subjectDirs = await listSubdirs(usersDirAbs);
|
|
3735
|
+
return (await Promise.all(subjectDirs.map(async (subjectDirAbs) => {
|
|
3736
|
+
const subject = basename(subjectDirAbs);
|
|
3737
|
+
const files = await listMdFilesShallow(subjectDirAbs);
|
|
3738
|
+
return (await Promise.all(files.map(async (file) => {
|
|
3739
|
+
const fm = await readFrontmatterOnly(file);
|
|
3740
|
+
if (!fm?.name || !fm?.description) return null;
|
|
3741
|
+
return {
|
|
3742
|
+
name: fm.name,
|
|
3743
|
+
description: fm.description,
|
|
3744
|
+
subject,
|
|
3745
|
+
relPath: relative(cwd, file)
|
|
3746
|
+
};
|
|
3747
|
+
}))).filter((e) => e !== null);
|
|
3748
|
+
}))).flat();
|
|
2378
3749
|
}
|
|
2379
3750
|
async function readFrontmatterOnly(filePath) {
|
|
2380
3751
|
let fh;
|
|
@@ -2405,77 +3776,189 @@ function parseFrontmatter(text) {
|
|
|
2405
3776
|
const result = FrontmatterSchema.safeParse(parsed);
|
|
2406
3777
|
return result.success ? result.data : null;
|
|
2407
3778
|
}
|
|
2408
|
-
function
|
|
2409
|
-
const
|
|
2410
|
-
|
|
2411
|
-
|
|
2412
|
-
|
|
2413
|
-
|
|
2414
|
-
}
|
|
2415
|
-
|
|
3779
|
+
function renderUserIndex(entries) {
|
|
3780
|
+
const bySubject = /* @__PURE__ */ new Map();
|
|
3781
|
+
for (const e of entries) {
|
|
3782
|
+
const list = bySubject.get(e.subject) ?? [];
|
|
3783
|
+
list.push(e);
|
|
3784
|
+
bySubject.set(e.subject, list);
|
|
3785
|
+
}
|
|
3786
|
+
const lines = ["### Users"];
|
|
3787
|
+
for (const subject of [...bySubject.keys()].sort()) {
|
|
3788
|
+
lines.push(`- **${subject}**`);
|
|
3789
|
+
for (const e of bySubject.get(subject) ?? []) lines.push(` - \`${e.relPath}\` — ${e.description}`);
|
|
3790
|
+
}
|
|
3791
|
+
return lines.join("\n");
|
|
3792
|
+
}
|
|
3793
|
+
/**
|
|
3794
|
+
* One-time migration for agents created before `MEMORY.md` existed. If there is
|
|
3795
|
+
* no hand-maintained `.memory/MEMORY.md` yet but the agent has shared memory
|
|
3796
|
+
* files from the old frontmatter-indexed layout (`projects/`, `feedback/`,
|
|
3797
|
+
* `reference/`), derive a `MEMORY.md` from their frontmatter and write it once.
|
|
3798
|
+
* After that the agent owns the file — this never runs again for that agent and
|
|
3799
|
+
* never clobbers an existing index.
|
|
3800
|
+
*
|
|
3801
|
+
* Returns the seeded contents (also written to disk), or `null` when nothing
|
|
3802
|
+
* was seeded (index already present, or no legacy shared files). A write
|
|
3803
|
+
* failure propagates so the caller can log it; the read path then falls back to
|
|
3804
|
+
* whatever is on disk.
|
|
3805
|
+
*/
|
|
3806
|
+
async function seedMemoryIndexFile({ cwd }) {
|
|
3807
|
+
const memoryDirAbs = join(cwd, MEMORY_DIRNAME);
|
|
3808
|
+
const indexPath = join(memoryDirAbs, MEMORY_INDEX_FILENAME);
|
|
3809
|
+
if (await stat(indexPath).catch(() => null)) return null;
|
|
3810
|
+
const perType = await Promise.all(LEGACY_SHARED_TYPES.map(async ({ dir, label }) => {
|
|
3811
|
+
const typeDirAbs = join(memoryDirAbs, dir);
|
|
3812
|
+
const files = [];
|
|
3813
|
+
await walkMdFiles(typeDirAbs, files);
|
|
3814
|
+
return {
|
|
3815
|
+
label,
|
|
3816
|
+
entries: (await Promise.all(files.map(async (file) => {
|
|
3817
|
+
const fm = await readFrontmatterOnly(file);
|
|
3818
|
+
if (!fm?.name || !fm?.description) return null;
|
|
3819
|
+
return {
|
|
3820
|
+
name: fm.name,
|
|
3821
|
+
description: fm.description,
|
|
3822
|
+
relPath: relative(cwd, file)
|
|
3823
|
+
};
|
|
3824
|
+
}))).filter((e) => !!e)
|
|
3825
|
+
};
|
|
3826
|
+
}));
|
|
3827
|
+
if (perType.every((group) => group.entries.length === 0)) return null;
|
|
3828
|
+
const content = renderSeededIndex(perType);
|
|
3829
|
+
await writeFile(indexPath, `${content}\n`, "utf-8");
|
|
3830
|
+
return content;
|
|
3831
|
+
}
|
|
3832
|
+
function renderSeededIndex(groups) {
|
|
2416
3833
|
const sections = [];
|
|
2417
|
-
for (const
|
|
2418
|
-
|
|
2419
|
-
|
|
2420
|
-
|
|
2421
|
-
|
|
2422
|
-
const bySubject = /* @__PURE__ */ new Map();
|
|
2423
|
-
for (const e of items) {
|
|
2424
|
-
const subject = e.subject ?? "(unknown)";
|
|
2425
|
-
const list = bySubject.get(subject) ?? [];
|
|
2426
|
-
list.push(e);
|
|
2427
|
-
bySubject.set(subject, list);
|
|
2428
|
-
}
|
|
2429
|
-
const subjects = [...bySubject.keys()].sort();
|
|
2430
|
-
for (const subject of subjects) {
|
|
2431
|
-
sections.push(`- **${subject}**`);
|
|
2432
|
-
for (const e of bySubject.get(subject) ?? []) sections.push(` - \`${e.relPath}\` — ${e.description}`);
|
|
2433
|
-
}
|
|
2434
|
-
} else for (const e of items) sections.push(`- \`${e.relPath}\` — ${e.description}`);
|
|
3834
|
+
for (const { label, entries } of groups) {
|
|
3835
|
+
if (entries.length === 0) continue;
|
|
3836
|
+
sections.push(`### ${label}`);
|
|
3837
|
+
const sorted = [...entries].sort((a, b) => a.relPath.localeCompare(b.relPath));
|
|
3838
|
+
for (const e of sorted) sections.push(`- [${e.name}](${e.relPath}) — ${e.description}`);
|
|
2435
3839
|
sections.push("");
|
|
2436
3840
|
}
|
|
2437
3841
|
return sections.join("\n").trimEnd();
|
|
2438
3842
|
}
|
|
3843
|
+
/**
|
|
3844
|
+
* The version at which the hand-maintained `MEMORY.md` layout was introduced.
|
|
3845
|
+
* Used to classify an unversioned `.memory/`: if it already has a `MEMORY.md`
|
|
3846
|
+
* it's on this layout (not the pre-MEMORY.md v1 frontmatter layout), so it
|
|
3847
|
+
* shouldn't be treated as v1 and re-seeded.
|
|
3848
|
+
*/
|
|
3849
|
+
const HAND_MAINTAINED_INDEX_VERSION = 2;
|
|
3850
|
+
const VERSION_FILENAME = ".version";
|
|
3851
|
+
const MEMORY_MIGRATIONS = [{
|
|
3852
|
+
from: 1,
|
|
3853
|
+
to: 2,
|
|
3854
|
+
apply: async ({ cwd }) => {
|
|
3855
|
+
await seedMemoryIndexFile({ cwd });
|
|
3856
|
+
}
|
|
3857
|
+
}];
|
|
3858
|
+
/**
|
|
3859
|
+
* The layout version of an agent's `.memory/`:
|
|
3860
|
+
* - `null` when there's no `.memory/` at all (a fresh agent is current by
|
|
3861
|
+
* construction; nothing to migrate).
|
|
3862
|
+
* - `1` when `.memory/` exists but carries no `.version` marker — i.e. it
|
|
3863
|
+
* predates versioning.
|
|
3864
|
+
* - otherwise the integer in `.memory/.version`.
|
|
3865
|
+
*/
|
|
3866
|
+
async function readMemoryVersion(cwd) {
|
|
3867
|
+
const memoryDirAbs = join(cwd, MEMORY_DIRNAME);
|
|
3868
|
+
if (!(await stat(memoryDirAbs).catch(() => null))?.isDirectory()) return null;
|
|
3869
|
+
const raw = await readFile(join(memoryDirAbs, VERSION_FILENAME), "utf-8").catch(() => null);
|
|
3870
|
+
if (raw === null) return await stat(join(memoryDirAbs, MEMORY_INDEX_FILENAME)).then((s) => s.isFile()).catch(() => false) ? HAND_MAINTAINED_INDEX_VERSION : 1;
|
|
3871
|
+
const parsed = Number.parseInt(raw.trim(), 10);
|
|
3872
|
+
return Number.isInteger(parsed) && parsed > 0 ? parsed : 1;
|
|
3873
|
+
}
|
|
3874
|
+
async function writeMemoryVersion(cwd, version) {
|
|
3875
|
+
await writeFile(join(cwd, MEMORY_DIRNAME, VERSION_FILENAME), `${version}\n`, "utf-8");
|
|
3876
|
+
}
|
|
3877
|
+
/**
|
|
3878
|
+
* Bring an agent's `.memory/` up to `CURRENT_MEMORY_VERSION` by applying the
|
|
3879
|
+
* ordered migrations. Runs on session start. No-op when there's no `.memory/`
|
|
3880
|
+
* yet or it's already current. Migrations must be idempotent, so a lost/unwritten
|
|
3881
|
+
* version marker (the file isn't committed by the harness) only costs a repeated
|
|
3882
|
+
* no-op, never corruption. Returns the `{ from, to }` actually applied, or
|
|
3883
|
+
* `null` when nothing ran.
|
|
3884
|
+
*/
|
|
3885
|
+
async function migrateMemory({ cwd }) {
|
|
3886
|
+
const from = await readMemoryVersion(cwd);
|
|
3887
|
+
if (from === null || from >= 2) return null;
|
|
3888
|
+
let version = from;
|
|
3889
|
+
while (version < 2) {
|
|
3890
|
+
const migration = MEMORY_MIGRATIONS.find((m) => m.from === version);
|
|
3891
|
+
if (!migration) break;
|
|
3892
|
+
await migration.apply({ cwd });
|
|
3893
|
+
version = migration.to;
|
|
3894
|
+
}
|
|
3895
|
+
await writeMemoryVersion(cwd, version);
|
|
3896
|
+
return {
|
|
3897
|
+
from,
|
|
3898
|
+
to: version
|
|
3899
|
+
};
|
|
3900
|
+
}
|
|
2439
3901
|
//#endregion
|
|
2440
3902
|
//#region src/extensions/memory.ts
|
|
2441
|
-
const log$
|
|
3903
|
+
const log$5 = logger.child({ module: "memory-extension" });
|
|
2442
3904
|
/**
|
|
2443
3905
|
* The standing instructions for the memory system. Always injected (even with
|
|
2444
|
-
*
|
|
3906
|
+
* no `MEMORY.md`) so the agent knows it can persist notes and how. `users/` is
|
|
2445
3907
|
* described by the platform memory extension, which is the only thing that can
|
|
2446
3908
|
* scope it to a person — here we just point at it.
|
|
2447
3909
|
*/
|
|
2448
3910
|
function memoryInstructions(cwd) {
|
|
2449
3911
|
return `## Memory across conversations
|
|
2450
3912
|
|
|
2451
|
-
Persistent notes across conversations live at \`${cwd}/.memory/\` — plain markdown files in your repo
|
|
3913
|
+
Persistent notes across conversations live at \`${cwd}/.memory/\` — plain markdown files in your repo, one fact per file. You maintain a hand-written index of them at \`${cwd}/.memory/MEMORY.md\`, and the harness injects that index into your system prompt every turn. **Bodies are NOT auto-loaded** — when an index line looks relevant, use your \`read\` tool to load that specific file.
|
|
2452
3914
|
|
|
2453
3915
|
Memory records **what happened**: facts you learned, events, investigation findings, project and system details worth carrying forward. It is NOT where behavior goes. A standing rule about how you should act — a "from now on, always/never …", a tone or format preference, a workflow convention a user wants you to follow — belongs in \`soul.md\` (see the Persona / Standing instructions section), not here. When a note is really an instruction about your behavior, write it to \`soul.md\`; when it is a fact or a record of something that occurred, write it here.
|
|
2454
3916
|
|
|
2455
|
-
Shared knowledge is laid out as \`projects/<slug>/<topic>.md\` for project and system context, \`feedback/<topic>.md\` for concrete lessons learned from something that happened (the event and what it taught you — not a free-floating rule; the rule itself, if durable, goes in \`soul.md\`), and \`reference/<topic>.md\` for how external systems work. (Notes about a specific person live under \`users/\` and are
|
|
3917
|
+
You own \`MEMORY.md\`. When you learn something durable, write the fact to its own \`.md\` file and add a one-line pointer to \`MEMORY.md\` in the form \`- [Title](relative/path.md) — one-line hook\`, where the hook is what tells future-you when to open the file. \`MEMORY.md\` is an *index*, never a store — put the actual content in the topic file and only a pointer line in \`MEMORY.md\`; do not inline a fact's body into the index even when it seems cheaper. Start each topic file with \`name:\`/\`description:\` frontmatter (the \`description\` is the one-line hook) so the index can be re-seeded, re-derived, or linted from the files themselves. When a fact changes, edit both the file and its line; when it stops being true, delete the file and its line. Shared knowledge is laid out as \`projects/<slug>/<topic>.md\` for project and system context, \`feedback/<topic>.md\` for concrete lessons learned from something that happened (the event and what it taught you — not a free-floating rule; the rule itself, if durable, goes in \`soul.md\`), and \`reference/<topic>.md\` for how external systems work. (Notes about a specific person live under \`users/\` and are indexed for you separately, scoped to whoever you're talking to — don't put people's private notes in the shared \`MEMORY.md\`.)
|
|
3918
|
+
|
|
3919
|
+
Keep \`MEMORY.md\` lean. It's re-injected on *every* turn, so a small, high-signal index is worth far more than an exhaustive one — curate it like a tightly-edited table of contents, not a log:
|
|
3920
|
+
- Be selective. Only record something durable that will matter in a *future* conversation. Don't record what only matters right now, what you can re-derive on demand, or what's already obvious from the repo.
|
|
3921
|
+
- Consolidate before you create. Before adding a line, scan \`MEMORY.md\` for one that already covers the topic; if it exists, \`read\` that file and rewrite it with the new facts merged in rather than adding a near-duplicate. One fact per file, but don't fragment a topic across many thin files.
|
|
3922
|
+
- Prune as you go. Delete lines (and their files) that are wrong, stale, or superseded. The index header reports its size — when it's flagged over budget, consolidate and delete before adding anything new.
|
|
3923
|
+
- Write dates absolute, not relative. Resolve "last week" / "yesterday" to a concrete date when you record it (e.g. "on 7/3 Dhruv told me to …"), so the note still reads correctly in a future conversation.
|
|
3924
|
+
|
|
3925
|
+
Commit and push after editing \`.memory/\` to persist it.`;
|
|
2456
3926
|
}
|
|
2457
3927
|
function composeBlock$1({ cwd, index }) {
|
|
2458
3928
|
const instructions = memoryInstructions(cwd);
|
|
2459
3929
|
if (!index || index.length === 0) return instructions;
|
|
2460
|
-
|
|
3930
|
+
const header = `## Memory index (${indexSizeNote(index)})`;
|
|
3931
|
+
const warning = indexBudgetWarning(index);
|
|
3932
|
+
const rendered = enforceMemoryIndexHardBudget(index);
|
|
3933
|
+
return `${instructions}\n\n${warning ? `${header}\n\n${warning}\n\n${rendered}` : `${header}\n\n${rendered}`}`;
|
|
2461
3934
|
}
|
|
2462
3935
|
const memoryExtension = (pi) => {
|
|
2463
3936
|
let cachedBlock = null;
|
|
2464
3937
|
pi.on("session_start", async (_event, ctx) => {
|
|
2465
3938
|
try {
|
|
2466
|
-
const
|
|
2467
|
-
|
|
2468
|
-
|
|
2469
|
-
|
|
3939
|
+
const migrated = await migrateMemory({ cwd: ctx.cwd });
|
|
3940
|
+
if (migrated !== null) log$5.info({
|
|
3941
|
+
event: "memory_migrated",
|
|
3942
|
+
from: migrated.from,
|
|
3943
|
+
to: migrated.to
|
|
3944
|
+
}, "migrated memory layout to current version");
|
|
3945
|
+
} catch (err) {
|
|
3946
|
+
log$5.warn({
|
|
3947
|
+
err,
|
|
3948
|
+
event: "memory_migration_failed"
|
|
3949
|
+
}, "memory migration failed; continuing with existing index");
|
|
3950
|
+
}
|
|
3951
|
+
try {
|
|
3952
|
+
const index = await readMemoryIndexFile({ cwd: ctx.cwd });
|
|
2470
3953
|
cachedBlock = composeBlock$1({
|
|
2471
3954
|
cwd: ctx.cwd,
|
|
2472
3955
|
index
|
|
2473
3956
|
});
|
|
2474
3957
|
} catch (err) {
|
|
2475
|
-
log$
|
|
3958
|
+
log$5.warn({
|
|
2476
3959
|
err,
|
|
2477
3960
|
event: "memory_index_failed"
|
|
2478
|
-
}, "memory index
|
|
3961
|
+
}, "memory index read failed; injecting instructions only");
|
|
2479
3962
|
cachedBlock = memoryInstructions(ctx.cwd);
|
|
2480
3963
|
}
|
|
2481
3964
|
});
|
|
@@ -2486,7 +3969,7 @@ const memoryExtension = (pi) => {
|
|
|
2486
3969
|
};
|
|
2487
3970
|
//#endregion
|
|
2488
3971
|
//#region src/extensions/platform-memory.ts
|
|
2489
|
-
const log$
|
|
3972
|
+
const log$4 = logger.child({ module: "platform-memory-extension" });
|
|
2490
3973
|
/**
|
|
2491
3974
|
* Resolve the human on this turn via the API, keyed by the message id.
|
|
2492
3975
|
* `/sandbox/channel-context` only returns a sender for a platform-known
|
|
@@ -2499,13 +3982,13 @@ const log$5 = logger.child({ module: "platform-memory-extension" });
|
|
|
2499
3982
|
async function resolveTurnUser(messageId) {
|
|
2500
3983
|
const client = sandboxClient();
|
|
2501
3984
|
if (!client) {
|
|
2502
|
-
log$
|
|
3985
|
+
log$4.debug({ event: "resolve_turn_user_no_api_url" }, "no API url in env; withholding user memory");
|
|
2503
3986
|
return null;
|
|
2504
3987
|
}
|
|
2505
3988
|
try {
|
|
2506
3989
|
const res = await client["channel-context"].$get({ query: { messageId } });
|
|
2507
3990
|
if (!res.ok) {
|
|
2508
|
-
log$
|
|
3991
|
+
log$4.warn({
|
|
2509
3992
|
event: "resolve_turn_user_failed",
|
|
2510
3993
|
status: res.status
|
|
2511
3994
|
}, "channel-context returned non-ok; withholding user memory");
|
|
@@ -2518,7 +4001,7 @@ async function resolveTurnUser(messageId) {
|
|
|
2518
4001
|
displayName: sender.displayName
|
|
2519
4002
|
};
|
|
2520
4003
|
} catch (err) {
|
|
2521
|
-
log$
|
|
4004
|
+
log$4.warn({
|
|
2522
4005
|
err,
|
|
2523
4006
|
event: "resolve_turn_user_failed"
|
|
2524
4007
|
}, "failed to resolve current user; withholding user memory");
|
|
@@ -2534,9 +4017,12 @@ function slugifyName(name) {
|
|
|
2534
4017
|
function composeBlock({ index, user }) {
|
|
2535
4018
|
const instructions = `## Current user memory
|
|
2536
4019
|
|
|
2537
|
-
Notes about the person on this turn — the only \`users/\` memory you can see. Store anything you learn about them under \`${`.memory/users/${user.id}-${slugifyName(user.displayName)}/`}<topic>.md\`, using exactly this directory. Other people's \`users/\` notes are never shown, so never address someone by a name you only find in memory.`;
|
|
4020
|
+
Notes about the person on this turn — the only \`users/\` memory you can see. Store anything you learn about them under \`${`.memory/users/${user.id}-${slugifyName(user.displayName)}/`}<topic>.md\`, using exactly this directory. Other people's \`users/\` notes are never shown, so never address someone by a name you only find in memory. Keep it lean and be selective — this is re-injected every turn; consolidate related facts into one file and delete what's stale rather than piling on near-duplicates.`;
|
|
2538
4021
|
if (!index || index.length === 0) return instructions;
|
|
2539
|
-
|
|
4022
|
+
const warning = indexBudgetWarning(index);
|
|
4023
|
+
const header = `Memory index (${indexSizeNote(index)}):`;
|
|
4024
|
+
const rendered = enforceMemoryIndexHardBudget(index);
|
|
4025
|
+
return `${instructions}\n\n${warning ? `${warning}\n\n${header}\n\n${rendered}` : `${header}\n\n${rendered}`}`;
|
|
2540
4026
|
}
|
|
2541
4027
|
/**
|
|
2542
4028
|
* Build the platform memory extension. `channelContext` is the per-turn ref
|
|
@@ -2557,15 +4043,12 @@ function createPlatformMemoryExtension({ channelContext }) {
|
|
|
2557
4043
|
cachedBlock = composeBlock({
|
|
2558
4044
|
index: await buildMemoryIndex({
|
|
2559
4045
|
cwd: ctx.cwd,
|
|
2560
|
-
|
|
2561
|
-
kind: "user",
|
|
2562
|
-
userId: user.id
|
|
2563
|
-
}
|
|
4046
|
+
userId: user.id
|
|
2564
4047
|
}),
|
|
2565
4048
|
user
|
|
2566
4049
|
});
|
|
2567
4050
|
} catch (err) {
|
|
2568
|
-
log$
|
|
4051
|
+
log$4.warn({
|
|
2569
4052
|
err,
|
|
2570
4053
|
event: "user_memory_index_failed"
|
|
2571
4054
|
}, "user memory index build failed; skipping injection");
|
|
@@ -2579,8 +4062,22 @@ function createPlatformMemoryExtension({ channelContext }) {
|
|
|
2579
4062
|
};
|
|
2580
4063
|
}
|
|
2581
4064
|
//#endregion
|
|
4065
|
+
//#region src/extensions/provider-api-compat.ts
|
|
4066
|
+
const OPENAI_COMPATIBLE_PROVIDERS = ["google", "openai"];
|
|
4067
|
+
const OPENAI_COMPLETIONS_API = "openai-completions";
|
|
4068
|
+
/**
|
|
4069
|
+
* Keep older harnesses compatible with provider-native model specs that use
|
|
4070
|
+
* the OpenAI-compatible wire through the transparent proxy.
|
|
4071
|
+
*/
|
|
4072
|
+
const providerApiCompatExtension = (pi) => {
|
|
4073
|
+
for (const provider of OPENAI_COMPATIBLE_PROVIDERS) pi.registerProvider(provider, { models: getModels(provider).map((model) => ({
|
|
4074
|
+
...model,
|
|
4075
|
+
api: OPENAI_COMPLETIONS_API
|
|
4076
|
+
})) });
|
|
4077
|
+
};
|
|
4078
|
+
//#endregion
|
|
2582
4079
|
//#region src/extensions/self-trace.ts
|
|
2583
|
-
const log$
|
|
4080
|
+
const log$3 = logger.child({ module: "self-trace-extension" });
|
|
2584
4081
|
/**
|
|
2585
4082
|
* Reports the agent's own execution as OpenTelemetry spans:
|
|
2586
4083
|
* agent.session → agent.run → agent.turn.N → tool.NAME, with token/cost
|
|
@@ -2605,7 +4102,7 @@ const selfTraceExtension = (pi) => {
|
|
|
2605
4102
|
sessionSpan = tracer.startSpan("agent.session", { attributes: { "agent.model": modelId } }, remoteCtx);
|
|
2606
4103
|
sessionCtx = trace.setSpan(remoteCtx, sessionSpan);
|
|
2607
4104
|
const sc = sessionSpan.spanContext();
|
|
2608
|
-
log$
|
|
4105
|
+
log$3.info({
|
|
2609
4106
|
event: "self_trace_session_start",
|
|
2610
4107
|
trace_id: sc.traceId,
|
|
2611
4108
|
span_id: sc.spanId,
|
|
@@ -2717,13 +4214,13 @@ const selfTraceExtension = (pi) => {
|
|
|
2717
4214
|
* Lives in the harness package — soul.md is content from the agent's
|
|
2718
4215
|
* own git repo, not from the platform — so its handling stays here.
|
|
2719
4216
|
*/
|
|
2720
|
-
const log$
|
|
4217
|
+
const log$2 = logger.child({ module: "soul-extension" });
|
|
2721
4218
|
async function readSoul(cwd) {
|
|
2722
4219
|
try {
|
|
2723
4220
|
return (await readFile(join(cwd, "soul.md"), "utf8")).trim() || null;
|
|
2724
4221
|
} catch (err) {
|
|
2725
4222
|
if (err?.code === "ENOENT") return null;
|
|
2726
|
-
log$
|
|
4223
|
+
log$2.warn({
|
|
2727
4224
|
err,
|
|
2728
4225
|
event: "soul_read_failed"
|
|
2729
4226
|
}, "soul.md read failed");
|
|
@@ -2755,19 +4252,34 @@ const soulExtension = (pi) => {
|
|
|
2755
4252
|
};
|
|
2756
4253
|
//#endregion
|
|
2757
4254
|
//#region src/extensions/subagent/index.ts
|
|
2758
|
-
const log$
|
|
4255
|
+
const log$1 = logger.child({ module: "subagent-ext" });
|
|
2759
4256
|
const MAX_TASKS = 8;
|
|
2760
|
-
const
|
|
4257
|
+
const SpawnTaskItem = Type.Object({
|
|
2761
4258
|
task: Type.String({ description: "The task to delegate to a subagent run." }),
|
|
2762
4259
|
title: Type.String({
|
|
2763
4260
|
description: "A SHORT name for this task — 3-6 words, sentence case, no trailing period. This is what the person in the chat sees as the row for this subagent, so name the work, do not restate the prompt. Good: \"Audit the billing gate\", \"Compare competitor pricing\", \"Draft the migration\". Bad: \"You are looking at apps/anyone/web and should check every component…\".",
|
|
2764
4261
|
maxLength: 120
|
|
2765
4262
|
}),
|
|
2766
4263
|
persona: Type.Optional(Type.String({ description: "Optional extra system prompt / role for this task, applied ON TOP of the child run's own default persona (your full identity and soul are still there underneath). Omit to run with just your default persona." })),
|
|
2767
|
-
model: Type.Optional(Type.String({ description: "Optional model id to run this subagent on. PREFER A LOWER-COST, FASTER MODEL when the task is well-scoped and does not need your full reasoning depth — most delegated subtasks (searching, summarizing, mechanical edits, gathering or reformatting data, running a check) run just as well on a lighter model and cost far less. Reserve a top-tier model for subtasks that genuinely need deep reasoning or careful judgment. Must be a real catalogued model id. Omit to inherit your own model. If you are locked to a Google-compliant model, only compliant models are accepted." }))
|
|
4264
|
+
model: Type.Optional(Type.String({ description: "Optional model id to run this subagent on. PREFER A LOWER-COST, FASTER MODEL when the task is well-scoped and does not need your full reasoning depth — most delegated subtasks (searching, summarizing, mechanical edits, gathering or reformatting data, running a check) run just as well on a lighter model and cost far less. Reserve a top-tier model for subtasks that genuinely need deep reasoning or careful judgment. Must be a real catalogued model id. Omit to inherit your own model. If you are locked to a Google-compliant model, only compliant models are accepted." })),
|
|
4265
|
+
timeoutMinutes: Type.Optional(Type.Integer({
|
|
4266
|
+
description: "Optional. Minutes until your next check in if the subagent is still running. You are woken to check progress; the child keeps working. If it needs more time, re-arm your next check in with tasks: [{ taskId, timeoutMinutes }]. Completion wakes you sooner. Default: 30 minutes. Range: 1–360.",
|
|
4267
|
+
minimum: 1,
|
|
4268
|
+
maximum: 360
|
|
4269
|
+
}))
|
|
2768
4270
|
});
|
|
4271
|
+
const RearmTaskItem = Type.Object({
|
|
4272
|
+
taskId: Type.String({ description: "The running subagent task to keep waiting for. Use the taskId returned by an earlier subagent call." }),
|
|
4273
|
+
timeoutMinutes: Type.Integer({
|
|
4274
|
+
description: "Wake again after this many minutes if the subagent is still running. This replaces no work and can be repeated after every timeout. Range 1-360.",
|
|
4275
|
+
minimum: 1,
|
|
4276
|
+
maximum: 360
|
|
4277
|
+
})
|
|
4278
|
+
});
|
|
4279
|
+
const TaskItem = Type.Union([SpawnTaskItem, RearmTaskItem]);
|
|
4280
|
+
const DEFAULT_SUBAGENT_TIMEOUT_MS = 30 * 6e4;
|
|
2769
4281
|
const SubagentParams = Type.Object({ tasks: Type.Array(TaskItem, {
|
|
2770
|
-
description: "
|
|
4282
|
+
description: "Spawn subagent runs with { task, title, ... }, or keep waiting for a running task with { taskId, timeoutMinutes }. Re-arming schedules another parent wake without restarting or stopping the child.",
|
|
2771
4283
|
minItems: 1,
|
|
2772
4284
|
maxItems: MAX_TASKS
|
|
2773
4285
|
}) });
|
|
@@ -2777,10 +4289,12 @@ function buildTool(messageId) {
|
|
|
2777
4289
|
name: SUBAGENT_TOOL_NAME,
|
|
2778
4290
|
label: "Subagent",
|
|
2779
4291
|
description: [
|
|
2780
|
-
"Delegate
|
|
4292
|
+
"Delegate tasks to isolated subagent runs, or re-arm a running task so it wakes you again later without stopping the child.",
|
|
2781
4293
|
"Use it to parallelize independent work, to keep a large or noisy subtask out of your own context, or to run a task under a specialized persona.",
|
|
2782
4294
|
"Fire-and-forget: this returns immediately after queueing. It does NOT wait for results. Each subagent runs on its own and, when it finishes, sends you its result on this thread — so queue the work, then keep going or end your turn. To chain, re-delegate after a result lands.",
|
|
2783
|
-
"
|
|
4295
|
+
"To spawn, pass { task, title, persona?, model?, timeoutMinutes? }. To keep waiting after a timeout, pass { taskId, timeoutMinutes }; this schedules another wake and leaves the same child running. title is the short human-visible name for new work. persona and model apply only when spawning.",
|
|
4296
|
+
"Peering: each queued task comes back with its own conversation id. A subagent is a real linked conversation, so to see what one is doing RIGHT NOW while it runs — its reasoning, the tools it has called and their results, its progress — read that conversation with `platform conversations show <conversationId>` (you are already authorized; it is your own delegated run). Check in that way instead of waiting blind for the final result; never sleep before a check-in, that blocks your turn pointlessly. If the child needs more time, re-arm it with tasks: [{ taskId, timeoutMinutes }] and end your turn; the timeout wake brings you back to react. The read reflects the child's persisted state, which lags a few seconds behind live (tool results land as they complete; in-progress reasoning can be up to ~5s stale), so it is a progress view for checkpoints, not something to poll in a tight loop.",
|
|
4297
|
+
"Steering: to add context, correct course, or answer a question a subagent needs mid-run, post to its conversation with `platform conversations post <conversationId> --message \"...\"`. If the subagent is still running, your message lands as a live steer picked up in that same turn; if it has gone idle, it queues as its next turn. This is the same primitive as any conversation message — there is no separate steer channel."
|
|
2784
4298
|
].join(" "),
|
|
2785
4299
|
promptSnippet: "subagent — delegate tasks to isolated subagent runs; each rewakes you with its result when done",
|
|
2786
4300
|
parameters: SubagentParams,
|
|
@@ -2794,39 +4308,102 @@ function buildTool(messageId) {
|
|
|
2794
4308
|
details: {},
|
|
2795
4309
|
isError: true
|
|
2796
4310
|
};
|
|
2797
|
-
const
|
|
2798
|
-
task
|
|
2799
|
-
|
|
2800
|
-
|
|
2801
|
-
|
|
4311
|
+
const spawnRequests = tasks.flatMap((task, index) => "task" in task ? [{
|
|
4312
|
+
task,
|
|
4313
|
+
index
|
|
4314
|
+
}] : []);
|
|
4315
|
+
const rearmRequests = tasks.flatMap((task, index) => "taskId" in task ? [{
|
|
4316
|
+
task,
|
|
4317
|
+
index
|
|
4318
|
+
}] : []);
|
|
4319
|
+
if (spawnRequests.length > 0 && rearmRequests.length > 0) return {
|
|
4320
|
+
content: [{
|
|
4321
|
+
type: "text",
|
|
4322
|
+
text: "A subagent call must either spawn new tasks or re-arm running tasks, not both."
|
|
4323
|
+
}],
|
|
4324
|
+
details: { error: "mixed subagent operations" }
|
|
4325
|
+
};
|
|
4326
|
+
const spawnTasks = spawnRequests.map(({ task }) => ({
|
|
4327
|
+
task: task.task,
|
|
4328
|
+
title: task.title ?? null,
|
|
4329
|
+
persona: task.persona ?? null,
|
|
4330
|
+
model: task.model ?? null,
|
|
4331
|
+
timeoutMs: task.timeoutMinutes != null ? task.timeoutMinutes * 6e4 : DEFAULT_SUBAGENT_TIMEOUT_MS
|
|
2802
4332
|
}));
|
|
2803
4333
|
try {
|
|
2804
|
-
const
|
|
4334
|
+
const spawned = spawnTasks.length ? await postSubagentSpawn({
|
|
2805
4335
|
messageId,
|
|
2806
4336
|
tasks: spawnTasks
|
|
2807
|
-
})
|
|
2808
|
-
|
|
4337
|
+
}) : {
|
|
4338
|
+
taskIds: [],
|
|
4339
|
+
tasks: []
|
|
4340
|
+
};
|
|
4341
|
+
const rearmed = await Promise.all(rearmRequests.map(async ({ task, index }) => {
|
|
4342
|
+
try {
|
|
4343
|
+
const result = await postSubagentRearm({
|
|
4344
|
+
messageId,
|
|
4345
|
+
taskId: task.taskId,
|
|
4346
|
+
timeoutMinutes: task.timeoutMinutes
|
|
4347
|
+
});
|
|
4348
|
+
return {
|
|
4349
|
+
index,
|
|
4350
|
+
timeoutMinutes: task.timeoutMinutes,
|
|
4351
|
+
...result
|
|
4352
|
+
};
|
|
4353
|
+
} catch (err) {
|
|
4354
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
4355
|
+
log$1.warn({
|
|
4356
|
+
err,
|
|
4357
|
+
event: "subagent_rearm_failed",
|
|
4358
|
+
taskId: task.taskId
|
|
4359
|
+
}, "subagent re-arm failed");
|
|
4360
|
+
return {
|
|
4361
|
+
index,
|
|
4362
|
+
taskId: task.taskId,
|
|
4363
|
+
error: message
|
|
4364
|
+
};
|
|
4365
|
+
}
|
|
4366
|
+
}));
|
|
4367
|
+
const { taskIds } = spawned;
|
|
4368
|
+
log$1.info({
|
|
2809
4369
|
event: "subagent_spawned",
|
|
2810
4370
|
count: taskIds.length
|
|
2811
4371
|
}, "subagent tasks queued");
|
|
2812
|
-
const
|
|
4372
|
+
const convByTask = new Map(spawned.tasks.map((t) => [t.taskId, t.conversationId]));
|
|
4373
|
+
const lines = tasks.map((_, index) => {
|
|
4374
|
+
const spawnIndex = spawnRequests.findIndex((request) => request.index === index);
|
|
4375
|
+
if (spawnIndex >= 0) {
|
|
4376
|
+
const id = taskIds[spawnIndex];
|
|
4377
|
+
const label = spawnTasks[spawnIndex]?.title ?? spawnTasks[spawnIndex]?.task ?? "";
|
|
4378
|
+
const conv = id ? convByTask.get(id) : void 0;
|
|
4379
|
+
return conv ? `- conversation ${conv}: ${label} (taskId ${id})` : `- ${label} (taskId ${id}, conversation id unavailable — cannot peer this run)`;
|
|
4380
|
+
}
|
|
4381
|
+
const rearm = rearmed.find((request) => request.index === index);
|
|
4382
|
+
if (!rearm) return "- task update unavailable";
|
|
4383
|
+
return "error" in rearm ? `- taskId ${rearm.taskId ?? "unknown"}: re-arm failed — ${rearm.error}` : `- taskId ${rearm.taskId}: wake again in ${rearm.timeoutMinutes} minutes`;
|
|
4384
|
+
}).join("\n");
|
|
4385
|
+
const peerHint = spawned.tasks.length ? "\nEach subagent runs on its own conversation (id shown per task above). To SEE what one is doing while it runs, read it with `platform conversations show <conversationId>`. To STEER one mid-run — add context, correct course, answer a question — post to its conversation with `platform conversations post <conversationId> --message \"...\"`; it lands as a live steer if the subagent is still running, or as its next turn if it has gone idle." : "";
|
|
2813
4386
|
return {
|
|
2814
4387
|
content: [{
|
|
2815
4388
|
type: "text",
|
|
2816
|
-
text: `Queued ${taskIds.length} subagent ${taskIds.length === 1 ? "run" : "runs"}.
|
|
4389
|
+
text: `${taskIds.length ? `Queued ${taskIds.length} subagent ${taskIds.length === 1 ? "run" : "runs"}.` : ""}${rearmed.length ? `${taskIds.length ? " " : ""}Re-armed ${rearmed.length} running ${rearmed.length === 1 ? "task" : "tasks"}.` : ""} Each will wake you on completion or at its next timeout.\n${lines}${peerHint}`
|
|
2817
4390
|
}],
|
|
2818
|
-
details: {
|
|
4391
|
+
details: {
|
|
4392
|
+
taskIds,
|
|
4393
|
+
tasks: spawned.tasks,
|
|
4394
|
+
rearmed
|
|
4395
|
+
}
|
|
2819
4396
|
};
|
|
2820
4397
|
} catch (err) {
|
|
2821
4398
|
const message = err instanceof Error ? err.message : String(err);
|
|
2822
|
-
log$
|
|
4399
|
+
log$1.warn({
|
|
2823
4400
|
err,
|
|
2824
4401
|
event: "subagent_spawn_failed"
|
|
2825
|
-
}, "subagent
|
|
4402
|
+
}, "subagent request failed");
|
|
2826
4403
|
return {
|
|
2827
4404
|
content: [{
|
|
2828
4405
|
type: "text",
|
|
2829
|
-
text: `Failed to
|
|
4406
|
+
text: `Failed to update subagent tasks: ${message}`
|
|
2830
4407
|
}],
|
|
2831
4408
|
details: {},
|
|
2832
4409
|
isError: true
|
|
@@ -2846,14 +4423,76 @@ function createSubagentExtension({ channelContext }) {
|
|
|
2846
4423
|
return (pi) => {
|
|
2847
4424
|
const messageId = extractMessageId(channelContext);
|
|
2848
4425
|
let registered = false;
|
|
2849
|
-
const registerOnce = () => {
|
|
4426
|
+
const registerOnce = () => {
|
|
4427
|
+
if (registered) return;
|
|
4428
|
+
registered = true;
|
|
4429
|
+
pi.registerTool(buildTool(messageId));
|
|
4430
|
+
log$1.info({ event: "subagent_enabled" }, "subagent tool registered");
|
|
4431
|
+
};
|
|
4432
|
+
pi.on("session_start", () => {
|
|
4433
|
+
registerOnce();
|
|
4434
|
+
});
|
|
4435
|
+
};
|
|
4436
|
+
}
|
|
4437
|
+
//#endregion
|
|
4438
|
+
//#region src/extensions/proactive-result.ts
|
|
4439
|
+
const contextSchema = z.object({
|
|
4440
|
+
proactiveAction: z.literal(true).optional(),
|
|
4441
|
+
runId: z.string().uuid(),
|
|
4442
|
+
messageId: z.string().uuid()
|
|
4443
|
+
});
|
|
4444
|
+
function createProactiveResultExtension({ channelContext }) {
|
|
4445
|
+
return (pi) => {
|
|
4446
|
+
let context;
|
|
4447
|
+
try {
|
|
4448
|
+
context = contextSchema.safeParse(JSON.parse(channelContext ?? ""));
|
|
4449
|
+
} catch (_error) {
|
|
4450
|
+
return;
|
|
4451
|
+
}
|
|
4452
|
+
if (!context.success || !context.data.proactiveAction) return;
|
|
4453
|
+
const { runId, messageId } = context.data;
|
|
4454
|
+
let registered = false;
|
|
4455
|
+
pi.on("session_start", async () => {
|
|
2850
4456
|
if (registered) return;
|
|
2851
|
-
|
|
2852
|
-
|
|
2853
|
-
|
|
2854
|
-
|
|
2855
|
-
|
|
2856
|
-
|
|
4457
|
+
const client = sandboxClient();
|
|
4458
|
+
if (!client) return;
|
|
4459
|
+
try {
|
|
4460
|
+
const response = await client["proactive-result"].$get({ query: {
|
|
4461
|
+
runId,
|
|
4462
|
+
messageId
|
|
4463
|
+
} });
|
|
4464
|
+
if (response.status === 404) return;
|
|
4465
|
+
if (!response.ok) throw new Error(`proactive result discovery failed: ${response.status}`);
|
|
4466
|
+
const { tool } = await response.json();
|
|
4467
|
+
if (!tool) return;
|
|
4468
|
+
pi.registerTool({
|
|
4469
|
+
name: tool.name,
|
|
4470
|
+
label: "Submit proactive result",
|
|
4471
|
+
description: tool.description,
|
|
4472
|
+
parameters: Type.Unsafe(tool.parameters),
|
|
4473
|
+
async execute(_toolCallId, result, signal) {
|
|
4474
|
+
const submitted = await client["proactive-result"].$post({ json: {
|
|
4475
|
+
runId,
|
|
4476
|
+
messageId,
|
|
4477
|
+
result
|
|
4478
|
+
} }, { init: { signal } });
|
|
4479
|
+
return {
|
|
4480
|
+
content: [{
|
|
4481
|
+
type: "text",
|
|
4482
|
+
text: await submitted.text()
|
|
4483
|
+
}],
|
|
4484
|
+
isError: !submitted.ok,
|
|
4485
|
+
details: {}
|
|
4486
|
+
};
|
|
4487
|
+
}
|
|
4488
|
+
});
|
|
4489
|
+
registered = true;
|
|
4490
|
+
} catch (err) {
|
|
4491
|
+
logger.warn({
|
|
4492
|
+
err,
|
|
4493
|
+
runId
|
|
4494
|
+
}, "proactive result tool unavailable");
|
|
4495
|
+
}
|
|
2857
4496
|
});
|
|
2858
4497
|
};
|
|
2859
4498
|
}
|
|
@@ -2884,212 +4523,31 @@ const toolCallEnvExtension = (pi) => {
|
|
|
2884
4523
|
});
|
|
2885
4524
|
};
|
|
2886
4525
|
//#endregion
|
|
2887
|
-
//#region src/extensions/tool-
|
|
2888
|
-
|
|
2889
|
-
|
|
2890
|
-
|
|
2891
|
-
|
|
2892
|
-
|
|
2893
|
-
|
|
2894
|
-
|
|
2895
|
-
|
|
2896
|
-
|
|
2897
|
-
|
|
2898
|
-
|
|
2899
|
-
|
|
2900
|
-
|
|
2901
|
-
|
|
2902
|
-
properties: z.record(z.string(), z.unknown()).optional(),
|
|
2903
|
-
required: z.array(z.string()).optional(),
|
|
2904
|
-
additionalProperties: z.unknown().optional()
|
|
2905
|
-
}).passthrough();
|
|
2906
|
-
const toolEntrySchema = z.object({
|
|
2907
|
-
name: z.string().optional(),
|
|
2908
|
-
input_schema: jsonSchemaObjectSchema.optional(),
|
|
2909
|
-
parameters: jsonSchemaObjectSchema.optional(),
|
|
2910
|
-
function: z.object({
|
|
2911
|
-
name: z.string().optional(),
|
|
2912
|
-
parameters: jsonSchemaObjectSchema.optional()
|
|
2913
|
-
}).passthrough().optional()
|
|
2914
|
-
}).passthrough();
|
|
2915
|
-
const payloadWithToolsSchema = z.object({ tools: z.array(z.unknown()) }).passthrough();
|
|
2916
|
-
/**
|
|
2917
|
-
* Add the summary property to one JSON Schema object. Returns the augmented
|
|
2918
|
-
* copy, or `null` when the tool should be left untouched: a strict schema
|
|
2919
|
-
* (`additionalProperties: false`) whose validation would reject the extra
|
|
2920
|
-
* field, or one that already declares a `__skydive_summary__` property of its own.
|
|
2921
|
-
*/
|
|
2922
|
-
function augmentSchema(schema) {
|
|
2923
|
-
if (schema.additionalProperties === false) return null;
|
|
2924
|
-
const properties = schema.properties ?? {};
|
|
2925
|
-
if ("__skydive_summary__" in properties) return null;
|
|
2926
|
-
const required = schema.required ?? [];
|
|
2927
|
-
return {
|
|
2928
|
-
...schema,
|
|
2929
|
-
type: schema.type ?? "object",
|
|
2930
|
-
properties: {
|
|
2931
|
-
[TOOL_CALL_SUMMARY_FIELD]: SUMMARY_PROPERTY,
|
|
2932
|
-
...properties
|
|
2933
|
-
},
|
|
2934
|
-
required: required.includes("__skydive_summary__") ? required : [...required, TOOL_CALL_SUMMARY_FIELD]
|
|
2935
|
-
};
|
|
2936
|
-
}
|
|
2937
|
-
/**
|
|
2938
|
-
* Augment a single tool entry, dispatching on which provider shape it is.
|
|
2939
|
-
* Returns the (possibly rebuilt) entry and whether anything changed. Skipped
|
|
2940
|
-
* tools — wrong shape, strict, or name in `strictToolNames` — return unchanged.
|
|
2941
|
-
*/
|
|
2942
|
-
function augmentToolEntry(entry, strictToolNames) {
|
|
2943
|
-
const parsed = toolEntrySchema.safeParse(entry);
|
|
2944
|
-
if (!parsed.success) return {
|
|
2945
|
-
entry,
|
|
2946
|
-
changed: false
|
|
2947
|
-
};
|
|
2948
|
-
const tool = parsed.data;
|
|
2949
|
-
const name = tool.name ?? tool.function?.name ?? null;
|
|
2950
|
-
if (name !== null && strictToolNames.has(name)) return {
|
|
2951
|
-
entry,
|
|
2952
|
-
changed: false
|
|
2953
|
-
};
|
|
2954
|
-
if (tool.input_schema) {
|
|
2955
|
-
const augmented = augmentSchema(tool.input_schema);
|
|
2956
|
-
if (!augmented) return {
|
|
2957
|
-
entry,
|
|
2958
|
-
changed: false
|
|
2959
|
-
};
|
|
2960
|
-
return {
|
|
2961
|
-
entry: {
|
|
2962
|
-
...tool,
|
|
2963
|
-
input_schema: augmented
|
|
2964
|
-
},
|
|
2965
|
-
changed: true
|
|
2966
|
-
};
|
|
2967
|
-
}
|
|
2968
|
-
if (tool.parameters) {
|
|
2969
|
-
const augmented = augmentSchema(tool.parameters);
|
|
2970
|
-
if (!augmented) return {
|
|
2971
|
-
entry,
|
|
2972
|
-
changed: false
|
|
2973
|
-
};
|
|
2974
|
-
return {
|
|
2975
|
-
entry: {
|
|
2976
|
-
...tool,
|
|
2977
|
-
parameters: augmented
|
|
2978
|
-
},
|
|
2979
|
-
changed: true
|
|
2980
|
-
};
|
|
2981
|
-
}
|
|
2982
|
-
if (tool.function?.parameters) {
|
|
2983
|
-
const augmented = augmentSchema(tool.function.parameters);
|
|
2984
|
-
if (!augmented) return {
|
|
2985
|
-
entry,
|
|
2986
|
-
changed: false
|
|
4526
|
+
//#region src/extensions/tool-history.ts
|
|
4527
|
+
function historyToolName(name) {
|
|
4528
|
+
if (/^[a-zA-Z0-9_-]{1,64}$/.test(name)) return name;
|
|
4529
|
+
return `invalid_tool_${createHash("sha256").update(name).digest("hex").slice(0, 32)}`;
|
|
4530
|
+
}
|
|
4531
|
+
/** Repair provider-bound history, never tool execution or persisted messages. */
|
|
4532
|
+
function repairToolHistory(messages) {
|
|
4533
|
+
if (!messages.some((message) => message.role === "assistant" && message.content.some((block) => block.type === "toolCall" && historyToolName(block.name) !== block.name) || message.role === "toolResult" && historyToolName(message.toolName) !== message.toolName)) return messages;
|
|
4534
|
+
return messages.map((message) => {
|
|
4535
|
+
if (message.role === "assistant") return {
|
|
4536
|
+
...message,
|
|
4537
|
+
content: message.content.map((block) => block.type === "toolCall" ? {
|
|
4538
|
+
...block,
|
|
4539
|
+
name: historyToolName(block.name)
|
|
4540
|
+
} : block)
|
|
2987
4541
|
};
|
|
2988
|
-
return {
|
|
2989
|
-
|
|
2990
|
-
|
|
2991
|
-
function: {
|
|
2992
|
-
...tool.function,
|
|
2993
|
-
parameters: augmented
|
|
2994
|
-
}
|
|
2995
|
-
},
|
|
2996
|
-
changed: true
|
|
4542
|
+
if (message.role === "toolResult") return {
|
|
4543
|
+
...message,
|
|
4544
|
+
toolName: historyToolName(message.toolName)
|
|
2997
4545
|
};
|
|
2998
|
-
|
|
2999
|
-
return {
|
|
3000
|
-
entry,
|
|
3001
|
-
changed: false
|
|
3002
|
-
};
|
|
3003
|
-
}
|
|
3004
|
-
/**
|
|
3005
|
-
* Inject the summary field into every eligible tool in a provider payload.
|
|
3006
|
-
* Returns a new payload when at least one tool was augmented, or `undefined`
|
|
3007
|
-
* to signal "no change" (which keeps the original payload, per the
|
|
3008
|
-
* `before_provider_request` contract).
|
|
3009
|
-
*
|
|
3010
|
-
* @param payload The outgoing provider payload (shape varies by provider).
|
|
3011
|
-
* @param strictToolNames Names of tools whose registered schema is strict and
|
|
3012
|
-
* must be skipped to avoid validation errors.
|
|
3013
|
-
*/
|
|
3014
|
-
function injectToolCallSummary(payload, strictToolNames) {
|
|
3015
|
-
const parsed = payloadWithToolsSchema.safeParse(payload);
|
|
3016
|
-
if (!parsed.success || parsed.data.tools.length === 0) return void 0;
|
|
3017
|
-
let changed = false;
|
|
3018
|
-
const tools = parsed.data.tools.map((entry) => {
|
|
3019
|
-
const result = augmentToolEntry(entry, strictToolNames);
|
|
3020
|
-
if (result.changed) changed = true;
|
|
3021
|
-
return result.entry;
|
|
4546
|
+
return message;
|
|
3022
4547
|
});
|
|
3023
|
-
if (!changed) return void 0;
|
|
3024
|
-
return {
|
|
3025
|
-
...parsed.data,
|
|
3026
|
-
tools
|
|
3027
|
-
};
|
|
3028
|
-
}
|
|
3029
|
-
/**
|
|
3030
|
-
* Names of registered tools whose schema sets `additionalProperties: false`.
|
|
3031
|
-
* Pi validates the model's tool args against this registered schema, so the
|
|
3032
|
-
* injected field would make a strict tool's call fail validation — skip them.
|
|
3033
|
-
*/
|
|
3034
|
-
function getStrictToolNames(pi) {
|
|
3035
|
-
const names = /* @__PURE__ */ new Set();
|
|
3036
|
-
for (const tool of pi.getAllTools()) {
|
|
3037
|
-
const parsed = jsonSchemaObjectSchema.safeParse(tool.parameters);
|
|
3038
|
-
if (parsed.success && parsed.data.additionalProperties === false) names.add(tool.name);
|
|
3039
|
-
}
|
|
3040
|
-
return names;
|
|
3041
|
-
}
|
|
3042
|
-
function toolDeclaresSummaryParam(pi, toolName) {
|
|
3043
|
-
const tool = pi.getAllTools().find((candidate) => candidate.name === toolName);
|
|
3044
|
-
if (!tool) return false;
|
|
3045
|
-
const parsed = jsonSchemaObjectSchema.safeParse(tool.parameters);
|
|
3046
|
-
return parsed.success && parsed.data.properties != null && "__skydive_summary__" in parsed.data.properties;
|
|
3047
|
-
}
|
|
3048
|
-
/**
|
|
3049
|
-
* Remove the injected summary from a tool's execution input. No-op when the
|
|
3050
|
-
* field is absent, or when the tool genuinely declares a `__skydive_summary__`
|
|
3051
|
-
* parameter of its own (which we never inject into, so its value is real).
|
|
3052
|
-
* Mutates `input` in place, matching the `tool_call` contract.
|
|
3053
|
-
*
|
|
3054
|
-
* Fails open: this runs on the critical path of tool execution, and the
|
|
3055
|
-
* `getAllTools()` lookup can throw. On any error we leave `input` untouched
|
|
3056
|
-
* (the sentinel may pass through to the tool, but a bug here can never break
|
|
3057
|
-
* tool execution).
|
|
3058
|
-
*/
|
|
3059
|
-
function stripInjectedSummary(pi, toolName, input) {
|
|
3060
|
-
try {
|
|
3061
|
-
if (!("__skydive_summary__" in input)) return;
|
|
3062
|
-
if (toolDeclaresSummaryParam(pi, toolName)) return;
|
|
3063
|
-
delete input[TOOL_CALL_SUMMARY_FIELD];
|
|
3064
|
-
} catch (err) {
|
|
3065
|
-
log$1.error({
|
|
3066
|
-
err,
|
|
3067
|
-
event: "tool_call_summary_strip_failed",
|
|
3068
|
-
toolName
|
|
3069
|
-
}, "tool_call_summary strip failed; leaving tool input untouched");
|
|
3070
|
-
}
|
|
3071
|
-
}
|
|
3072
|
-
/**
|
|
3073
|
-
* Compute the rewritten payload for a `before_provider_request` event, failing
|
|
3074
|
-
* open: on any error the original payload is left untouched so a bug here can
|
|
3075
|
-
* never break an LLM call.
|
|
3076
|
-
*/
|
|
3077
|
-
function buildInjectedPayload(pi, payload) {
|
|
3078
|
-
try {
|
|
3079
|
-
return injectToolCallSummary(payload, getStrictToolNames(pi));
|
|
3080
|
-
} catch (err) {
|
|
3081
|
-
log$1.error({
|
|
3082
|
-
err,
|
|
3083
|
-
event: "tool_call_summary_injection_failed"
|
|
3084
|
-
}, "tool_call_summary injection failed; passing payload through unchanged");
|
|
3085
|
-
return;
|
|
3086
|
-
}
|
|
3087
4548
|
}
|
|
3088
|
-
const
|
|
3089
|
-
pi.on("
|
|
3090
|
-
pi.on("tool_call", (event) => {
|
|
3091
|
-
stripInjectedSummary(pi, event.toolName, event.input);
|
|
3092
|
-
});
|
|
4549
|
+
const toolHistoryExtension = (pi) => {
|
|
4550
|
+
pi.on("context", ({ messages }) => ({ messages: repairToolHistory(messages) }));
|
|
3093
4551
|
};
|
|
3094
4552
|
//#endregion
|
|
3095
4553
|
//#region src/extensions/background-tasks.ts
|
|
@@ -3131,21 +4589,43 @@ const toolCallSummaryExtension = (pi) => {
|
|
|
3131
4589
|
* from the origin messageId in its channel context (`resolveConversationFromApi`)
|
|
3132
4590
|
* — and every run of the same conversation resolves to the same id, keeping the
|
|
3133
4591
|
* shared map correctly scoped across turns. `bg_*`, the completion wake, and the
|
|
3134
|
-
* next-session injection all filter
|
|
3135
|
-
*
|
|
4592
|
+
* next-session injection all filter by a **scope key** — the resolved
|
|
4593
|
+
* conversation id, or, when a session's conversation is unresolvable (a bare
|
|
4594
|
+
* CLI session, or a run whose channel-context ref carries no messageId), a
|
|
4595
|
+
* sentinel unique to that one session instance. Comparing on the raw
|
|
4596
|
+
* `conversationId` would bucket every unresolvable session together under
|
|
4597
|
+
* `null` and leak one's completion wake / status / next-session injection into
|
|
4598
|
+
* another; the sentinel keeps each isolated so an agent never sees or is woken
|
|
4599
|
+
* by a task from a different chat. Only the output log
|
|
3136
4600
|
* spills to disk (/home/user/.anyone/bg-tasks/<id>.log) to avoid buffering a chatty job
|
|
3137
4601
|
* in memory; exit code and run state live on the in-memory task.
|
|
3138
4602
|
*
|
|
3139
|
-
* **
|
|
3140
|
-
*
|
|
3141
|
-
* respawn,
|
|
3142
|
-
*
|
|
3143
|
-
*
|
|
3144
|
-
*
|
|
3145
|
-
*
|
|
3146
|
-
*
|
|
3147
|
-
*
|
|
3148
|
-
*
|
|
4603
|
+
* **Cross-restart survival is OPT-IN, via `bg_run({ resumable: true })`.**
|
|
4604
|
+
* A non-resumable task's state lives only in the running harness process: a
|
|
4605
|
+
* harness restart (crash → supervisord respawn, `platform harness reload`, or
|
|
4606
|
+
* a sandbox recycle on idle timeout / redeploy / template rebuild) drops the
|
|
4607
|
+
* map and pi's exec children are reaped with it, and the task is gone — the
|
|
4608
|
+
* right behavior for a one-shot side-effecting command, which must never
|
|
4609
|
+
* silently re-run.
|
|
4610
|
+
*
|
|
4611
|
+
* A RESUMABLE task additionally checkpoints its *spec* (command, cwd, labels,
|
|
4612
|
+
* origin messageId — not its live output/exit state) to a durable, api-side
|
|
4613
|
+
* journal keyed by conversation (`/sandbox/bg-task-journal`, redis). On the
|
|
4614
|
+
* NEXT run's `session_start` — which usually lands on a *different*,
|
|
4615
|
+
* cold-provisioned sandbox, which is exactly why the journal is api-side and
|
|
4616
|
+
* not on the sandbox disk — the harness lists the journal and relaunches any
|
|
4617
|
+
* spec it isn't already running, keeping the original id and prepending a
|
|
4618
|
+
* `<resumed-after-restart>` banner so the agent knows it re-ran from scratch,
|
|
4619
|
+
* not continued. The spec is dropped from the journal when the task
|
|
4620
|
+
* finishes/kills. Because relaunch RE-EXECUTES the command, resumable is only
|
|
4621
|
+
* for idempotent, long-lived work (pollers, watchers, retry loops); the tool
|
|
4622
|
+
* description enforces this and the default is false.
|
|
4623
|
+
*
|
|
4624
|
+
* The idle-completion wake (a task finishing while the agent is idle) crosses
|
|
4625
|
+
* the sandbox → platform boundary via a fresh run (`bg-task-done`) for both
|
|
4626
|
+
* resumable and non-resumable tasks; the journal is a separate, additive layer
|
|
4627
|
+
* that only handles a task whose harness dies BEFORE it finishes. This stays
|
|
4628
|
+
* distinct from the scheduled-run (cron) system.
|
|
3149
4629
|
*/
|
|
3150
4630
|
const log = logger.child({ module: "background-tasks-ext" });
|
|
3151
4631
|
const ops = createLocalBashOperations();
|
|
@@ -3156,6 +4636,7 @@ const WATCHDOG_INTERVAL_MS = 3e4;
|
|
|
3156
4636
|
const KEEPALIVE_EVERY_MS = 6e4;
|
|
3157
4637
|
const KEEPALIVE_MAX_MS = 3600 * 1e3;
|
|
3158
4638
|
const STALL_HINT_AFTER_MS = 120 * 1e3;
|
|
4639
|
+
const PUBLISH_DEBOUNCE_MS = 300;
|
|
3159
4640
|
const MAX_LOG_BYTES = 100 * 1024 * 1024;
|
|
3160
4641
|
const DEFAULT_TAIL_LINES = 30;
|
|
3161
4642
|
const TAIL_READ_BYTES = 64 * 1024;
|
|
@@ -3163,11 +4644,12 @@ function taskLabel(meta) {
|
|
|
3163
4644
|
return `${meta.id} "${meta.description ?? meta.command.slice(0, 60)}"`;
|
|
3164
4645
|
}
|
|
3165
4646
|
let taskCounter = 0;
|
|
4647
|
+
let sessionScopeCounter = 0;
|
|
3166
4648
|
const tasks = /* @__PURE__ */ new Map();
|
|
3167
4649
|
let watchdogInterval = null;
|
|
3168
4650
|
let lastKeepaliveAt = 0;
|
|
3169
|
-
function
|
|
3170
|
-
return meta.
|
|
4651
|
+
function sameScope(meta, scopeKey) {
|
|
4652
|
+
return meta.scopeKey === scopeKey;
|
|
3171
4653
|
}
|
|
3172
4654
|
function logPath(id) {
|
|
3173
4655
|
return join(tasksDir(), `${id}.log`);
|
|
@@ -3251,7 +4733,9 @@ function watchdogTick(now) {
|
|
|
3251
4733
|
}
|
|
3252
4734
|
if (now - lastKeepaliveAt >= KEEPALIVE_EVERY_MS) {
|
|
3253
4735
|
lastKeepaliveAt = now;
|
|
3254
|
-
|
|
4736
|
+
const messageIds = new Set(running.map((task) => task.messageId));
|
|
4737
|
+
if (messageIds.size === 0) postHeartbeat({ messageId: null });
|
|
4738
|
+
else for (const messageId of messageIds) postHeartbeat({ messageId });
|
|
3255
4739
|
}
|
|
3256
4740
|
}
|
|
3257
4741
|
function stopWatchdog() {
|
|
@@ -3280,6 +4764,10 @@ function createBackgroundTasksExtension({ channelContext }) {
|
|
|
3280
4764
|
return id;
|
|
3281
4765
|
});
|
|
3282
4766
|
}
|
|
4767
|
+
const unresolvedScopeSentinel = `unresolved:${process.pid.toString(36)}:${(sessionScopeCounter += 1).toString(36)}`;
|
|
4768
|
+
function scopeKey() {
|
|
4769
|
+
return conversationId ?? unresolvedScopeSentinel;
|
|
4770
|
+
}
|
|
3283
4771
|
let agentActive = false;
|
|
3284
4772
|
pi.on("agent_start", async () => {
|
|
3285
4773
|
agentActive = true;
|
|
@@ -3307,13 +4795,16 @@ ${recentOutput}
|
|
|
3307
4795
|
</background-task-finished>
|
|
3308
4796
|
Run bg_logs for the full output.
|
|
3309
4797
|
|
|
3310
|
-
This is a background-task completion, not a message from the user.
|
|
4798
|
+
This is a background-task completion, not a message from the user.
|
|
4799
|
+
For a routine or expected completion, output nothing.
|
|
4800
|
+
Only reply if the outcome changes what the user should know or do,
|
|
4801
|
+
or if you were explicitly waiting to report it.`,
|
|
3311
4802
|
display: false
|
|
3312
4803
|
};
|
|
3313
4804
|
}
|
|
3314
4805
|
async function notifyCompletion(meta) {
|
|
3315
4806
|
if (meta.notified) return;
|
|
3316
|
-
if (agentActive &&
|
|
4807
|
+
if (agentActive && sameScope(meta, scopeKey())) {
|
|
3317
4808
|
meta.notified = true;
|
|
3318
4809
|
pi.sendMessage(await taskDoneMessage(meta), {
|
|
3319
4810
|
triggerTurn: true,
|
|
@@ -3344,9 +4835,83 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3344
4835
|
});
|
|
3345
4836
|
} else log.info({ taskId: meta.id }, "bg task completed idle with no origin message; deferring to next session_start");
|
|
3346
4837
|
}
|
|
3347
|
-
|
|
4838
|
+
const SNAPSHOT_TAIL_LINES = 40;
|
|
4839
|
+
let lastPublishedSignature = null;
|
|
4840
|
+
let publishInFlight = null;
|
|
4841
|
+
let publishQueued = false;
|
|
4842
|
+
function snapshotSignature(snapshot) {
|
|
4843
|
+
return JSON.stringify(snapshot.map((t) => ({
|
|
4844
|
+
id: t.id,
|
|
4845
|
+
state: t.state,
|
|
4846
|
+
exitCode: t.exitCode,
|
|
4847
|
+
killedReason: t.killedReason,
|
|
4848
|
+
outputTail: t.outputTail
|
|
4849
|
+
})));
|
|
4850
|
+
}
|
|
4851
|
+
async function doPublishSnapshot() {
|
|
4852
|
+
if (!messageId) return;
|
|
4853
|
+
const mine = [...tasks.values()].filter((t) => sameScope(t, scopeKey()));
|
|
4854
|
+
try {
|
|
4855
|
+
const snapshot = await Promise.all(mine.map(async (t) => ({
|
|
4856
|
+
id: t.id,
|
|
4857
|
+
command: t.command,
|
|
4858
|
+
description: t.description,
|
|
4859
|
+
startedAt: t.startedAt,
|
|
4860
|
+
state: t.running ? "running" : "finished",
|
|
4861
|
+
exitCode: t.exitCode,
|
|
4862
|
+
killedReason: t.killedReason,
|
|
4863
|
+
outputTail: await tailLog(t.id, SNAPSHOT_TAIL_LINES)
|
|
4864
|
+
})));
|
|
4865
|
+
const signature = snapshotSignature(snapshot);
|
|
4866
|
+
if (signature === lastPublishedSignature) return;
|
|
4867
|
+
lastPublishedSignature = signature;
|
|
4868
|
+
postBackgroundTasksSnapshot({
|
|
4869
|
+
messageId,
|
|
4870
|
+
tasks: snapshot
|
|
4871
|
+
});
|
|
4872
|
+
} catch (err) {
|
|
4873
|
+
log.debug({
|
|
4874
|
+
err,
|
|
4875
|
+
event: "bg_tasks_snapshot_build_failed"
|
|
4876
|
+
}, "building bg-tasks snapshot failed");
|
|
4877
|
+
}
|
|
4878
|
+
}
|
|
4879
|
+
async function publishSnapshotNow() {
|
|
4880
|
+
if (publishInFlight) {
|
|
4881
|
+
publishQueued = true;
|
|
4882
|
+
return;
|
|
4883
|
+
}
|
|
4884
|
+
publishInFlight = (async () => {
|
|
4885
|
+
try {
|
|
4886
|
+
do {
|
|
4887
|
+
publishQueued = false;
|
|
4888
|
+
await doPublishSnapshot();
|
|
4889
|
+
} while (publishQueued);
|
|
4890
|
+
} finally {
|
|
4891
|
+
publishInFlight = null;
|
|
4892
|
+
}
|
|
4893
|
+
})();
|
|
4894
|
+
await publishInFlight;
|
|
4895
|
+
}
|
|
4896
|
+
let publishTimer = null;
|
|
4897
|
+
function schedulePublishSnapshot() {
|
|
4898
|
+
if (publishTimer) return;
|
|
4899
|
+
publishTimer = setTimeout(() => {
|
|
4900
|
+
publishTimer = null;
|
|
4901
|
+
publishSnapshotNow();
|
|
4902
|
+
}, PUBLISH_DEBOUNCE_MS);
|
|
4903
|
+
publishTimer.unref?.();
|
|
4904
|
+
}
|
|
4905
|
+
async function flushPublishSnapshot() {
|
|
4906
|
+
if (publishTimer) {
|
|
4907
|
+
clearTimeout(publishTimer);
|
|
4908
|
+
publishTimer = null;
|
|
4909
|
+
}
|
|
4910
|
+
await publishSnapshotNow();
|
|
4911
|
+
}
|
|
4912
|
+
async function launchTask({ command, description, cwd, resumable = false, resumedFromJournal = false, id: providedId, startedAt: providedStartedAt }) {
|
|
3348
4913
|
taskCounter += 1;
|
|
3349
|
-
const id = `bg-${process.pid.toString(36)}-${taskCounter}`;
|
|
4914
|
+
const id = providedId ?? `bg-${process.pid.toString(36)}-${taskCounter}`;
|
|
3350
4915
|
try {
|
|
3351
4916
|
await mkdir(tasksDir(), { recursive: true });
|
|
3352
4917
|
} catch (err) {
|
|
@@ -3362,7 +4927,7 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3362
4927
|
taskId: id
|
|
3363
4928
|
}, "bg task log write failed");
|
|
3364
4929
|
});
|
|
3365
|
-
const startedAt = Date.now();
|
|
4930
|
+
const startedAt = providedStartedAt ?? Date.now();
|
|
3366
4931
|
const meta = {
|
|
3367
4932
|
id,
|
|
3368
4933
|
command: stripPlatformExportsForDisplay(command),
|
|
@@ -3370,6 +4935,7 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3370
4935
|
logBytes: 0,
|
|
3371
4936
|
lastOutputAt: startedAt,
|
|
3372
4937
|
conversationId,
|
|
4938
|
+
scopeKey: scopeKey(),
|
|
3373
4939
|
messageId,
|
|
3374
4940
|
description,
|
|
3375
4941
|
notified: false,
|
|
@@ -3377,9 +4943,23 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3377
4943
|
controller: new AbortController(),
|
|
3378
4944
|
running: true,
|
|
3379
4945
|
exitCode: null,
|
|
3380
|
-
error: null
|
|
4946
|
+
error: null,
|
|
4947
|
+
resumable,
|
|
4948
|
+
cwd,
|
|
4949
|
+
resumedFromJournal
|
|
3381
4950
|
};
|
|
3382
4951
|
tasks.set(id, meta);
|
|
4952
|
+
if (resumable && messageId) putBackgroundTaskJournalSpec({
|
|
4953
|
+
messageId,
|
|
4954
|
+
spec: {
|
|
4955
|
+
id,
|
|
4956
|
+
command,
|
|
4957
|
+
cwd,
|
|
4958
|
+
description,
|
|
4959
|
+
startedAt,
|
|
4960
|
+
messageId
|
|
4961
|
+
}
|
|
4962
|
+
});
|
|
3383
4963
|
ops.exec(command, cwd, {
|
|
3384
4964
|
onData: (chunk) => {
|
|
3385
4965
|
meta.logBytes += chunk.length;
|
|
@@ -3405,6 +4985,11 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3405
4985
|
taskId: id,
|
|
3406
4986
|
exitCode: meta.exitCode
|
|
3407
4987
|
}, "bg task finished");
|
|
4988
|
+
if (meta.resumable && meta.messageId) deleteBackgroundTaskJournalSpec({
|
|
4989
|
+
messageId: meta.messageId,
|
|
4990
|
+
taskId: id
|
|
4991
|
+
});
|
|
4992
|
+
schedulePublishSnapshot();
|
|
3408
4993
|
await notifyCompletion(meta);
|
|
3409
4994
|
});
|
|
3410
4995
|
ensureWatchdog();
|
|
@@ -3412,19 +4997,55 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3412
4997
|
taskId: id,
|
|
3413
4998
|
conversationId
|
|
3414
4999
|
}, "bg task started");
|
|
5000
|
+
schedulePublishSnapshot();
|
|
3415
5001
|
return meta;
|
|
3416
5002
|
}
|
|
3417
5003
|
function knownTaskIds() {
|
|
3418
|
-
return [...tasks.values()].filter((t) =>
|
|
5004
|
+
return [...tasks.values()].filter((t) => sameScope(t, scopeKey())).map((t) => t.id).join(", ") || "(none)";
|
|
5005
|
+
}
|
|
5006
|
+
let journalChecked = false;
|
|
5007
|
+
async function rehydrateJournaledTasks() {
|
|
5008
|
+
if (!messageId || journalChecked) return;
|
|
5009
|
+
journalChecked = true;
|
|
5010
|
+
const specs = await listBackgroundTaskJournalSpecs({ messageId });
|
|
5011
|
+
if (specs.length === 0) return;
|
|
5012
|
+
for (const spec of specs) {
|
|
5013
|
+
const live = tasks.get(spec.id);
|
|
5014
|
+
if (live && sameScope(live, scopeKey())) continue;
|
|
5015
|
+
log.info({
|
|
5016
|
+
taskId: spec.id,
|
|
5017
|
+
conversationId
|
|
5018
|
+
}, "relaunching journaled resumable bg task after restart");
|
|
5019
|
+
const meta = await launchTask({
|
|
5020
|
+
command: spec.command,
|
|
5021
|
+
description: spec.description,
|
|
5022
|
+
cwd: spec.cwd,
|
|
5023
|
+
resumable: true,
|
|
5024
|
+
resumedFromJournal: true,
|
|
5025
|
+
id: spec.id,
|
|
5026
|
+
startedAt: spec.startedAt
|
|
5027
|
+
});
|
|
5028
|
+
try {
|
|
5029
|
+
const stream = createWriteStream(logPath(meta.id), { flags: "a" });
|
|
5030
|
+
stream.write(`<resumed-after-restart>requeued and relaunched on a new sandbox after the previous one was drained; this is a fresh execution of the command from the start, not a continuation</resumed-after-restart>\n`);
|
|
5031
|
+
stream.end();
|
|
5032
|
+
} catch (err) {
|
|
5033
|
+
log.debug({
|
|
5034
|
+
err,
|
|
5035
|
+
taskId: meta.id
|
|
5036
|
+
}, "resume banner write failed");
|
|
5037
|
+
}
|
|
5038
|
+
}
|
|
3419
5039
|
}
|
|
3420
5040
|
pi.on("session_start", async () => {
|
|
3421
|
-
if (tasks.size === 0) return;
|
|
5041
|
+
if (tasks.size === 0 && (!messageId || journalChecked)) return;
|
|
3422
5042
|
await ensureConversationId();
|
|
3423
|
-
|
|
5043
|
+
await rehydrateJournaledTasks();
|
|
5044
|
+
for (const [id, meta] of tasks) if (sameScope(meta, scopeKey()) && !meta.running && meta.notified) {
|
|
3424
5045
|
tasks.delete(id);
|
|
3425
5046
|
await unlink(logPath(id)).catch(() => {});
|
|
3426
5047
|
}
|
|
3427
|
-
const unnotified = [...tasks.values()].filter((t) =>
|
|
5048
|
+
const unnotified = [...tasks.values()].filter((t) => sameScope(t, scopeKey()) && !t.notified && !t.running);
|
|
3428
5049
|
for (const meta of unnotified) {
|
|
3429
5050
|
meta.notified = true;
|
|
3430
5051
|
pi.sendMessage(await taskDoneMessage(meta));
|
|
@@ -3433,7 +5054,9 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3433
5054
|
conversationId,
|
|
3434
5055
|
count: unnotified.length
|
|
3435
5056
|
}, "injected completed bg tasks at session_start");
|
|
3436
|
-
if ([...tasks.values()].some((t) =>
|
|
5057
|
+
if ([...tasks.values()].some((t) => sameScope(t, scopeKey()) && t.running)) ensureWatchdog();
|
|
5058
|
+
lastPublishedSignature = null;
|
|
5059
|
+
await flushPublishSnapshot();
|
|
3437
5060
|
});
|
|
3438
5061
|
function err(text) {
|
|
3439
5062
|
return {
|
|
@@ -3450,11 +5073,11 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3450
5073
|
}
|
|
3451
5074
|
function resolveTask(taskId) {
|
|
3452
5075
|
const exact = tasks.get(taskId);
|
|
3453
|
-
if (exact &&
|
|
5076
|
+
if (exact && sameScope(exact, scopeKey())) return {
|
|
3454
5077
|
error: null,
|
|
3455
5078
|
meta: exact
|
|
3456
5079
|
};
|
|
3457
|
-
const matches = [...tasks.values()].filter((t) =>
|
|
5080
|
+
const matches = [...tasks.values()].filter((t) => sameScope(t, scopeKey()) && t.id.startsWith(taskId));
|
|
3458
5081
|
if (matches.length === 1) return {
|
|
3459
5082
|
error: null,
|
|
3460
5083
|
meta: matches[0]
|
|
@@ -3463,7 +5086,7 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3463
5086
|
return err(`Unknown task ${taskId}. Known tasks: ${knownTaskIds()}`);
|
|
3464
5087
|
}
|
|
3465
5088
|
function listTasks() {
|
|
3466
|
-
const mine = [...tasks.values()].filter((t) =>
|
|
5089
|
+
const mine = [...tasks.values()].filter((t) => sameScope(t, scopeKey()));
|
|
3467
5090
|
if (mine.length === 0) return "No background tasks.";
|
|
3468
5091
|
return mine.map((t) => {
|
|
3469
5092
|
const state = t.running ? "running" : t.exitCode !== null ? `exited ${t.exitCode}${t.killedReason ? ` (killed: ${t.killedReason})` : ""}` : t.killedReason ? `killed: ${t.killedReason}` : "ended";
|
|
@@ -3476,24 +5099,27 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3476
5099
|
const bgRun = {
|
|
3477
5100
|
name: "bg_run",
|
|
3478
5101
|
label: "Run in background",
|
|
3479
|
-
description: "Run a bash command in the background. Returns immediately with a task id; output streams to a log file. You are sent a message when it finishes — keep working or end your turn meanwhile. Use for anything over a couple of minutes (builds, batch jobs, retry loops, downloads). Inspect with bg_status / bg_logs, stop with bg_kill.",
|
|
3480
|
-
promptSnippet: "bg_run — run a long command without blocking; you are notified on completion",
|
|
5102
|
+
description: "Run a bash command in the background. Returns immediately with a task id; output streams to a log file. You are sent a message when it finishes — keep working or end your turn meanwhile. Use for anything over a couple of minutes (builds, batch jobs, retry loops, downloads). Inspect with bg_status / bg_logs, stop with bg_kill. Pass resumable:true ONLY for an idempotent, long-lived command (a poller/watcher/retry loop) that should be requeued and relaunched on your next run if the sandbox goes away before it finishes (a sandbox is drained and replaced with a fresh one, not restarted in place) — the requeue re-runs the command from scratch, so never mark a one-shot side-effecting job (a migration, an apply, a send) resumable.",
|
|
5103
|
+
promptSnippet: "bg_run — run a long command without blocking; you are notified on completion (resumable:true requeues the task onto a fresh sandbox if the current one is drained, idempotent commands only)",
|
|
3481
5104
|
parameters: Type.Object({
|
|
3482
5105
|
command: Type.String({ description: "Bash command to execute" }),
|
|
3483
|
-
description: Type.Optional(Type.String({ description: "Clear, concise description of what this command does in active voice (2-6 words)." }))
|
|
5106
|
+
description: Type.Optional(Type.String({ description: "Clear, concise description of what this command does in active voice (2-6 words)." })),
|
|
5107
|
+
resumable: Type.Optional(Type.Boolean({ description: "When true, this task is requeued and relaunched on your next run if the sandbox it is running on goes away before it finishes (a sandbox is drained and replaced with a fresh one, rather than restarted in place, so an in-flight task would otherwise be lost). Use ONLY for idempotent, long-lived commands (pollers, watchers, retry loops) — the requeue re-runs the command from scratch on the new sandbox, so never set this on a one-shot side-effecting command." }))
|
|
3484
5108
|
}),
|
|
3485
5109
|
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
|
|
3486
5110
|
await ensureConversationId();
|
|
3487
|
-
const { command, description = null } = params;
|
|
5111
|
+
const { command, description = null, resumable = false } = params;
|
|
3488
5112
|
const meta = await launchTask({
|
|
3489
5113
|
command,
|
|
3490
5114
|
description,
|
|
3491
|
-
cwd: ctx.cwd
|
|
5115
|
+
cwd: ctx.cwd,
|
|
5116
|
+
resumable
|
|
3492
5117
|
});
|
|
5118
|
+
const resumeNote = resumable && meta.messageId ? "\nResumable: if this sandbox is drained before the task finishes, it will be requeued and relaunched from the start on your next run." : resumable ? "\nNote: resumable was requested but this session has no durable conversation, so it will NOT be requeued if the sandbox is drained." : "";
|
|
3493
5119
|
return {
|
|
3494
5120
|
content: [{
|
|
3495
5121
|
type: "text",
|
|
3496
|
-
text: `Started background task ${taskLabel(meta)}.\nlog: ${logPath(meta.id)}\nYou will get a message when it finishes. Check on it with bg_status {"taskId":"${meta.id}"}.`
|
|
5122
|
+
text: `Started background task ${taskLabel(meta)}.\nlog: ${logPath(meta.id)}\nYou will get a message when it finishes. Check on it with bg_status {"taskId":"${meta.id}"}.${resumeNote}\n\nProgress is already visible to the user. Output nothing unless the user asked for an update.`
|
|
3497
5123
|
}],
|
|
3498
5124
|
details: {}
|
|
3499
5125
|
};
|
|
@@ -3522,7 +5148,7 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3522
5148
|
return {
|
|
3523
5149
|
content: [{
|
|
3524
5150
|
type: "text",
|
|
3525
|
-
text: await describeStatus(resolved.meta, lines)
|
|
5151
|
+
text: `${await describeStatus(resolved.meta, lines)}${resolved.meta.running ? "\n\nProgress is already visible to the user. Output nothing unless the user asked for an update." : ""}`
|
|
3526
5152
|
}],
|
|
3527
5153
|
details: {}
|
|
3528
5154
|
};
|
|
@@ -3613,6 +5239,7 @@ const all = [
|
|
|
3613
5239
|
localToolsExtension,
|
|
3614
5240
|
toolCallEnvExtension,
|
|
3615
5241
|
bashDefaultTimeoutExtension,
|
|
5242
|
+
diskGuardExtension,
|
|
3616
5243
|
toolCallSummaryExtension
|
|
3617
5244
|
];
|
|
3618
5245
|
/**
|
|
@@ -3623,6 +5250,7 @@ const all = [
|
|
|
3623
5250
|
*/
|
|
3624
5251
|
function platformExtensions({ sessionId, channelContext }) {
|
|
3625
5252
|
return [
|
|
5253
|
+
providerApiCompatExtension,
|
|
3626
5254
|
createPlatformExtensions({
|
|
3627
5255
|
sessionId,
|
|
3628
5256
|
channelContext
|
|
@@ -3631,8 +5259,11 @@ function platformExtensions({ sessionId, channelContext }) {
|
|
|
3631
5259
|
selfTraceExtension,
|
|
3632
5260
|
createBackgroundTasksExtension({ channelContext }),
|
|
3633
5261
|
createSubagentExtension({ channelContext }),
|
|
3634
|
-
|
|
5262
|
+
createProactiveResultExtension({ channelContext }),
|
|
5263
|
+
createContextManagementExtension(),
|
|
5264
|
+
resourcePressureWarningExtension,
|
|
5265
|
+
toolHistoryExtension
|
|
3635
5266
|
];
|
|
3636
5267
|
}
|
|
3637
5268
|
//#endregion
|
|
3638
|
-
export { all, createHarness, createHealthHandler, createPlatformEnvMiddleware, installToolUpdateAutoStop, isPlatformConfigLoaded, loadPlatformConfig, localToolsExtension, mcp_default as mcpExtension, memoryExtension, platformExtensions, runToolUpdateLoop, soulExtension, toolCallEnvExtension };
|
|
5269
|
+
export { all, createHarness, createHealthHandler, createPlatformEnvMiddleware, installToolUpdateAutoStop, isPlatformConfigLoaded, loadPlatformConfig, localToolsExtension, mcp_default as mcpExtension, memoryExtension, platformExtensions, providerApiCompatExtension, runClosingTextLoop, runToolUpdateLoop, soulExtension, toolCallEnvExtension, turnEndedWithoutVisibleText };
|