@skydiveai/pi-extensions 0.1.0-beta.211 → 0.1.0-beta.2113
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.mts +49 -3
- package/dist/index.mjs +2044 -676
- package/package.json +2 -9
package/dist/index.mjs
CHANGED
|
@@ -2,24 +2,27 @@ import { createRequire } from "node:module";
|
|
|
2
2
|
import { DefaultExecutionEventBusManager, DefaultRequestHandler, InMemoryTaskStore } from "@a2a-js/sdk/server";
|
|
3
3
|
import { UserBuilder, restHandler } from "@a2a-js/sdk/server/express";
|
|
4
4
|
import { buildAgentCard, chainMiddleware, composeHandlers, createAgentExecutor, createProtocolHandlers, getCurrentTraceparent, logger, mountAt, requestHeaders, requestUrl, webHandlerToMiddleware } from "@skydiveai/pi-server";
|
|
5
|
-
import { mkdir, open, readFile, readdir, stat, unlink } from "node:fs/promises";
|
|
5
|
+
import { mkdir, open, readFile, readdir, stat, unlink, writeFile } from "node:fs/promises";
|
|
6
6
|
import { basename, dirname, join, relative, resolve } from "node:path";
|
|
7
7
|
import { z } from "zod";
|
|
8
8
|
import { pathToFileURL } from "node:url";
|
|
9
9
|
import { CallToolResultSchema } from "@modelcontextprotocol/sdk/types.js";
|
|
10
10
|
import { Type } from "typebox";
|
|
11
11
|
import { Client } from "@modelcontextprotocol/sdk/client/index.js";
|
|
12
|
-
import { StdioClientTransport } from "@modelcontextprotocol/sdk/client/stdio.js";
|
|
12
|
+
import { StdioClientTransport, getDefaultEnvironment } from "@modelcontextprotocol/sdk/client/stdio.js";
|
|
13
13
|
import { StreamableHTTPClientTransport, StreamableHTTPError } from "@modelcontextprotocol/sdk/client/streamableHttp.js";
|
|
14
14
|
import { UnauthorizedError } from "@modelcontextprotocol/sdk/client/auth.js";
|
|
15
15
|
import { Check, Errors } from "typebox/value";
|
|
16
|
+
import { hc } from "hono/client";
|
|
16
17
|
import { ROOT_CONTEXT, SpanStatusCode, propagation, trace } from "@opentelemetry/api";
|
|
17
18
|
import { W3CTraceContextPropagator } from "@opentelemetry/core";
|
|
18
19
|
import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-http";
|
|
19
20
|
import { Resource } from "@opentelemetry/resources";
|
|
20
21
|
import { BatchSpanProcessor, NodeTracerProvider } from "@opentelemetry/sdk-trace-node";
|
|
21
22
|
import { ATTR_SERVICE_NAME } from "@opentelemetry/semantic-conventions";
|
|
22
|
-
import {
|
|
23
|
+
import { execFile } from "node:child_process";
|
|
24
|
+
import { availableParallelism } from "node:os";
|
|
25
|
+
import { promisify } from "node:util";
|
|
23
26
|
import { parse } from "yaml";
|
|
24
27
|
import { quote } from "shell-quote";
|
|
25
28
|
import { createWriteStream } from "node:fs";
|
|
@@ -256,7 +259,7 @@ function createHealthHandler({ metadata }) {
|
|
|
256
259
|
* read on the hot path before every LLM call), it falls back to the default
|
|
257
260
|
* for that knob and logs once.
|
|
258
261
|
*/
|
|
259
|
-
const log$
|
|
262
|
+
const log$15 = logger.child({ module: "context-management-config" });
|
|
260
263
|
const DEFAULT_CONTEXT_MANAGEMENT_CONFIG = {
|
|
261
264
|
enabled: false,
|
|
262
265
|
perResultMaxBytes: 16 * 1024,
|
|
@@ -304,7 +307,7 @@ function resolveContextManagementConfig(env = process.env) {
|
|
|
304
307
|
maxModelCallsPerTurn: env.SKYDIVE_CTX_MAX_MODEL_CALLS
|
|
305
308
|
});
|
|
306
309
|
if (!parsed.success) {
|
|
307
|
-
log$
|
|
310
|
+
log$15.warn({
|
|
308
311
|
event: "context_management_config_invalid",
|
|
309
312
|
err: parsed.error
|
|
310
313
|
}, "falling back to default context-management config");
|
|
@@ -415,6 +418,275 @@ function installIterationCap({ session, log }, configOverride = null) {
|
|
|
415
418
|
*/
|
|
416
419
|
const CAPABILITY_SOUL_NUDGE = "New capability gained — once the current task is done, if this changes what you can do for the user, record it in `soul.md` so it carries into future conversations rather than being rediscovered from scratch (then commit and push).";
|
|
417
420
|
//#endregion
|
|
421
|
+
//#region src/extensions/tool-call-summary.ts
|
|
422
|
+
const log$14 = logger.child({ module: "tool-call-summary-extension" });
|
|
423
|
+
/**
|
|
424
|
+
* The injected parameter name: a namespaced sentinel, so it can never collide
|
|
425
|
+
* with a real tool argument and is unmistakable in transcripts and logs. The
|
|
426
|
+
* frontend renderer (ANY-2723) duplicates this literal — keep the two in sync.
|
|
427
|
+
*/
|
|
428
|
+
const TOOL_CALL_SUMMARY_FIELD = "__skydive_summary__";
|
|
429
|
+
/** JSON Schema fragment for the injected parameter. */
|
|
430
|
+
const SUMMARY_PROPERTY = {
|
|
431
|
+
type: "string",
|
|
432
|
+
description: "Required for every tool call. A concise, specific summary (max ~8 words) of what THIS call does and why, written for a person watching the conversation, e.g. \"Searching feedback for billing complaints\" or \"Reading the auth middleware\". Address the user directly in second person: the summary is read by the user, so refer to their things as \"your\", never in third person — \"Reading your emails\", not \"Reading his emails\". Always use the present progressive tense, since it is shown while the call runs: \"Updating your Slack\", never \"Updated your Slack\". Make each summary distinct from your other tool calls; never reuse a generic label like \"Search query\" or \"Running command\"."
|
|
433
|
+
};
|
|
434
|
+
const jsonSchemaObjectSchema = z.object({
|
|
435
|
+
type: z.unknown().optional(),
|
|
436
|
+
properties: z.record(z.string(), z.unknown()).optional(),
|
|
437
|
+
required: z.array(z.string()).optional(),
|
|
438
|
+
additionalProperties: z.unknown().optional()
|
|
439
|
+
}).passthrough();
|
|
440
|
+
const toolEntrySchema = z.object({
|
|
441
|
+
name: z.string().optional(),
|
|
442
|
+
input_schema: jsonSchemaObjectSchema.optional(),
|
|
443
|
+
parameters: jsonSchemaObjectSchema.optional(),
|
|
444
|
+
function: z.object({
|
|
445
|
+
name: z.string().optional(),
|
|
446
|
+
parameters: jsonSchemaObjectSchema.optional()
|
|
447
|
+
}).passthrough().optional()
|
|
448
|
+
}).passthrough();
|
|
449
|
+
const payloadWithToolsSchema = z.object({ tools: z.array(z.unknown()) }).passthrough();
|
|
450
|
+
/**
|
|
451
|
+
* Add the summary property to one JSON Schema object. Returns the augmented
|
|
452
|
+
* copy, or `null` when the tool should be left untouched: a strict schema
|
|
453
|
+
* (`additionalProperties: false`) whose validation would reject the extra
|
|
454
|
+
* field, or one that already declares a `__skydive_summary__` property of its own.
|
|
455
|
+
*/
|
|
456
|
+
function augmentSchema(schema) {
|
|
457
|
+
if (schema.additionalProperties === false) return null;
|
|
458
|
+
const properties = schema.properties ?? {};
|
|
459
|
+
if ("__skydive_summary__" in properties) return null;
|
|
460
|
+
const required = schema.required ?? [];
|
|
461
|
+
return {
|
|
462
|
+
...schema,
|
|
463
|
+
type: schema.type ?? "object",
|
|
464
|
+
properties: {
|
|
465
|
+
[TOOL_CALL_SUMMARY_FIELD]: SUMMARY_PROPERTY,
|
|
466
|
+
...properties
|
|
467
|
+
},
|
|
468
|
+
required: required.includes("__skydive_summary__") ? required : [...required, TOOL_CALL_SUMMARY_FIELD]
|
|
469
|
+
};
|
|
470
|
+
}
|
|
471
|
+
/**
|
|
472
|
+
* Augment a single tool entry, dispatching on which provider shape it is.
|
|
473
|
+
* Returns the (possibly rebuilt) entry and whether anything changed. Skipped
|
|
474
|
+
* tools — wrong shape, strict, or name in `strictToolNames` — return unchanged.
|
|
475
|
+
*/
|
|
476
|
+
function augmentToolEntry(entry, strictToolNames) {
|
|
477
|
+
const parsed = toolEntrySchema.safeParse(entry);
|
|
478
|
+
if (!parsed.success) return {
|
|
479
|
+
entry,
|
|
480
|
+
changed: false
|
|
481
|
+
};
|
|
482
|
+
const tool = parsed.data;
|
|
483
|
+
const name = tool.name ?? tool.function?.name ?? null;
|
|
484
|
+
if (name !== null && strictToolNames.has(name)) return {
|
|
485
|
+
entry,
|
|
486
|
+
changed: false
|
|
487
|
+
};
|
|
488
|
+
if (tool.input_schema) {
|
|
489
|
+
const augmented = augmentSchema(tool.input_schema);
|
|
490
|
+
if (!augmented) return {
|
|
491
|
+
entry,
|
|
492
|
+
changed: false
|
|
493
|
+
};
|
|
494
|
+
return {
|
|
495
|
+
entry: {
|
|
496
|
+
...tool,
|
|
497
|
+
input_schema: augmented
|
|
498
|
+
},
|
|
499
|
+
changed: true
|
|
500
|
+
};
|
|
501
|
+
}
|
|
502
|
+
if (tool.parameters) {
|
|
503
|
+
const augmented = augmentSchema(tool.parameters);
|
|
504
|
+
if (!augmented) return {
|
|
505
|
+
entry,
|
|
506
|
+
changed: false
|
|
507
|
+
};
|
|
508
|
+
return {
|
|
509
|
+
entry: {
|
|
510
|
+
...tool,
|
|
511
|
+
parameters: augmented
|
|
512
|
+
},
|
|
513
|
+
changed: true
|
|
514
|
+
};
|
|
515
|
+
}
|
|
516
|
+
if (tool.function?.parameters) {
|
|
517
|
+
const augmented = augmentSchema(tool.function.parameters);
|
|
518
|
+
if (!augmented) return {
|
|
519
|
+
entry,
|
|
520
|
+
changed: false
|
|
521
|
+
};
|
|
522
|
+
return {
|
|
523
|
+
entry: {
|
|
524
|
+
...tool,
|
|
525
|
+
function: {
|
|
526
|
+
...tool.function,
|
|
527
|
+
parameters: augmented
|
|
528
|
+
}
|
|
529
|
+
},
|
|
530
|
+
changed: true
|
|
531
|
+
};
|
|
532
|
+
}
|
|
533
|
+
return {
|
|
534
|
+
entry,
|
|
535
|
+
changed: false
|
|
536
|
+
};
|
|
537
|
+
}
|
|
538
|
+
/**
|
|
539
|
+
* Inject the summary field into every eligible tool in a provider payload.
|
|
540
|
+
* Returns a new payload when at least one tool was augmented, or `undefined`
|
|
541
|
+
* to signal "no change" (which keeps the original payload, per the
|
|
542
|
+
* `before_provider_request` contract).
|
|
543
|
+
*
|
|
544
|
+
* @param payload The outgoing provider payload (shape varies by provider).
|
|
545
|
+
* @param strictToolNames Names of tools whose registered schema is strict and
|
|
546
|
+
* must be skipped to avoid validation errors.
|
|
547
|
+
*/
|
|
548
|
+
function injectToolCallSummary(payload, strictToolNames) {
|
|
549
|
+
const parsed = payloadWithToolsSchema.safeParse(payload);
|
|
550
|
+
if (!parsed.success || parsed.data.tools.length === 0) return void 0;
|
|
551
|
+
let changed = false;
|
|
552
|
+
const tools = parsed.data.tools.map((entry) => {
|
|
553
|
+
const result = augmentToolEntry(entry, strictToolNames);
|
|
554
|
+
if (result.changed) changed = true;
|
|
555
|
+
return result.entry;
|
|
556
|
+
});
|
|
557
|
+
if (!changed) return void 0;
|
|
558
|
+
return {
|
|
559
|
+
...parsed.data,
|
|
560
|
+
tools
|
|
561
|
+
};
|
|
562
|
+
}
|
|
563
|
+
/**
|
|
564
|
+
* Names of registered tools whose schema sets `additionalProperties: false`.
|
|
565
|
+
* Pi validates the model's tool args against this registered schema, so the
|
|
566
|
+
* injected field would make a strict tool's call fail validation — skip them.
|
|
567
|
+
*/
|
|
568
|
+
function getStrictToolNames(pi) {
|
|
569
|
+
const names = /* @__PURE__ */ new Set();
|
|
570
|
+
for (const tool of pi.getAllTools()) {
|
|
571
|
+
const parsed = jsonSchemaObjectSchema.safeParse(tool.parameters);
|
|
572
|
+
if (parsed.success && parsed.data.additionalProperties === false) names.add(tool.name);
|
|
573
|
+
}
|
|
574
|
+
return names;
|
|
575
|
+
}
|
|
576
|
+
/**
|
|
577
|
+
* Whether a tool's own schema declares a `__skydive_summary__` property. We
|
|
578
|
+
* never inject into such a tool, so any value it carries is a real argument
|
|
579
|
+
* and must be left alone.
|
|
580
|
+
*/
|
|
581
|
+
function schemaDeclaresSummary(parameters) {
|
|
582
|
+
const parsed = jsonSchemaObjectSchema.safeParse(parameters);
|
|
583
|
+
return parsed.success && parsed.data.properties != null && "__skydive_summary__" in parsed.data.properties;
|
|
584
|
+
}
|
|
585
|
+
function toolDeclaresSummaryParam(pi, toolName) {
|
|
586
|
+
const tool = pi.getAllTools().find((candidate) => candidate.name === toolName);
|
|
587
|
+
if (!tool) return false;
|
|
588
|
+
return schemaDeclaresSummary(tool.parameters);
|
|
589
|
+
}
|
|
590
|
+
/** A copy of `args` without the injected summary. Never mutates its input. */
|
|
591
|
+
function withoutInjectedSummary(args) {
|
|
592
|
+
if (args == null || typeof args !== "object" || Array.isArray(args)) return args;
|
|
593
|
+
if (!("__skydive_summary__" in args)) return args;
|
|
594
|
+
const { [TOOL_CALL_SUMMARY_FIELD]: _summary, ...rest } = args;
|
|
595
|
+
return rest;
|
|
596
|
+
}
|
|
597
|
+
/**
|
|
598
|
+
* Register a tool so the injected summary is removed *before* pi validates
|
|
599
|
+
* the model's arguments against the tool's schema.
|
|
600
|
+
*
|
|
601
|
+
* Needed because the `tool_call` strip below runs too late. pi-agent-core's
|
|
602
|
+
* `prepareToolCall` goes: `tool.prepareArguments` → `validateToolArguments` →
|
|
603
|
+
* `beforeToolCall` (which is what dispatches `tool_call`). A tool whose schema
|
|
604
|
+
* sets `additionalProperties: false` therefore rejects the summary and errors
|
|
605
|
+
* the call — client-side, before any request reaches the server — while the
|
|
606
|
+
* strip meant to prevent exactly that sits one step further down.
|
|
607
|
+
*
|
|
608
|
+
* `augmentSchema` already skips strict tools, but that only controls what the
|
|
609
|
+
* model is *told*. It still emits the field on those tools, because every
|
|
610
|
+
* other tool in the list declares it as required for every call. So a server
|
|
611
|
+
* can connect, bind its tools, present a healthy inventory, and have every
|
|
612
|
+
* call fail — and it reads as the server's fault when it is ours.
|
|
613
|
+
*
|
|
614
|
+
* Apply to tools registered from schemas we do not author: MCP servers and
|
|
615
|
+
* local `tools/*.ts`. Tools built here with `Type.Object(...)` do not need it
|
|
616
|
+
* (TypeBox emits no `additionalProperties`, so the field validates fine and
|
|
617
|
+
* the `tool_call` strip removes it in time).
|
|
618
|
+
*
|
|
619
|
+
* Fails open, like the rest of this module: if the strip throws, the original
|
|
620
|
+
* arguments are used rather than failing the call.
|
|
621
|
+
*/
|
|
622
|
+
function withSummaryStrippedBeforeValidation(tool) {
|
|
623
|
+
if (schemaDeclaresSummary(tool.parameters)) return tool;
|
|
624
|
+
const toolPrepare = tool.prepareArguments;
|
|
625
|
+
const prepareArguments = ((args) => {
|
|
626
|
+
let stripped = args;
|
|
627
|
+
try {
|
|
628
|
+
stripped = withoutInjectedSummary(args);
|
|
629
|
+
} catch (err) {
|
|
630
|
+
log$14.error({
|
|
631
|
+
err,
|
|
632
|
+
event: "tool_call_summary_prepare_strip_failed",
|
|
633
|
+
toolName: tool.name
|
|
634
|
+
}, "tool_call_summary pre-validation strip failed; leaving arguments untouched");
|
|
635
|
+
}
|
|
636
|
+
return toolPrepare ? toolPrepare(stripped) : stripped;
|
|
637
|
+
});
|
|
638
|
+
return {
|
|
639
|
+
...tool,
|
|
640
|
+
prepareArguments
|
|
641
|
+
};
|
|
642
|
+
}
|
|
643
|
+
/**
|
|
644
|
+
* Remove the injected summary from a tool's execution input. No-op when the
|
|
645
|
+
* field is absent, or when the tool genuinely declares a `__skydive_summary__`
|
|
646
|
+
* parameter of its own (which we never inject into, so its value is real).
|
|
647
|
+
* Mutates `input` in place, matching the `tool_call` contract.
|
|
648
|
+
*
|
|
649
|
+
* Fails open: this runs on the critical path of tool execution, and the
|
|
650
|
+
* `getAllTools()` lookup can throw. On any error we leave `input` untouched
|
|
651
|
+
* (the sentinel may pass through to the tool, but a bug here can never break
|
|
652
|
+
* tool execution).
|
|
653
|
+
*/
|
|
654
|
+
function stripInjectedSummary(pi, toolName, input) {
|
|
655
|
+
try {
|
|
656
|
+
if (!("__skydive_summary__" in input)) return;
|
|
657
|
+
if (toolDeclaresSummaryParam(pi, toolName)) return;
|
|
658
|
+
delete input[TOOL_CALL_SUMMARY_FIELD];
|
|
659
|
+
} catch (err) {
|
|
660
|
+
log$14.error({
|
|
661
|
+
err,
|
|
662
|
+
event: "tool_call_summary_strip_failed",
|
|
663
|
+
toolName
|
|
664
|
+
}, "tool_call_summary strip failed; leaving tool input untouched");
|
|
665
|
+
}
|
|
666
|
+
}
|
|
667
|
+
/**
|
|
668
|
+
* Compute the rewritten payload for a `before_provider_request` event, failing
|
|
669
|
+
* open: on any error the original payload is left untouched so a bug here can
|
|
670
|
+
* never break an LLM call.
|
|
671
|
+
*/
|
|
672
|
+
function buildInjectedPayload(pi, payload) {
|
|
673
|
+
try {
|
|
674
|
+
return injectToolCallSummary(payload, getStrictToolNames(pi));
|
|
675
|
+
} catch (err) {
|
|
676
|
+
log$14.error({
|
|
677
|
+
err,
|
|
678
|
+
event: "tool_call_summary_injection_failed"
|
|
679
|
+
}, "tool_call_summary injection failed; passing payload through unchanged");
|
|
680
|
+
return;
|
|
681
|
+
}
|
|
682
|
+
}
|
|
683
|
+
const toolCallSummaryExtension = (pi) => {
|
|
684
|
+
pi.on("before_provider_request", (event) => buildInjectedPayload(pi, event.payload));
|
|
685
|
+
pi.on("tool_call", (event) => {
|
|
686
|
+
stripInjectedSummary(pi, event.toolName, event.input);
|
|
687
|
+
});
|
|
688
|
+
};
|
|
689
|
+
//#endregion
|
|
418
690
|
//#region src/extensions/local-tools.ts
|
|
419
691
|
/**
|
|
420
692
|
* Local-tools adapter as a pi extension. Mirrors the mcp.ts hot-reload pattern
|
|
@@ -438,7 +710,7 @@ const CAPABILITY_SOUL_NUDGE = "New capability gained — once the current task i
|
|
|
438
710
|
* or `ToolDefinition[]`. Files starting with `_` or `.` are skipped, so
|
|
439
711
|
* `tools/_example.ts` documents the shape without registering.
|
|
440
712
|
*/
|
|
441
|
-
const log$
|
|
713
|
+
const log$13 = logger.child({ module: "local-tools-extension" });
|
|
442
714
|
const TOOLS_DIRNAME = "tools";
|
|
443
715
|
const fileState = /* @__PURE__ */ new Map();
|
|
444
716
|
let pendingLocalToolsUpdate = null;
|
|
@@ -565,7 +837,7 @@ async function reconcileLocalTools({ pi, dir }) {
|
|
|
565
837
|
action = existing ? "refreshed" : "added";
|
|
566
838
|
}
|
|
567
839
|
for (const tool of tools) {
|
|
568
|
-
pi.registerTool(withDefaultPromptSnippet(tool));
|
|
840
|
+
pi.registerTool(withSummaryStrippedBeforeValidation(withDefaultPromptSnippet(tool)));
|
|
569
841
|
summary.totalTools++;
|
|
570
842
|
}
|
|
571
843
|
if (action === "added") summary.added.push(file);
|
|
@@ -582,7 +854,7 @@ async function reconcileAndQueue({ pi, dir, reason }) {
|
|
|
582
854
|
dir
|
|
583
855
|
});
|
|
584
856
|
if (reason !== "session_start" && summaryHasChanges$1(summary)) pendingLocalToolsUpdate = summary;
|
|
585
|
-
log$
|
|
857
|
+
log$13.info({
|
|
586
858
|
event: "local_tools_reconcile",
|
|
587
859
|
reason,
|
|
588
860
|
total_tools: summary.totalTools,
|
|
@@ -604,7 +876,7 @@ const localToolsExtension = (pi) => {
|
|
|
604
876
|
reason: "session_start"
|
|
605
877
|
});
|
|
606
878
|
} catch (err) {
|
|
607
|
-
log$
|
|
879
|
+
log$13.error({
|
|
608
880
|
err,
|
|
609
881
|
event: "local_tools_reconcile_failed"
|
|
610
882
|
}, "local tools reconcile failed");
|
|
@@ -616,7 +888,7 @@ const localToolsExtension = (pi) => {
|
|
|
616
888
|
try {
|
|
617
889
|
current = await listToolFiles(dir);
|
|
618
890
|
} catch (err) {
|
|
619
|
-
log$
|
|
891
|
+
log$13.warn({
|
|
620
892
|
err,
|
|
621
893
|
event: "local_tools_listing_failed"
|
|
622
894
|
}, "tools/ listing failed");
|
|
@@ -638,7 +910,7 @@ const localToolsExtension = (pi) => {
|
|
|
638
910
|
reason: "auto_reload"
|
|
639
911
|
});
|
|
640
912
|
} catch (err) {
|
|
641
|
-
log$
|
|
913
|
+
log$13.error({
|
|
642
914
|
err,
|
|
643
915
|
event: "local_tools_auto_reload_failed"
|
|
644
916
|
}, "auto-reload after tools/ change failed");
|
|
@@ -685,6 +957,47 @@ function createStderrBuffer({ maxBytes }) {
|
|
|
685
957
|
const DEFAULT_CONNECT_TIMEOUT_MS = 5e3;
|
|
686
958
|
const STDERR_BUFFER_BYTES = 4096;
|
|
687
959
|
/**
|
|
960
|
+
* Sandbox egress variables a stdio MCP server needs to make outbound HTTPS
|
|
961
|
+
* calls. The SDK's default child environment is a tiny whitelist (HOME,
|
|
962
|
+
* LOGNAME, PATH, SHELL, TERM, USER on Linux), which drops NODE_EXTRA_CA_CERTS
|
|
963
|
+
* — but the sandbox's egress proxy terminates TLS with a private CA, so a
|
|
964
|
+
* child spawned without it fails every HTTPS request with a bare
|
|
965
|
+
* `fetch failed` (cause: UNABLE_TO_VERIFY_LEAF_SIGNATURE) that looks like the
|
|
966
|
+
* remote service is down. Same story for SSL_CERT_FILE (OpenSSL-based
|
|
967
|
+
* runtimes) and the *_PROXY set. Inherit them from this process so a stdio
|
|
968
|
+
* server gets working egress like every other process in the sandbox.
|
|
969
|
+
*/
|
|
970
|
+
const INHERITED_EGRESS_ENV_VARS = [
|
|
971
|
+
"NODE_EXTRA_CA_CERTS",
|
|
972
|
+
"SSL_CERT_FILE",
|
|
973
|
+
"SSL_CERT_DIR",
|
|
974
|
+
"REQUESTS_CA_BUNDLE",
|
|
975
|
+
"CURL_CA_BUNDLE",
|
|
976
|
+
"HTTP_PROXY",
|
|
977
|
+
"HTTPS_PROXY",
|
|
978
|
+
"NO_PROXY",
|
|
979
|
+
"http_proxy",
|
|
980
|
+
"https_proxy",
|
|
981
|
+
"no_proxy"
|
|
982
|
+
];
|
|
983
|
+
/**
|
|
984
|
+
* Environment for a stdio MCP server child: the SDK's safe defaults, plus the
|
|
985
|
+
* sandbox's egress/TLS variables, plus (last, so it wins) the server's own
|
|
986
|
+
* configured `env`. Building the merge here — instead of only when `env` is
|
|
987
|
+
* unset — also fixes the workaround trap where supplying any `env` in
|
|
988
|
+
* mcp.config.json silently replaced the ENTIRE default set, so a config that
|
|
989
|
+
* added one API key lost PATH/HOME and the server failed to spawn at all.
|
|
990
|
+
*/
|
|
991
|
+
function buildStdioEnv(configured, processEnv = process.env) {
|
|
992
|
+
const env = { ...getDefaultEnvironment() };
|
|
993
|
+
for (const key of INHERITED_EGRESS_ENV_VARS) {
|
|
994
|
+
const value = processEnv[key];
|
|
995
|
+
if (value !== void 0) env[key] = value;
|
|
996
|
+
}
|
|
997
|
+
if (configured) Object.assign(env, configured);
|
|
998
|
+
return env;
|
|
999
|
+
}
|
|
1000
|
+
/**
|
|
688
1001
|
* undici's fetch throws `TypeError: fetch failed` with the actual
|
|
689
1002
|
* network error hung off `.cause` (e.g. `getaddrinfo ENOTFOUND ...`,
|
|
690
1003
|
* `ECONNREFUSED`, TLS errors). Surfacing only `err.message` makes
|
|
@@ -692,6 +1005,19 @@ const STDERR_BUFFER_BYTES = 4096;
|
|
|
692
1005
|
* indistinguishable from any other transport problem. Walk the cause
|
|
693
1006
|
* chain so the agent sees the real underlying error.
|
|
694
1007
|
*/
|
|
1008
|
+
/**
|
|
1009
|
+
* True when an error from an http MCP transport (connect, listTools, or a tool
|
|
1010
|
+
* call) is an authentication failure. With no authProvider configured the SDK
|
|
1011
|
+
* surfaces a 401 as `StreamableHTTPError(401)`; older paths translate it to
|
|
1012
|
+
* `UnauthorizedError`. A dead/expired OAuth token (the proxy can no longer
|
|
1013
|
+
* mint one) shows up here on the NEXT request against a previously-connected
|
|
1014
|
+
* client — not just at connect — so reconcile must re-classify such a failure
|
|
1015
|
+
* as `pending_auth` instead of a generic `failed`, keeping the "waiting on
|
|
1016
|
+
* auth" report consistent with `platform auth`.
|
|
1017
|
+
*/
|
|
1018
|
+
function isUnauthorizedError(err) {
|
|
1019
|
+
return err instanceof UnauthorizedError || err instanceof StreamableHTTPError && err.code === 401;
|
|
1020
|
+
}
|
|
695
1021
|
function formatError(err) {
|
|
696
1022
|
if (!(err instanceof Error)) return String(err);
|
|
697
1023
|
const parts = [err.message];
|
|
@@ -703,6 +1029,15 @@ function formatError(err) {
|
|
|
703
1029
|
}
|
|
704
1030
|
return parts.join(": ");
|
|
705
1031
|
}
|
|
1032
|
+
function probeAlive(pid) {
|
|
1033
|
+
if (pid === null) return null;
|
|
1034
|
+
try {
|
|
1035
|
+
process.kill(pid, 0);
|
|
1036
|
+
return true;
|
|
1037
|
+
} catch {
|
|
1038
|
+
return false;
|
|
1039
|
+
}
|
|
1040
|
+
}
|
|
706
1041
|
async function connectHttp(_id, config, client) {
|
|
707
1042
|
const transport = new StreamableHTTPClientTransport(new URL(config.url), { ...config.headers !== null ? { requestInit: { headers: config.headers } } : {} });
|
|
708
1043
|
try {
|
|
@@ -713,12 +1048,12 @@ async function connectHttp(_id, config, client) {
|
|
|
713
1048
|
stderr: null
|
|
714
1049
|
};
|
|
715
1050
|
} catch (err) {
|
|
716
|
-
if (err
|
|
1051
|
+
if (isUnauthorizedError(err)) return {
|
|
717
1052
|
status: "pending_auth",
|
|
718
1053
|
client,
|
|
719
1054
|
stderr: "",
|
|
720
1055
|
stderrBuffer: null,
|
|
721
|
-
cliHint: `platform auth mcp ${config.url}`
|
|
1056
|
+
cliHint: `platform auth mcp ${config.url} --service "<Product>"`
|
|
722
1057
|
};
|
|
723
1058
|
return {
|
|
724
1059
|
status: "failed",
|
|
@@ -737,9 +1072,9 @@ async function connectClient(id, config, opts = {}) {
|
|
|
737
1072
|
const params = {
|
|
738
1073
|
command: config.command,
|
|
739
1074
|
args: config.args,
|
|
740
|
-
stderr: "pipe"
|
|
1075
|
+
stderr: "pipe",
|
|
1076
|
+
env: buildStdioEnv(config.env)
|
|
741
1077
|
};
|
|
742
|
-
if (config.env !== null) params.env = config.env;
|
|
743
1078
|
if (config.cwd !== null) params.cwd = config.cwd;
|
|
744
1079
|
const transport = new StdioClientTransport(params);
|
|
745
1080
|
const stderrBuffer = createStderrBuffer({ maxBytes: STDERR_BUFFER_BYTES });
|
|
@@ -751,6 +1086,10 @@ async function connectClient(id, config, opts = {}) {
|
|
|
751
1086
|
resolve();
|
|
752
1087
|
};
|
|
753
1088
|
});
|
|
1089
|
+
let stdoutMessages = 0;
|
|
1090
|
+
transport.onmessage = () => {
|
|
1091
|
+
if (stdoutMessages < 1e4) stdoutMessages += 1;
|
|
1092
|
+
};
|
|
754
1093
|
const connectPromise = client.connect(transport);
|
|
755
1094
|
const TIMEOUT_SENTINEL = Symbol("timeout");
|
|
756
1095
|
const result = await Promise.race([
|
|
@@ -768,12 +1107,20 @@ async function connectClient(id, config, opts = {}) {
|
|
|
768
1107
|
error: `server "${id}" exited before completing initialize`,
|
|
769
1108
|
stderr: stderrBuffer.read()
|
|
770
1109
|
};
|
|
771
|
-
if (result === TIMEOUT_SENTINEL)
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
|
|
775
|
-
|
|
776
|
-
|
|
1110
|
+
if (result === TIMEOUT_SENTINEL) {
|
|
1111
|
+
const pid = transport.pid;
|
|
1112
|
+
return {
|
|
1113
|
+
status: "timeout",
|
|
1114
|
+
client,
|
|
1115
|
+
stderr: stderrBuffer.read(),
|
|
1116
|
+
stderrBuffer,
|
|
1117
|
+
diagnostics: {
|
|
1118
|
+
pid,
|
|
1119
|
+
alive: probeAlive(pid),
|
|
1120
|
+
stdoutMessages
|
|
1121
|
+
}
|
|
1122
|
+
};
|
|
1123
|
+
}
|
|
777
1124
|
try {
|
|
778
1125
|
await client.close();
|
|
779
1126
|
} catch {}
|
|
@@ -784,13 +1131,133 @@ async function connectClient(id, config, opts = {}) {
|
|
|
784
1131
|
};
|
|
785
1132
|
}
|
|
786
1133
|
//#endregion
|
|
787
|
-
//#region src/extensions/mcp/
|
|
1134
|
+
//#region src/extensions/mcp/limits.ts
|
|
788
1135
|
/**
|
|
789
|
-
*
|
|
790
|
-
*
|
|
1136
|
+
* Tool-budget guardrails for MCP registration.
|
|
1137
|
+
*
|
|
1138
|
+
* Why this exists: the agent loop sends the *full* tool list — every tool's
|
|
1139
|
+
* name, description, and JSON-schema `parameters` — to the model on every
|
|
1140
|
+
* prompt. MCP servers add tools without bound: a handful of chatty servers (or
|
|
1141
|
+
* one server that exposes 100+ tools, or a few tools with enormous schemas) can
|
|
1142
|
+
* push the registered set past what the model can accept, and the request is
|
|
1143
|
+
* rejected before the turn even runs. Because `reconcile` re-registers from
|
|
1144
|
+
* `mcp.config.json` on *every* turn, an over-limit config bricks the harness on
|
|
1145
|
+
* a loop — the agent can't get a turn to run in order to edit the config back
|
|
1146
|
+
* down. Worse, the person can't tell *why*: tools just stop working.
|
|
1147
|
+
*
|
|
1148
|
+
* What actually overflows the request is *tokens*, not tool count — a few tools
|
|
1149
|
+
* with deeply-nested schemas and long descriptions cost more than a hundred
|
|
1150
|
+
* trivial ones. And how many tokens are safe depends on the *model*: a 200K
|
|
1151
|
+
* context window can afford far more tool surface than a 32K one. So the primary
|
|
1152
|
+
* limiter is a **token budget derived from the active model's context window**,
|
|
1153
|
+
* with a fixed tool-count cap as a coarse secondary guard (and the fallback
|
|
1154
|
+
* when the model — hence its window — isn't known at reconcile time).
|
|
1155
|
+
*
|
|
1156
|
+
* The fix is to make registration bounded and fail-soft. We register in
|
|
1157
|
+
* deterministic config order and stop before we blow the budget, recording how
|
|
1158
|
+
* much we dropped so the agent is *told* it hit the limit and which servers
|
|
1159
|
+
* were truncated. A config that would have bricked the harness now degrades to
|
|
1160
|
+
* "a bounded set of tools plus a loud warning", which the agent can act on by
|
|
1161
|
+
* pruning servers.
|
|
1162
|
+
*
|
|
1163
|
+
* Everything is env-overridable so the ceilings can be tuned per deployment
|
|
1164
|
+
* without a release, but ships with conservative defaults. A cap value of 0 (or
|
|
1165
|
+
* a non-finite / negative override) disables that cap — an explicit escape
|
|
1166
|
+
* hatch, not the default.
|
|
791
1167
|
*/
|
|
792
|
-
const
|
|
793
|
-
|
|
1168
|
+
const env = process.env;
|
|
1169
|
+
/**
|
|
1170
|
+
* Fraction of the model's context window we're willing to spend on MCP tool
|
|
1171
|
+
* schemas. Tool definitions are sent on every prompt, so they permanently eat
|
|
1172
|
+
* into the window available for the conversation, but a generous tool budget is
|
|
1173
|
+
* worth more than a marginally larger conversation window given how the harness
|
|
1174
|
+
* is used. 0.35 of a 200K window is ~70K tokens of tool schema, comfortably
|
|
1175
|
+
* more than any sane MCP setup; of a 32K window it's ~11.2K, which still forces
|
|
1176
|
+
* truncation before a small model chokes.
|
|
1177
|
+
*/
|
|
1178
|
+
const DEFAULT_MCP_TOOL_TOKEN_BUDGET_FRACTION = .35;
|
|
1179
|
+
/**
|
|
1180
|
+
* Floor for the token budget when the model's context window is unknown at
|
|
1181
|
+
* reconcile time (e.g. the model hasn't been resolved yet). Generous enough not
|
|
1182
|
+
* to truncate an ordinary tool set, low enough to still catch a runaway.
|
|
1183
|
+
*/
|
|
1184
|
+
const DEFAULT_MCP_TOOL_TOKEN_BUDGET_FLOOR = 16e3;
|
|
1185
|
+
/**
|
|
1186
|
+
* Read a positive-integer cap from an env var, falling back to `fallback`.
|
|
1187
|
+
* A `0` override (or any non-finite / negative value) means "no cap" and is
|
|
1188
|
+
* returned as `Infinity`, so callers can compare against it directly.
|
|
1189
|
+
*/
|
|
1190
|
+
function readCap(raw, fallback) {
|
|
1191
|
+
if (raw === void 0 || raw.trim() === "") return fallback;
|
|
1192
|
+
const parsed = Number(raw);
|
|
1193
|
+
if (!Number.isFinite(parsed) || parsed < 0) return Number.POSITIVE_INFINITY;
|
|
1194
|
+
if (parsed === 0) return Number.POSITIVE_INFINITY;
|
|
1195
|
+
return Math.floor(parsed);
|
|
1196
|
+
}
|
|
1197
|
+
/** Read a fraction in (0, 1] from an env var, falling back to `fallback`. */
|
|
1198
|
+
function readFraction(raw, fallback) {
|
|
1199
|
+
if (raw === void 0 || raw.trim() === "") return fallback;
|
|
1200
|
+
const parsed = Number(raw);
|
|
1201
|
+
if (!Number.isFinite(parsed) || parsed <= 0 || parsed > 1) return fallback;
|
|
1202
|
+
return parsed;
|
|
1203
|
+
}
|
|
1204
|
+
/** Read a non-negative integer from an env var, falling back to `fallback`. */
|
|
1205
|
+
function readNonNegativeInt(raw, fallback) {
|
|
1206
|
+
if (raw === void 0 || raw.trim() === "") return fallback;
|
|
1207
|
+
const parsed = Number(raw);
|
|
1208
|
+
if (!Number.isFinite(parsed) || parsed < 0) return fallback;
|
|
1209
|
+
return Math.floor(parsed);
|
|
1210
|
+
}
|
|
1211
|
+
/**
|
|
1212
|
+
* Resolve the active caps from the environment. Read once per reconcile so a
|
|
1213
|
+
* deployment can retune without a restart, cheap enough not to cache.
|
|
1214
|
+
*/
|
|
1215
|
+
function resolveMcpToolLimits() {
|
|
1216
|
+
return {
|
|
1217
|
+
maxTotalTools: readCap(env["SKYDIVE_MCP_MAX_TOTAL_TOOLS"], 128),
|
|
1218
|
+
maxToolsPerServer: readCap(env["SKYDIVE_MCP_MAX_TOOLS_PER_SERVER"], 50),
|
|
1219
|
+
tokenBudgetFraction: readFraction(env["SKYDIVE_MCP_TOOL_TOKEN_BUDGET_FRACTION"], DEFAULT_MCP_TOOL_TOKEN_BUDGET_FRACTION),
|
|
1220
|
+
tokenBudgetFloor: readNonNegativeInt(env["SKYDIVE_MCP_TOOL_TOKEN_BUDGET_FLOOR"], DEFAULT_MCP_TOOL_TOKEN_BUDGET_FLOOR)
|
|
1221
|
+
};
|
|
1222
|
+
}
|
|
1223
|
+
/**
|
|
1224
|
+
* The token budget for MCP tool schemas, given the active model's context
|
|
1225
|
+
* window (or undefined when the model isn't known yet).
|
|
1226
|
+
*
|
|
1227
|
+
* With a known window we spend `tokenBudgetFraction` of it, but never less than
|
|
1228
|
+
* the floor — a tiny window shouldn't collapse the budget to near-zero and
|
|
1229
|
+
* strand every tool. With no window we fall back to the floor outright.
|
|
1230
|
+
*/
|
|
1231
|
+
function resolveTokenBudget(limits, contextWindow) {
|
|
1232
|
+
if (contextWindow === void 0 || !Number.isFinite(contextWindow)) return limits.tokenBudgetFloor;
|
|
1233
|
+
return Math.max(limits.tokenBudgetFloor, Math.floor(contextWindow * limits.tokenBudgetFraction));
|
|
1234
|
+
}
|
|
1235
|
+
/**
|
|
1236
|
+
* Estimate the tokens an MCP tool's *definition* costs in the request. We
|
|
1237
|
+
* serialize what's actually sent to the model — the tool name, its description,
|
|
1238
|
+
* and its JSON-schema parameters — and apply pi's own ~chars/4 heuristic
|
|
1239
|
+
* (`estimateTokens` in the coding agent uses the same convention; there is no
|
|
1240
|
+
* per-provider tokenizer to lean on, and the provider's real usage is only
|
|
1241
|
+
* known after the response). Rough by design, but it tracks the true cost far
|
|
1242
|
+
* better than a flat per-tool count: a fat schema is charged for its fatness.
|
|
1243
|
+
*/
|
|
1244
|
+
function estimateToolTokens(tool) {
|
|
1245
|
+
let chars = tool.name.length + (tool.description?.length ?? 0);
|
|
1246
|
+
if (tool.inputSchema !== void 0 && tool.inputSchema !== null) try {
|
|
1247
|
+
chars += JSON.stringify(tool.inputSchema).length;
|
|
1248
|
+
} catch {
|
|
1249
|
+
chars += 256;
|
|
1250
|
+
}
|
|
1251
|
+
return Math.ceil(chars / 4);
|
|
1252
|
+
}
|
|
1253
|
+
//#endregion
|
|
1254
|
+
//#region src/extensions/mcp/mcp-config.ts
|
|
1255
|
+
/**
|
|
1256
|
+
* mcp.config.json schema + loader. Split out from the extension so it has
|
|
1257
|
+
* a small, focused unit-test surface (typebox validation, error formatting).
|
|
1258
|
+
*/
|
|
1259
|
+
const MCP_CONFIG_FILENAME = "mcp.config.json";
|
|
1260
|
+
const StdioServerSchema = Type.Object({
|
|
794
1261
|
transport: Type.Literal("stdio"),
|
|
795
1262
|
command: Type.String(),
|
|
796
1263
|
args: Type.Array(Type.String()),
|
|
@@ -863,7 +1330,7 @@ async function loadMcpConfig(path) {
|
|
|
863
1330
|
* Clients are keyed by JSON-stringified config and reused across
|
|
864
1331
|
* reloads — only changed configs reconnect.
|
|
865
1332
|
*/
|
|
866
|
-
const log$
|
|
1333
|
+
const log$12 = logger.child({ module: "mcp-extension" });
|
|
867
1334
|
async function closeConnected(connected) {
|
|
868
1335
|
try {
|
|
869
1336
|
await connected.client.close();
|
|
@@ -933,6 +1400,57 @@ function mcpResultToPiContent(result) {
|
|
|
933
1400
|
text: "MCP tool returned no content."
|
|
934
1401
|
}];
|
|
935
1402
|
}
|
|
1403
|
+
const MAX_ENUM_VALUES_LISTED = 12;
|
|
1404
|
+
function formatEnumValues(values) {
|
|
1405
|
+
const shown = values.slice(0, MAX_ENUM_VALUES_LISTED).map((v) => JSON.stringify(v));
|
|
1406
|
+
if (values.length > MAX_ENUM_VALUES_LISTED) shown.push(`… (${values.length} total)`);
|
|
1407
|
+
return shown.join(", ");
|
|
1408
|
+
}
|
|
1409
|
+
function extractEnum(propSchema) {
|
|
1410
|
+
if (Array.isArray(propSchema.enum) && propSchema.enum.length > 0) return propSchema.enum;
|
|
1411
|
+
const items = propSchema.items;
|
|
1412
|
+
if (items !== null && typeof items === "object" && Array.isArray(items.enum) && items.enum.length > 0) return items.enum;
|
|
1413
|
+
return null;
|
|
1414
|
+
}
|
|
1415
|
+
/**
|
|
1416
|
+
* Build a compact, human-readable constraint hint from an MCP tool's raw JSON
|
|
1417
|
+
* schema so the model sees required fields and each param's allowed enum values
|
|
1418
|
+
* inline in the tool description — the schema pass-through hands pi the server's
|
|
1419
|
+
* schema verbatim, but the model attends to the prose far more than the raw
|
|
1420
|
+
* `enum`/`required` keys. Returns '' when there is nothing worth surfacing.
|
|
1421
|
+
*/
|
|
1422
|
+
function describeToolConstraints(inputSchema) {
|
|
1423
|
+
if (inputSchema === null || typeof inputSchema !== "object") return "";
|
|
1424
|
+
const schema = inputSchema;
|
|
1425
|
+
const required = Array.isArray(schema.required) ? schema.required.filter((r) => typeof r === "string") : [];
|
|
1426
|
+
const properties = schema.properties !== null && typeof schema.properties === "object" ? schema.properties : {};
|
|
1427
|
+
const enumLines = [];
|
|
1428
|
+
for (const [propName, rawProp] of Object.entries(properties)) {
|
|
1429
|
+
if (rawProp === null || typeof rawProp !== "object") continue;
|
|
1430
|
+
const values = extractEnum(rawProp);
|
|
1431
|
+
if (values) enumLines.push(`- \`${propName}\` must be one of: ${formatEnumValues(values)}`);
|
|
1432
|
+
}
|
|
1433
|
+
if (required.length === 0 && enumLines.length === 0) return "";
|
|
1434
|
+
const lines = ["Argument constraints:"];
|
|
1435
|
+
if (required.length > 0) lines.push(`- required: ${required.map((r) => `\`${r}\``).join(", ")}`);
|
|
1436
|
+
lines.push(...enumLines);
|
|
1437
|
+
return lines.join("\n");
|
|
1438
|
+
}
|
|
1439
|
+
/**
|
|
1440
|
+
* Build the `promptSnippet` a registered MCP tool exposes to the model: the
|
|
1441
|
+
* server's description with the argument-constraint hint appended, falling back
|
|
1442
|
+
* to a labelled placeholder when the server omitted a description AND the schema
|
|
1443
|
+
* carries no constraints (pi's system-prompt builder hides tools whose snippet
|
|
1444
|
+
* is empty — system-prompt.js:49). This is the layer where `describeToolConstraints`
|
|
1445
|
+
* actually reaches the model, so it is exported and tested directly: the pure
|
|
1446
|
+
* hint being correct is necessary but not sufficient; what the model sees is
|
|
1447
|
+
* this composed snippet.
|
|
1448
|
+
*/
|
|
1449
|
+
function buildToolPromptSnippet({ description, inputSchema, serverId }) {
|
|
1450
|
+
const constraintHint = describeToolConstraints(inputSchema);
|
|
1451
|
+
const enriched = constraintHint.length > 0 ? `${description}${description.length > 0 ? "\n\n" : ""}${constraintHint}` : description;
|
|
1452
|
+
return enriched.length > 0 ? enriched : `MCP tool from server "${serverId}".`;
|
|
1453
|
+
}
|
|
936
1454
|
var McpExtension = class {
|
|
937
1455
|
mcpClients = /* @__PURE__ */ new Map();
|
|
938
1456
|
registeredMcpToolNames = /* @__PURE__ */ new Set();
|
|
@@ -952,8 +1470,12 @@ var McpExtension = class {
|
|
|
952
1470
|
const name = makeToolName(serverId, tool.name);
|
|
953
1471
|
const parameters = Type.Unsafe(tool.inputSchema);
|
|
954
1472
|
const description = tool.description?.trim() ?? "";
|
|
955
|
-
const promptSnippet =
|
|
956
|
-
|
|
1473
|
+
const promptSnippet = buildToolPromptSnippet({
|
|
1474
|
+
description,
|
|
1475
|
+
inputSchema: tool.inputSchema,
|
|
1476
|
+
serverId
|
|
1477
|
+
});
|
|
1478
|
+
pi.registerTool(withSummaryStrippedBeforeValidation({
|
|
957
1479
|
name,
|
|
958
1480
|
label: `MCP: ${serverId}/${tool.name}`,
|
|
959
1481
|
description,
|
|
@@ -987,22 +1509,29 @@ var McpExtension = class {
|
|
|
987
1509
|
};
|
|
988
1510
|
}
|
|
989
1511
|
}
|
|
990
|
-
});
|
|
1512
|
+
}));
|
|
991
1513
|
this.registeredMcpToolNames.add(name);
|
|
992
1514
|
}
|
|
993
|
-
async reconcile({ pi, configPath, connectTimeoutMs }) {
|
|
1515
|
+
async reconcile({ pi, configPath, connectTimeoutMs, contextWindow }) {
|
|
994
1516
|
let config;
|
|
995
1517
|
try {
|
|
996
1518
|
config = await loadMcpConfig(configPath);
|
|
997
1519
|
} catch (err) {
|
|
998
1520
|
throw new Error(`Failed to load MCP config: ${err instanceof Error ? err.message : String(err)}`);
|
|
999
1521
|
}
|
|
1522
|
+
const limits = resolveMcpToolLimits();
|
|
1523
|
+
const tokenBudget = resolveTokenBudget(limits, contextWindow);
|
|
1000
1524
|
const summary = {
|
|
1001
1525
|
added: [],
|
|
1002
1526
|
removed: [],
|
|
1003
1527
|
refreshed: [],
|
|
1004
1528
|
errors: [],
|
|
1005
1529
|
totalTools: 0,
|
|
1530
|
+
droppedTools: 0,
|
|
1531
|
+
limits,
|
|
1532
|
+
tokenBudget,
|
|
1533
|
+
tokensUsed: 0,
|
|
1534
|
+
toolCounts: {},
|
|
1006
1535
|
servers: {}
|
|
1007
1536
|
};
|
|
1008
1537
|
const desiredIds = new Set(Object.keys(config.servers));
|
|
@@ -1026,14 +1555,46 @@ var McpExtension = class {
|
|
|
1026
1555
|
if (outcome.error) summary.errors.push(outcome.error);
|
|
1027
1556
|
if (outcome.change === "added") summary.added.push(id);
|
|
1028
1557
|
else if (outcome.change === "refreshed") summary.refreshed.push(id);
|
|
1029
|
-
if (outcome.tools)
|
|
1030
|
-
|
|
1031
|
-
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
|
|
1035
|
-
|
|
1036
|
-
|
|
1558
|
+
if (outcome.tools) {
|
|
1559
|
+
const advertised = outcome.tools.list.length;
|
|
1560
|
+
const perServerRoom = Math.min(advertised, limits.maxToolsPerServer);
|
|
1561
|
+
let registered = 0;
|
|
1562
|
+
let serverTokens = 0;
|
|
1563
|
+
let droppedReason = null;
|
|
1564
|
+
for (const tool of outcome.tools.list) {
|
|
1565
|
+
if (summary.totalTools >= limits.maxTotalTools) {
|
|
1566
|
+
droppedReason = "total";
|
|
1567
|
+
break;
|
|
1568
|
+
}
|
|
1569
|
+
if (registered >= perServerRoom) {
|
|
1570
|
+
droppedReason = "per_server";
|
|
1571
|
+
break;
|
|
1572
|
+
}
|
|
1573
|
+
const cost = estimateToolTokens(tool);
|
|
1574
|
+
if (summary.tokensUsed + cost > tokenBudget && summary.totalTools > 0) {
|
|
1575
|
+
droppedReason = "tokens";
|
|
1576
|
+
break;
|
|
1577
|
+
}
|
|
1578
|
+
this.registerMcpTool({
|
|
1579
|
+
pi,
|
|
1580
|
+
serverId: id,
|
|
1581
|
+
client: outcome.tools.client,
|
|
1582
|
+
tool
|
|
1583
|
+
});
|
|
1584
|
+
registered++;
|
|
1585
|
+
serverTokens += cost;
|
|
1586
|
+
summary.totalTools++;
|
|
1587
|
+
summary.tokensUsed += cost;
|
|
1588
|
+
}
|
|
1589
|
+
const dropped = advertised - registered;
|
|
1590
|
+
if (dropped > 0) summary.droppedTools += dropped;
|
|
1591
|
+
else droppedReason = null;
|
|
1592
|
+
summary.toolCounts[id] = {
|
|
1593
|
+
advertised,
|
|
1594
|
+
registered,
|
|
1595
|
+
tokens: serverTokens,
|
|
1596
|
+
droppedReason
|
|
1597
|
+
};
|
|
1037
1598
|
}
|
|
1038
1599
|
}
|
|
1039
1600
|
return summary;
|
|
@@ -1099,7 +1660,8 @@ var McpExtension = class {
|
|
|
1099
1660
|
},
|
|
1100
1661
|
serverStatus: {
|
|
1101
1662
|
status: "timeout",
|
|
1102
|
-
stderr: result.stderr
|
|
1663
|
+
stderr: result.stderr,
|
|
1664
|
+
diagnostics: result.diagnostics
|
|
1103
1665
|
},
|
|
1104
1666
|
change,
|
|
1105
1667
|
error: null,
|
|
@@ -1146,7 +1708,8 @@ var McpExtension = class {
|
|
|
1146
1708
|
},
|
|
1147
1709
|
serverStatus: {
|
|
1148
1710
|
status: "timeout",
|
|
1149
|
-
stderr: retry.stderr
|
|
1711
|
+
stderr: retry.stderr,
|
|
1712
|
+
diagnostics: retry.diagnostics
|
|
1150
1713
|
},
|
|
1151
1714
|
change: null,
|
|
1152
1715
|
error: null,
|
|
@@ -1179,7 +1742,8 @@ var McpExtension = class {
|
|
|
1179
1742
|
store: connected,
|
|
1180
1743
|
serverStatus: {
|
|
1181
1744
|
status: "timeout",
|
|
1182
|
-
stderr
|
|
1745
|
+
stderr,
|
|
1746
|
+
diagnostics: null
|
|
1183
1747
|
},
|
|
1184
1748
|
change: null,
|
|
1185
1749
|
error: null,
|
|
@@ -1190,6 +1754,64 @@ var McpExtension = class {
|
|
|
1190
1754
|
try {
|
|
1191
1755
|
mcpTools = (await connected.client.listTools()).tools;
|
|
1192
1756
|
} catch (err) {
|
|
1757
|
+
if (isUnauthorizedError(err) && serverConfig.transport === "http") {
|
|
1758
|
+
await closeConnected(connected);
|
|
1759
|
+
const retry = await connectClient(id, serverConfig, { connectTimeoutMs });
|
|
1760
|
+
if (retry.status === "pending_auth") return {
|
|
1761
|
+
id,
|
|
1762
|
+
store: {
|
|
1763
|
+
client: retry.client,
|
|
1764
|
+
configKey,
|
|
1765
|
+
status: "pending_auth",
|
|
1766
|
+
stderrBuffer: null,
|
|
1767
|
+
cliHint: retry.cliHint
|
|
1768
|
+
},
|
|
1769
|
+
serverStatus: {
|
|
1770
|
+
status: "pending_auth",
|
|
1771
|
+
stderr: "",
|
|
1772
|
+
cliHint: retry.cliHint
|
|
1773
|
+
},
|
|
1774
|
+
change: null,
|
|
1775
|
+
error: null,
|
|
1776
|
+
tools: null
|
|
1777
|
+
};
|
|
1778
|
+
if (retry.status === "failed") return {
|
|
1779
|
+
id,
|
|
1780
|
+
store: null,
|
|
1781
|
+
serverStatus: {
|
|
1782
|
+
status: "failed",
|
|
1783
|
+
error: retry.error,
|
|
1784
|
+
stderr: retry.stderr
|
|
1785
|
+
},
|
|
1786
|
+
change: null,
|
|
1787
|
+
error: {
|
|
1788
|
+
serverId: id,
|
|
1789
|
+
message: retry.error
|
|
1790
|
+
},
|
|
1791
|
+
tools: null
|
|
1792
|
+
};
|
|
1793
|
+
if (retry.status === "connected") {
|
|
1794
|
+
connected = {
|
|
1795
|
+
client: retry.client,
|
|
1796
|
+
configKey,
|
|
1797
|
+
status: "connected",
|
|
1798
|
+
stderrBuffer: retry.stderr,
|
|
1799
|
+
cliHint: null
|
|
1800
|
+
};
|
|
1801
|
+
mcpTools = (await connected.client.listTools()).tools;
|
|
1802
|
+
return {
|
|
1803
|
+
id,
|
|
1804
|
+
store: connected,
|
|
1805
|
+
serverStatus: { status: "connected" },
|
|
1806
|
+
change: action === "reused" ? "refreshed" : action,
|
|
1807
|
+
error: null,
|
|
1808
|
+
tools: {
|
|
1809
|
+
client: connected.client,
|
|
1810
|
+
list: mcpTools
|
|
1811
|
+
}
|
|
1812
|
+
};
|
|
1813
|
+
}
|
|
1814
|
+
}
|
|
1193
1815
|
const message = err instanceof Error ? err.message : String(err);
|
|
1194
1816
|
const stderr = connected.stderrBuffer?.read() ?? "";
|
|
1195
1817
|
return {
|
|
@@ -1220,17 +1842,21 @@ var McpExtension = class {
|
|
|
1220
1842
|
}
|
|
1221
1843
|
};
|
|
1222
1844
|
}
|
|
1223
|
-
async reconcileAndRecordMtime({ pi, configPath, reason }) {
|
|
1845
|
+
async reconcileAndRecordMtime({ pi, configPath, reason, contextWindow }) {
|
|
1224
1846
|
const summary = await this.reconcile({
|
|
1225
1847
|
pi,
|
|
1226
|
-
configPath
|
|
1848
|
+
configPath,
|
|
1849
|
+
contextWindow
|
|
1227
1850
|
});
|
|
1228
1851
|
this.lastConfigMtimeMs = await readConfigMtimeMs(configPath);
|
|
1229
1852
|
if (reason !== "session_start" && summaryHasChanges(summary)) this.pendingMcpUpdate = summary;
|
|
1230
|
-
log$
|
|
1853
|
+
log$12.info({
|
|
1231
1854
|
event: "mcp_reconcile",
|
|
1232
1855
|
reason,
|
|
1233
1856
|
total_tools: summary.totalTools,
|
|
1857
|
+
tokens_used: summary.tokensUsed,
|
|
1858
|
+
token_budget: summary.tokenBudget,
|
|
1859
|
+
dropped_tools: summary.droppedTools,
|
|
1234
1860
|
added: summary.added,
|
|
1235
1861
|
removed: summary.removed,
|
|
1236
1862
|
refreshed: summary.refreshed,
|
|
@@ -1247,10 +1873,11 @@ var McpExtension = class {
|
|
|
1247
1873
|
await this.reconcileAndRecordMtime({
|
|
1248
1874
|
pi,
|
|
1249
1875
|
configPath,
|
|
1250
|
-
reason: "session_start"
|
|
1876
|
+
reason: "session_start",
|
|
1877
|
+
contextWindow: ctx.model?.contextWindow
|
|
1251
1878
|
});
|
|
1252
1879
|
} catch (err) {
|
|
1253
|
-
log$
|
|
1880
|
+
log$12.error({
|
|
1254
1881
|
err,
|
|
1255
1882
|
event: "mcp_reconcile_failed"
|
|
1256
1883
|
}, "MCP reconcile failed");
|
|
@@ -1262,7 +1889,7 @@ var McpExtension = class {
|
|
|
1262
1889
|
try {
|
|
1263
1890
|
mtime = await readConfigMtimeMs(configPath);
|
|
1264
1891
|
} catch (err) {
|
|
1265
|
-
log$
|
|
1892
|
+
log$12.warn({
|
|
1266
1893
|
err,
|
|
1267
1894
|
event: "mcp_mtime_check_failed"
|
|
1268
1895
|
}, "mtime check on mcp.config.json failed");
|
|
@@ -1273,10 +1900,11 @@ var McpExtension = class {
|
|
|
1273
1900
|
await this.reconcileAndRecordMtime({
|
|
1274
1901
|
pi,
|
|
1275
1902
|
configPath,
|
|
1276
|
-
reason: "auto_reload"
|
|
1903
|
+
reason: "auto_reload",
|
|
1904
|
+
contextWindow: ctx.model?.contextWindow
|
|
1277
1905
|
});
|
|
1278
1906
|
} catch (err) {
|
|
1279
|
-
log$
|
|
1907
|
+
log$12.error({
|
|
1280
1908
|
err,
|
|
1281
1909
|
event: "mcp_auto_reload_failed"
|
|
1282
1910
|
}, "auto-reload after mcp.config.json change failed");
|
|
@@ -1293,7 +1921,8 @@ var McpExtension = class {
|
|
|
1293
1921
|
const summary = await this.reconcileAndRecordMtime({
|
|
1294
1922
|
pi,
|
|
1295
1923
|
configPath,
|
|
1296
|
-
reason: "tool"
|
|
1924
|
+
reason: "tool",
|
|
1925
|
+
contextWindow: ctx.model?.contextWindow
|
|
1297
1926
|
});
|
|
1298
1927
|
return {
|
|
1299
1928
|
content: [{
|
|
@@ -1335,9 +1964,21 @@ function pendingAuthEntries(summary) {
|
|
|
1335
1964
|
function timeoutEntries(summary) {
|
|
1336
1965
|
return Object.entries(summary.servers).flatMap(([id, status]) => status.status === "timeout" ? [{
|
|
1337
1966
|
id,
|
|
1338
|
-
stderr: status.stderr
|
|
1967
|
+
stderr: status.stderr,
|
|
1968
|
+
diagnostics: status.diagnostics
|
|
1339
1969
|
}] : []);
|
|
1340
1970
|
}
|
|
1971
|
+
/**
|
|
1972
|
+
* One line describing what the connect probe learned about a timed-out
|
|
1973
|
+
* stdio child, so an empty stderr is no longer a dead end. Empty array
|
|
1974
|
+
* when the entry was re-surfaced without a fresh probe.
|
|
1975
|
+
*/
|
|
1976
|
+
function timeoutDiagnosticLines(diagnostics) {
|
|
1977
|
+
if (diagnostics === null) return [];
|
|
1978
|
+
const alive = diagnostics.alive === null ? "unknown (no pid)" : diagnostics.alive ? "alive" : "DEAD";
|
|
1979
|
+
const spoke = diagnostics.stdoutMessages === 0 ? "no JSON-RPC received on stdout (initialize never answered — slow startup, or blocked e.g. on OAuth)" : `${diagnostics.stdoutMessages} JSON-RPC message(s) received on stdout (server spoke but the handshake stalled)`;
|
|
1980
|
+
return [` child process: ${alive}${diagnostics.pid !== null ? ` (pid ${diagnostics.pid})` : ""}; ${spoke}`];
|
|
1981
|
+
}
|
|
1341
1982
|
function failedEntries(summary) {
|
|
1342
1983
|
return Object.entries(summary.servers).flatMap(([id, status]) => status.status === "failed" ? [{
|
|
1343
1984
|
id,
|
|
@@ -1345,6 +1986,13 @@ function failedEntries(summary) {
|
|
|
1345
1986
|
stderr: status.stderr
|
|
1346
1987
|
}] : []);
|
|
1347
1988
|
}
|
|
1989
|
+
function appendTimeoutStderrBlock(lines, stderr) {
|
|
1990
|
+
if (!stderr) {
|
|
1991
|
+
lines.push(" stderr: (empty — the child wrote nothing to stderr)");
|
|
1992
|
+
return;
|
|
1993
|
+
}
|
|
1994
|
+
appendStderrBlock(lines, stderr);
|
|
1995
|
+
}
|
|
1348
1996
|
function appendStderrBlock(lines, stderr) {
|
|
1349
1997
|
if (!stderr) return;
|
|
1350
1998
|
lines.push(" stderr:");
|
|
@@ -1352,9 +2000,29 @@ function appendStderrBlock(lines, stderr) {
|
|
|
1352
2000
|
for (const line of stderr.trimEnd().split("\n")) lines.push(` ${line}`);
|
|
1353
2001
|
lines.push(" ---");
|
|
1354
2002
|
}
|
|
2003
|
+
/**
|
|
2004
|
+
* Human-readable lines describing any budget-driven truncation, shared by
|
|
2005
|
+
* `summaryText` (the reload_mcp tool output) and `formatMcpUpdateMessage` (the
|
|
2006
|
+
* synthetic continuation). Empty when nothing was dropped.
|
|
2007
|
+
*/
|
|
2008
|
+
function truncationLines(summary) {
|
|
2009
|
+
if (summary.droppedTools <= 0) return [];
|
|
2010
|
+
const lines = [];
|
|
2011
|
+
const { maxTotalTools, maxToolsPerServer } = summary.limits;
|
|
2012
|
+
const totalCapHint = Number.isFinite(maxTotalTools) ? `${maxTotalTools}` : "unlimited";
|
|
2013
|
+
lines.push(` WARNING: ${summary.droppedTools} MCP tool(s) were NOT registered because a tool budget was hit (token budget: ~${summary.tokenBudget} tokens, used ~${summary.tokensUsed}; total-tool cap: ${totalCapHint}, per-server cap: ${Number.isFinite(maxToolsPerServer) ? maxToolsPerServer : "unlimited"}).`);
|
|
2014
|
+
lines.push(" Tool definitions are sent to the model on every prompt; too many (or too-large) tool schemas push the request over the limit, so they are budgeted against the model context window. Prune servers from mcp.config.json to bring the tool surface down.");
|
|
2015
|
+
for (const [id, count] of Object.entries(summary.toolCounts)) {
|
|
2016
|
+
if (count.droppedReason === null) continue;
|
|
2017
|
+
const why = count.droppedReason === "tokens" ? "token budget exhausted" : count.droppedReason === "total" ? "total-tool cap reached" : "per-server cap";
|
|
2018
|
+
lines.push(` - ${id}: registered ${count.registered}/${count.advertised} tools (~${count.tokens} tokens, ${why}).`);
|
|
2019
|
+
}
|
|
2020
|
+
return lines;
|
|
2021
|
+
}
|
|
1355
2022
|
function summaryText(summary) {
|
|
1356
2023
|
const lines = [];
|
|
1357
2024
|
lines.push(`MCP reconcile complete: ${summary.totalTools} tool(s) live.`);
|
|
2025
|
+
lines.push(...truncationLines(summary));
|
|
1358
2026
|
if (summary.added.length > 0) lines.push(` Added: ${summary.added.join(", ")}`);
|
|
1359
2027
|
if (summary.refreshed.length > 0) lines.push(` Refreshed: ${summary.refreshed.join(", ")}`);
|
|
1360
2028
|
if (summary.removed.length > 0) lines.push(` Removed: ${summary.removed.join(", ")}`);
|
|
@@ -1363,11 +2031,12 @@ function summaryText(summary) {
|
|
|
1363
2031
|
lines.push(` run \`${cliHint}\` in the shell to authenticate.`);
|
|
1364
2032
|
appendStderrBlock(lines, stderr);
|
|
1365
2033
|
}
|
|
1366
|
-
for (const { id, stderr } of timeoutEntries(summary)) {
|
|
2034
|
+
for (const { id, stderr, diagnostics } of timeoutEntries(summary)) {
|
|
1367
2035
|
lines.push(` TIMED OUT ${id}:`);
|
|
1368
2036
|
lines.push(` bridge spawned but did not complete initialize in time`);
|
|
1369
2037
|
lines.push(` (usually mcp-remote mid-OAuth — see captured stderr).`);
|
|
1370
|
-
|
|
2038
|
+
lines.push(...timeoutDiagnosticLines(diagnostics));
|
|
2039
|
+
appendTimeoutStderrBlock(lines, stderr);
|
|
1371
2040
|
}
|
|
1372
2041
|
for (const { id, error, stderr } of failedEntries(summary)) {
|
|
1373
2042
|
lines.push(` FAILED ${id}: ${error}`);
|
|
@@ -1380,7 +2049,7 @@ function summaryText(summary) {
|
|
|
1380
2049
|
return lines.join("\n");
|
|
1381
2050
|
}
|
|
1382
2051
|
function summaryHasChanges(summary) {
|
|
1383
|
-
return summary.added.length > 0 || summary.removed.length > 0 || summary.refreshed.length > 0 || summary.errors.length > 0;
|
|
2052
|
+
return summary.added.length > 0 || summary.removed.length > 0 || summary.refreshed.length > 0 || summary.errors.length > 0 || summary.droppedTools > 0;
|
|
1384
2053
|
}
|
|
1385
2054
|
/**
|
|
1386
2055
|
* Format a queued tool-update as a synthetic system-style message for
|
|
@@ -1393,6 +2062,10 @@ function formatMcpUpdateMessage(summary) {
|
|
|
1393
2062
|
if (summary.added.length > 0) lines.push(`Newly available servers: ${summary.added.join(", ")}`);
|
|
1394
2063
|
if (summary.refreshed.length > 0) lines.push(`Refreshed servers: ${summary.refreshed.join(", ")}`);
|
|
1395
2064
|
if (summary.removed.length > 0) lines.push(`Removed servers (and their tools): ${summary.removed.join(", ")}`);
|
|
2065
|
+
if (summary.droppedTools > 0) {
|
|
2066
|
+
lines.push("");
|
|
2067
|
+
lines.push(...truncationLines(summary));
|
|
2068
|
+
}
|
|
1396
2069
|
const pending = pendingAuthEntries(summary);
|
|
1397
2070
|
if (pending.length > 0) {
|
|
1398
2071
|
lines.push("");
|
|
@@ -1408,9 +2081,10 @@ function formatMcpUpdateMessage(summary) {
|
|
|
1408
2081
|
if (timedOut.length > 0) {
|
|
1409
2082
|
lines.push("");
|
|
1410
2083
|
lines.push("Servers that timed out during initialize (bridge is still alive in the background; usually mcp-remote-style stdio bridges mid-OAuth):");
|
|
1411
|
-
for (const { id, stderr } of timedOut) {
|
|
2084
|
+
for (const { id, stderr, diagnostics } of timedOut) {
|
|
1412
2085
|
lines.push(` - ${id}:`);
|
|
1413
|
-
|
|
2086
|
+
lines.push(...timeoutDiagnosticLines(diagnostics));
|
|
2087
|
+
appendTimeoutStderrBlock(lines, stderr);
|
|
1414
2088
|
}
|
|
1415
2089
|
lines.push("If the bridge prints an auth URL in its stderr, share it with the user. Then call `reload_mcp` once they've finished.");
|
|
1416
2090
|
}
|
|
@@ -1499,156 +2173,16 @@ async function runToolUpdateLoop({ session, log }) {
|
|
|
1499
2173
|
}, "reached tool-update continuation cap; further updates will land on next request");
|
|
1500
2174
|
}
|
|
1501
2175
|
//#endregion
|
|
1502
|
-
//#region src/tracing.ts
|
|
1503
|
-
const SERVICE_NAME = "skydive-agent-harness";
|
|
1504
|
-
const DAEMON_TRACES_URL = "http://localhost:38994/v1/traces";
|
|
1505
|
-
let provider = null;
|
|
1506
|
-
function initTracing() {
|
|
1507
|
-
if (provider) return;
|
|
1508
|
-
propagation.setGlobalPropagator(new W3CTraceContextPropagator());
|
|
1509
|
-
provider = new NodeTracerProvider({ resource: new Resource({ [ATTR_SERVICE_NAME]: SERVICE_NAME }) });
|
|
1510
|
-
const exporter = new OTLPTraceExporter({ url: DAEMON_TRACES_URL });
|
|
1511
|
-
provider.addSpanProcessor(new BatchSpanProcessor(exporter, { scheduledDelayMillis: 1e3 }));
|
|
1512
|
-
provider.register();
|
|
1513
|
-
logger.info({
|
|
1514
|
-
event: "tracing_enabled",
|
|
1515
|
-
endpoint: DAEMON_TRACES_URL
|
|
1516
|
-
}, "OTel tracing initialized — exporting via daemon");
|
|
1517
|
-
const shutdown = async () => {
|
|
1518
|
-
await shutdownTracing();
|
|
1519
|
-
process.exit(0);
|
|
1520
|
-
};
|
|
1521
|
-
process.on("SIGTERM", shutdown);
|
|
1522
|
-
process.on("SIGINT", shutdown);
|
|
1523
|
-
}
|
|
1524
|
-
function getTracer() {
|
|
1525
|
-
return trace.getTracer(SERVICE_NAME);
|
|
1526
|
-
}
|
|
1527
|
-
function extractRemoteContext() {
|
|
1528
|
-
const traceparent = getCurrentTraceparent();
|
|
1529
|
-
if (!traceparent) return ROOT_CONTEXT;
|
|
1530
|
-
const carrier = { traceparent };
|
|
1531
|
-
return propagation.extract(ROOT_CONTEXT, carrier, {
|
|
1532
|
-
get: (c, key) => c[key],
|
|
1533
|
-
keys: (c) => Object.keys(c)
|
|
1534
|
-
});
|
|
1535
|
-
}
|
|
1536
|
-
async function shutdownTracing() {
|
|
1537
|
-
if (provider) await provider.shutdown();
|
|
1538
|
-
}
|
|
1539
|
-
//#endregion
|
|
1540
|
-
//#region src/harness.ts
|
|
1541
|
-
/**
|
|
1542
|
-
* Skydive composition over @skydiveai/pi-server: wires the platform
|
|
1543
|
-
* defaults (tracing to the daemon, tool-update hot-reload hooks, prewarm
|
|
1544
|
-
* paths, header passthrough for proxy routing, Skydive agent-card
|
|
1545
|
-
* branding) into the generic protocol server, and returns mountable
|
|
1546
|
-
* handlers. The agent supplies the pi session factory (their session.ts)
|
|
1547
|
-
* and owns the express app:
|
|
1548
|
-
*
|
|
1549
|
-
* const { platform, protocols } = createHarness({
|
|
1550
|
-
* cwd: process.cwd(),
|
|
1551
|
-
* createSession,
|
|
1552
|
-
* });
|
|
1553
|
-
* app.get('/health', platform.handlers.health);
|
|
1554
|
-
* app.use(platform.handlers.injectEnv);
|
|
1555
|
-
* app.use(platform.handlers.prewarm); // matches /_skydive/prewarm + legacy alias
|
|
1556
|
-
* app.use(protocols.handlers.all);
|
|
1557
|
-
*/
|
|
1558
|
-
const A2A_PATH = "/a2a";
|
|
1559
|
-
const AGENT_CARD_PATH = "/.well-known/agent-card.json";
|
|
1560
|
-
const PREWARM_PATHS = ["/_skydive/prewarm", "/_anyone/prewarm"];
|
|
1561
|
-
const PASSTHROUGH_HEADER_PREFIXES = ["x-anyone-", "x-skydive-"];
|
|
1562
|
-
function createHarness(options) {
|
|
1563
|
-
initTracing();
|
|
1564
|
-
readPlatformVersions().then((runtimeVersions) => logger.info({
|
|
1565
|
-
event: "runtime_versions",
|
|
1566
|
-
runtimeVersions
|
|
1567
|
-
}, "platform runtime versions"));
|
|
1568
|
-
const serverOptions = {
|
|
1569
|
-
...options,
|
|
1570
|
-
onSessionSetup: options.onSessionSetup ?? ((args) => {
|
|
1571
|
-
installToolUpdateAutoStop(args);
|
|
1572
|
-
installIterationCap(args);
|
|
1573
|
-
}),
|
|
1574
|
-
postPrompt: options.postPrompt ?? runToolUpdateLoop,
|
|
1575
|
-
passthroughHeaderPrefixes: options.passthroughHeaderPrefixes ?? PASSTHROUGH_HEADER_PREFIXES,
|
|
1576
|
-
prewarmPaths: options.prewarmPaths ?? PREWARM_PATHS
|
|
1577
|
-
};
|
|
1578
|
-
const webHandlers = createProtocolHandlers(serverOptions);
|
|
1579
|
-
const cardOverrides = {
|
|
1580
|
-
name: "Skydive Agent",
|
|
1581
|
-
description: "An AI coding agent powered by the Skydive platform.",
|
|
1582
|
-
...options.agentCard
|
|
1583
|
-
};
|
|
1584
|
-
const agentCardWeb = async (request) => {
|
|
1585
|
-
const url = new URL(request.url);
|
|
1586
|
-
if (request.method !== "GET" || url.pathname !== AGENT_CARD_PATH) return null;
|
|
1587
|
-
return Response.json(buildAgentCard(cardOverrides.url ?? `${url.origin}${A2A_PATH}`, cardOverrides));
|
|
1588
|
-
};
|
|
1589
|
-
const a2a = restHandler({
|
|
1590
|
-
requestHandler: new DefaultRequestHandler(buildAgentCard(cardOverrides.url ?? A2A_PATH, cardOverrides), new InMemoryTaskStore(), createAgentExecutor(serverOptions), new DefaultExecutionEventBusManager()),
|
|
1591
|
-
userBuilder: UserBuilder.noAuthentication
|
|
1592
|
-
});
|
|
1593
|
-
return {
|
|
1594
|
-
platform: { handlers: {
|
|
1595
|
-
/**
|
|
1596
|
-
* Platform health plus the agent's `healthMetadata`. Mount above
|
|
1597
|
-
* injectEnv — health must respond immediately for prewarm
|
|
1598
|
-
* stashing and readiness probes, and injectEnv can wait up to
|
|
1599
|
-
* 10s for env vars during boot.
|
|
1600
|
-
*/
|
|
1601
|
-
health: createHealthHandler({ metadata: options.healthMetadata ?? null }),
|
|
1602
|
-
/** Loads platform env (e2b envd / daemon long-poll). */
|
|
1603
|
-
injectEnv: createPlatformEnvMiddleware(),
|
|
1604
|
-
prewarm: webHandlerToMiddleware(webHandlers.prewarm)
|
|
1605
|
-
} },
|
|
1606
|
-
protocols: { handlers: {
|
|
1607
|
-
/** Express-style; mount at /a2a. */
|
|
1608
|
-
a2a,
|
|
1609
|
-
/** GET /.well-known/agent-card.json. */
|
|
1610
|
-
agentCard: webHandlerToMiddleware(agentCardWeb),
|
|
1611
|
-
/** Mirrors each vendor's API shape. */
|
|
1612
|
-
openai: { v1: {
|
|
1613
|
-
chat: { completions: webHandlerToMiddleware(webHandlers.chatCompletions) },
|
|
1614
|
-
responses: webHandlerToMiddleware(webHandlers.responses)
|
|
1615
|
-
} },
|
|
1616
|
-
anthropic: { v1: { messages: webHandlerToMiddleware(webHandlers.messages) } },
|
|
1617
|
-
/**
|
|
1618
|
-
* Everything in one mount: a2a (+ agent card) at their well-known
|
|
1619
|
-
* paths, then chat-completions / anthropic-messages / responses.
|
|
1620
|
-
* Calls next() when nothing matches.
|
|
1621
|
-
*/
|
|
1622
|
-
all: chainMiddleware([mountAt(A2A_PATH, a2a), webHandlerToMiddleware(composeHandlers([
|
|
1623
|
-
agentCardWeb,
|
|
1624
|
-
webHandlers.chatCompletions,
|
|
1625
|
-
webHandlers.messages,
|
|
1626
|
-
webHandlers.responses
|
|
1627
|
-
]))])
|
|
1628
|
-
} }
|
|
1629
|
-
};
|
|
1630
|
-
}
|
|
1631
|
-
/** Effective bash timeout: the model's value when it gave a positive number, else the default. */
|
|
1632
|
-
function resolveBashTimeout(provided) {
|
|
1633
|
-
return typeof provided === "number" && provided > 0 ? provided : 600;
|
|
1634
|
-
}
|
|
1635
|
-
const bashDefaultTimeoutExtension = (pi) => {
|
|
1636
|
-
pi.on("tool_call", async (event) => {
|
|
1637
|
-
if (event.toolName !== "bash") return;
|
|
1638
|
-
event.input.timeout = resolveBashTimeout(event.input.timeout);
|
|
1639
|
-
});
|
|
1640
|
-
};
|
|
1641
|
-
//#endregion
|
|
1642
2176
|
//#region src/channel-context-ref.ts
|
|
1643
2177
|
/**
|
|
1644
|
-
* The worker injects
|
|
1645
|
-
* sandbox env (`SKYDIVE_CHANNEL_CONTEXT`) rather than the full context.
|
|
2178
|
+
* The worker injects a small reference — `{ channel, messageId, runId }` —
|
|
2179
|
+
* into the sandbox env (`SKYDIVE_CHANNEL_CONTEXT`) rather than the full context.
|
|
1646
2180
|
*
|
|
1647
2181
|
* The canonical `ChannelContextRef` type + `parseChannelContextRef` live in
|
|
1648
2182
|
* `@createinc/anyone-channels`, but the harness (`@skydiveai/*`) keeps zero
|
|
1649
2183
|
* `@createinc/*` dependencies — importing that package would pull the whole
|
|
1650
|
-
* platform channel stack (Slack/email/Linq SDKs, messaging)
|
|
1651
|
-
*
|
|
2184
|
+
* platform channel stack (Slack/email/Linq SDKs, messaging) just to read one
|
|
2185
|
+
* field. So we validate the field this consumer needs locally instead.
|
|
1652
2186
|
*/
|
|
1653
2187
|
const channelContextRefSchema = z.object({ messageId: z.string().nullable() });
|
|
1654
2188
|
/**
|
|
@@ -1680,20 +2214,84 @@ function apiBaseUrl() {
|
|
|
1680
2214
|
}
|
|
1681
2215
|
//#endregion
|
|
1682
2216
|
//#region src/extensions/platform.ts
|
|
2217
|
+
/**
|
|
2218
|
+
* Platform extension — bridges the agent harness to the Skydive platform daemon.
|
|
2219
|
+
*
|
|
2220
|
+
* Responsibilities:
|
|
2221
|
+
* - Heartbeat: periodic POST to the API so the sandbox manager knows the
|
|
2222
|
+
* agent is alive. Throttled to once per minute, triggered by tool events.
|
|
2223
|
+
* - Session tracking: registers the session with the daemon on start,
|
|
2224
|
+
* streams tool_call / tool_result events so the daemon can track which
|
|
2225
|
+
* session is actively executing, and signals session end on agent_end.
|
|
2226
|
+
* - Channel context: passes the SKYDIVE_CHANNEL_CONTEXT (containing the
|
|
2227
|
+
* messageId) to the daemon so file writes can be attributed to the
|
|
2228
|
+
* correct conversation.
|
|
2229
|
+
*
|
|
2230
|
+
* All daemon POSTs are fire-and-forget — failures are logged but never
|
|
2231
|
+
* block the agent. The daemon may not be running (e.g. local dev without
|
|
2232
|
+
* a sandbox), and that's fine.
|
|
2233
|
+
*/
|
|
1683
2234
|
const HEARTBEAT_THROTTLE_MS = 6e4;
|
|
1684
2235
|
const TOOL_HEARTBEAT_INTERVAL_MS = 5e3;
|
|
1685
2236
|
const MAX_TOOL_HEARTBEATS = 1440 * 60 * 1e3 / TOOL_HEARTBEAT_INTERVAL_MS;
|
|
1686
2237
|
const DAEMON_URL = "http://localhost:38994";
|
|
1687
|
-
const log$
|
|
2238
|
+
const log$11 = logger.child({ module: "platform-ext" });
|
|
1688
2239
|
function sandboxClient() {
|
|
1689
2240
|
const apiUrl = apiBaseUrl();
|
|
1690
2241
|
if (!apiUrl) return null;
|
|
1691
2242
|
return hc(`${apiUrl}/api/v1/sandbox`);
|
|
1692
2243
|
}
|
|
1693
2244
|
/**
|
|
1694
|
-
*
|
|
1695
|
-
*
|
|
1696
|
-
*
|
|
2245
|
+
* Is this box still an unclaimed warm-pool sandbox? (ANY-6000, the
|
|
2246
|
+
* feature-flags half of the ANY-5184 pool 403 wave.)
|
|
2247
|
+
*
|
|
2248
|
+
* `GET /sandbox/feature-flags` is agent-only, so the shared poller's request
|
|
2249
|
+
* from a pool box can only 403 — a guaranteed-failing GET every 60s for the
|
|
2250
|
+
* life of the pool phase. The discriminator is the sandbox token's `type`
|
|
2251
|
+
* claim, read UNVERIFIED (this box never holds the signing secret): not an
|
|
2252
|
+
* authorization decision, only "should I bother calling?", and the api still
|
|
2253
|
+
* authorizes every request.
|
|
2254
|
+
*
|
|
2255
|
+
* Read per call from the daemon's persisted env file, NOT process.env:
|
|
2256
|
+
* claiming a pool box rebinds the token in place (the daemon rewrites this
|
|
2257
|
+
* file) while the harness's process.env keeps the boot snapshot, so a
|
|
2258
|
+
* process-env gate would leave a claimed box permanently skipping — trading a
|
|
2259
|
+
* wasted request for silently frozen flags, which is strictly worse. "Cannot
|
|
2260
|
+
* tell" (no file, no token, unparseable payload) reports false so the poll
|
|
2261
|
+
* proceeds.
|
|
2262
|
+
*/
|
|
2263
|
+
const daemonEnvIdentitySchema = z.object({
|
|
2264
|
+
ANYONE_SANDBOX_TOKEN: z.string().optional(),
|
|
2265
|
+
SKYDIVE_SANDBOX_TOKEN: z.string().optional()
|
|
2266
|
+
}).passthrough();
|
|
2267
|
+
const tokenTypeSchema = z.object({ type: z.string() }).passthrough();
|
|
2268
|
+
async function isPoolIdentity() {
|
|
2269
|
+
try {
|
|
2270
|
+
const override = process.env.ANYONE_DAEMON_ENV_CACHE;
|
|
2271
|
+
const candidates = override ? [override] : ["/run/anyone-system/daemon-env.json", "/tmp/.anyone/daemon-env.json"];
|
|
2272
|
+
let raw = null;
|
|
2273
|
+
for (const file of candidates) {
|
|
2274
|
+
raw = await readFile(file, "utf8").catch(() => null);
|
|
2275
|
+
if (raw !== null) break;
|
|
2276
|
+
}
|
|
2277
|
+
if (raw === null) return false;
|
|
2278
|
+
const env = daemonEnvIdentitySchema.safeParse(JSON.parse(raw));
|
|
2279
|
+
if (!env.success) return false;
|
|
2280
|
+
const token = env.data.ANYONE_SANDBOX_TOKEN ?? env.data.SKYDIVE_SANDBOX_TOKEN;
|
|
2281
|
+
if (typeof token !== "string" || token === "") return false;
|
|
2282
|
+
const payload = token.split(".")[1];
|
|
2283
|
+
if (!payload) return false;
|
|
2284
|
+
const claims = tokenTypeSchema.safeParse(JSON.parse(Buffer.from(payload, "base64url").toString("utf8")));
|
|
2285
|
+
return claims.success && claims.data.type === "onboarding-pool";
|
|
2286
|
+
} catch (_err) {
|
|
2287
|
+
return false;
|
|
2288
|
+
}
|
|
2289
|
+
}
|
|
2290
|
+
/**
|
|
2291
|
+
* Fetch every harness feature flag in one GET (`{ contextManagement, ... }`
|
|
2292
|
+
* — see apps/anyone/api/src/routes/sandbox-feature-flags.ts). Returns
|
|
2293
|
+
* null when indeterminate (no api url, the request failed, or the box is an
|
|
2294
|
+
* unclaimed pool sandbox whose token the route would 403) so the shared
|
|
1697
2295
|
* poller keeps the last-known values rather than flipping on a transient error.
|
|
1698
2296
|
* This is the single fetch behind `feature-flags-poll.ts`; extensions read the
|
|
1699
2297
|
* polled values there instead of issuing their own GET.
|
|
@@ -1701,10 +2299,11 @@ function sandboxClient() {
|
|
|
1701
2299
|
async function fetchHarnessFlags() {
|
|
1702
2300
|
const client = sandboxClient();
|
|
1703
2301
|
if (!client) return null;
|
|
2302
|
+
if (await isPoolIdentity()) return null;
|
|
1704
2303
|
try {
|
|
1705
2304
|
const res = await client["feature-flags"].$get();
|
|
1706
2305
|
if (!res.ok) {
|
|
1707
|
-
log$
|
|
2306
|
+
log$11.debug({
|
|
1708
2307
|
status: res.status,
|
|
1709
2308
|
event: "feature_flags_fetch_failed"
|
|
1710
2309
|
}, "feature-flags fetch failed");
|
|
@@ -1713,10 +2312,10 @@ async function fetchHarnessFlags() {
|
|
|
1713
2312
|
const body = await res.json();
|
|
1714
2313
|
return {
|
|
1715
2314
|
contextManagement: body.contextManagement ?? null,
|
|
1716
|
-
|
|
2315
|
+
closingText: body.closingText ?? null
|
|
1717
2316
|
};
|
|
1718
2317
|
} catch (err) {
|
|
1719
|
-
log$
|
|
2318
|
+
log$11.debug({
|
|
1720
2319
|
err,
|
|
1721
2320
|
event: "feature_flags_fetch_error"
|
|
1722
2321
|
}, "feature-flags request errored");
|
|
@@ -1727,7 +2326,7 @@ function postHeartbeat({ messageId }) {
|
|
|
1727
2326
|
const client = sandboxClient();
|
|
1728
2327
|
if (!client) return;
|
|
1729
2328
|
client.heartbeat.$post({ json: { messageId } }).catch((err) => {
|
|
1730
|
-
log$
|
|
2329
|
+
log$11.debug({
|
|
1731
2330
|
err,
|
|
1732
2331
|
event: "heartbeat_failed"
|
|
1733
2332
|
}, "heartbeat failed");
|
|
@@ -1739,7 +2338,7 @@ async function resolveConversationFromApi(messageId) {
|
|
|
1739
2338
|
try {
|
|
1740
2339
|
const res = await client["message-conversation"].$get({ query: { messageId } });
|
|
1741
2340
|
if (!res.ok) {
|
|
1742
|
-
log$
|
|
2341
|
+
log$11.warn({
|
|
1743
2342
|
status: res.status,
|
|
1744
2343
|
messageId,
|
|
1745
2344
|
event: "resolve_conversation_failed"
|
|
@@ -1748,7 +2347,7 @@ async function resolveConversationFromApi(messageId) {
|
|
|
1748
2347
|
}
|
|
1749
2348
|
return (await res.json()).conversationId ?? null;
|
|
1750
2349
|
} catch (err) {
|
|
1751
|
-
log$
|
|
2350
|
+
log$11.warn({
|
|
1752
2351
|
err,
|
|
1753
2352
|
messageId,
|
|
1754
2353
|
event: "resolve_conversation_error"
|
|
@@ -1765,25 +2364,101 @@ async function postBackgroundTaskDone({ messageId, content }) {
|
|
|
1765
2364
|
} });
|
|
1766
2365
|
if (!res.ok) throw new Error(`bg-task-done POST failed: ${res.status}`);
|
|
1767
2366
|
}
|
|
1768
|
-
async function
|
|
2367
|
+
async function putBackgroundTaskJournalSpec({ messageId, spec }) {
|
|
1769
2368
|
const client = sandboxClient();
|
|
1770
|
-
if (!client)
|
|
1771
|
-
|
|
1772
|
-
|
|
1773
|
-
|
|
1774
|
-
|
|
1775
|
-
|
|
1776
|
-
|
|
2369
|
+
if (!client) return;
|
|
2370
|
+
try {
|
|
2371
|
+
const res = await client["bg-task-journal"].$put({ json: {
|
|
2372
|
+
messageId,
|
|
2373
|
+
spec
|
|
2374
|
+
} });
|
|
2375
|
+
if (!res.ok) log$11.warn({
|
|
2376
|
+
status: res.status,
|
|
2377
|
+
taskId: spec.id,
|
|
2378
|
+
event: "bg_journal_put_failed"
|
|
2379
|
+
}, "bg-task journal PUT failed");
|
|
2380
|
+
} catch (err) {
|
|
2381
|
+
log$11.warn({
|
|
2382
|
+
err,
|
|
2383
|
+
taskId: spec.id,
|
|
2384
|
+
event: "bg_journal_put_failed"
|
|
2385
|
+
}, "bg-task journal PUT threw");
|
|
2386
|
+
}
|
|
1777
2387
|
}
|
|
1778
|
-
function
|
|
1779
|
-
|
|
1780
|
-
|
|
1781
|
-
|
|
1782
|
-
|
|
1783
|
-
|
|
1784
|
-
|
|
1785
|
-
|
|
1786
|
-
|
|
2388
|
+
async function deleteBackgroundTaskJournalSpec({ messageId, taskId }) {
|
|
2389
|
+
const client = sandboxClient();
|
|
2390
|
+
if (!client) return;
|
|
2391
|
+
try {
|
|
2392
|
+
await client["bg-task-journal"].$delete({ json: {
|
|
2393
|
+
messageId,
|
|
2394
|
+
taskId
|
|
2395
|
+
} });
|
|
2396
|
+
} catch (err) {
|
|
2397
|
+
log$11.debug({
|
|
2398
|
+
err,
|
|
2399
|
+
taskId,
|
|
2400
|
+
event: "bg_journal_delete_failed"
|
|
2401
|
+
}, "bg-task journal DELETE failed");
|
|
2402
|
+
}
|
|
2403
|
+
}
|
|
2404
|
+
async function listBackgroundTaskJournalSpecs({ messageId }) {
|
|
2405
|
+
const client = sandboxClient();
|
|
2406
|
+
if (!client) return [];
|
|
2407
|
+
try {
|
|
2408
|
+
const res = await client["bg-task-journal"].$get({ query: { messageId } });
|
|
2409
|
+
if (!res.ok) return [];
|
|
2410
|
+
return (await res.json()).specs ?? [];
|
|
2411
|
+
} catch (err) {
|
|
2412
|
+
log$11.debug({
|
|
2413
|
+
err,
|
|
2414
|
+
event: "bg_journal_list_failed"
|
|
2415
|
+
}, "bg-task journal GET failed");
|
|
2416
|
+
return [];
|
|
2417
|
+
}
|
|
2418
|
+
}
|
|
2419
|
+
function postBackgroundTasksSnapshot({ messageId, tasks }) {
|
|
2420
|
+
const client = sandboxClient();
|
|
2421
|
+
if (!client || !messageId) return;
|
|
2422
|
+
client["bg-tasks"].$post({ json: {
|
|
2423
|
+
messageId,
|
|
2424
|
+
tasks
|
|
2425
|
+
} }).catch((err) => {
|
|
2426
|
+
log$11.debug({
|
|
2427
|
+
err,
|
|
2428
|
+
event: "bg_tasks_snapshot_failed"
|
|
2429
|
+
}, "bg-tasks snapshot publish failed");
|
|
2430
|
+
});
|
|
2431
|
+
}
|
|
2432
|
+
async function postSubagentSpawn({ messageId, tasks }) {
|
|
2433
|
+
const client = sandboxClient();
|
|
2434
|
+
if (!client) throw new Error("no api url for subagent-spawn");
|
|
2435
|
+
const res = await client["subagent-spawn"].$post({ json: {
|
|
2436
|
+
messageId,
|
|
2437
|
+
tasks
|
|
2438
|
+
} });
|
|
2439
|
+
if (!res.ok) {
|
|
2440
|
+
let detail = "";
|
|
2441
|
+
try {
|
|
2442
|
+
const errBody = await res.json();
|
|
2443
|
+
if (errBody && typeof errBody.error === "string") detail = `: ${errBody.error}`;
|
|
2444
|
+
} catch {}
|
|
2445
|
+
throw new Error(`subagent-spawn POST failed (${res.status})${detail}`);
|
|
2446
|
+
}
|
|
2447
|
+
const body = await res.json();
|
|
2448
|
+
return {
|
|
2449
|
+
taskIds: body.taskIds,
|
|
2450
|
+
tasks: body.tasks ?? []
|
|
2451
|
+
};
|
|
2452
|
+
}
|
|
2453
|
+
function createHeartbeatThrottle({ messageId }) {
|
|
2454
|
+
let lastAt = 0;
|
|
2455
|
+
let pending = null;
|
|
2456
|
+
function send() {
|
|
2457
|
+
const now = Date.now();
|
|
2458
|
+
if (now - lastAt < HEARTBEAT_THROTTLE_MS) {
|
|
2459
|
+
if (!pending) pending = setTimeout(() => {
|
|
2460
|
+
pending = null;
|
|
2461
|
+
send();
|
|
1787
2462
|
}, HEARTBEAT_THROTTLE_MS - (now - lastAt));
|
|
1788
2463
|
return;
|
|
1789
2464
|
}
|
|
@@ -1822,7 +2497,7 @@ function createToolHeartbeat({ messageId }) {
|
|
|
1822
2497
|
}
|
|
1823
2498
|
heartbeatCount++;
|
|
1824
2499
|
if (heartbeatCount > MAX_TOOL_HEARTBEATS) {
|
|
1825
|
-
log$
|
|
2500
|
+
log$11.warn({
|
|
1826
2501
|
heartbeatCount,
|
|
1827
2502
|
activeToolCalls: [...activeToolCalls]
|
|
1828
2503
|
}, "tool heartbeat max reached, stopping");
|
|
@@ -1853,7 +2528,7 @@ function postToDaemon(path, body) {
|
|
|
1853
2528
|
headers: { "content-type": "application/json" },
|
|
1854
2529
|
body: JSON.stringify(body)
|
|
1855
2530
|
}).catch((err) => {
|
|
1856
|
-
log$
|
|
2531
|
+
log$11.debug({
|
|
1857
2532
|
err,
|
|
1858
2533
|
path,
|
|
1859
2534
|
event: "daemon_post_failed"
|
|
@@ -1862,7 +2537,7 @@ function postToDaemon(path, body) {
|
|
|
1862
2537
|
}
|
|
1863
2538
|
function createPlatformExtensions({ sessionId, channelContext }) {
|
|
1864
2539
|
return (pi) => {
|
|
1865
|
-
log$
|
|
2540
|
+
log$11.info({
|
|
1866
2541
|
sessionId,
|
|
1867
2542
|
hasChannelContext: Boolean(channelContext)
|
|
1868
2543
|
}, "platform extension initialized");
|
|
@@ -1911,7 +2586,7 @@ function createPlatformExtensions({ sessionId, channelContext }) {
|
|
|
1911
2586
|
});
|
|
1912
2587
|
});
|
|
1913
2588
|
pi.on("agent_end", () => {
|
|
1914
|
-
log$
|
|
2589
|
+
log$11.info({ sessionId }, "session ending");
|
|
1915
2590
|
postToDaemon("/session/end", { sessionId });
|
|
1916
2591
|
});
|
|
1917
2592
|
};
|
|
@@ -1922,37 +2597,34 @@ function createPlatformExtensions({ sessionId, channelContext }) {
|
|
|
1922
2597
|
* Shared harness feature-flag poll.
|
|
1923
2598
|
*
|
|
1924
2599
|
* The api exposes one `/feature-flags` GET that returns every harness flag in a
|
|
1925
|
-
* single response (`{ contextManagement,
|
|
2600
|
+
* single response (`{ contextManagement, closingText, commandFlags }` — see
|
|
1926
2601
|
* apps/anyone/api/src/routes/sandbox-feature-flags.ts). Rather than each
|
|
1927
2602
|
* extension issuing its own GET — and, worse, a *blocking* GET on the
|
|
1928
2603
|
* pre-first-token `session_start` path — a single background poller fetches
|
|
1929
2604
|
* that response once per interval and fans the values out to every subscriber.
|
|
1930
2605
|
*
|
|
1931
|
-
* Why one poller:
|
|
1932
|
-
*
|
|
1933
|
-
*
|
|
1934
|
-
*
|
|
1935
|
-
*
|
|
1936
|
-
* the hot path allocation-only. A cold cache reads as `null` (fail-open to
|
|
1937
|
-
* unregistered); a newly-flipped flag takes effect on the next poll, matching
|
|
1938
|
-
* how context-management already treats its flag.
|
|
2606
|
+
* Why one poller: context-management consumes the `contextManagement` flag and
|
|
2607
|
+
* the closing-text loop consumes `closingText`, both without a blocking GET on
|
|
2608
|
+
* the pre-first-token `session_start` path. Reading the last-polled value keeps
|
|
2609
|
+
* the hot path allocation-only; a cold cache reads as `null` and a
|
|
2610
|
+
* newly-flipped flag takes effect on the next poll.
|
|
1939
2611
|
*
|
|
1940
2612
|
* The poll is fire-and-forget and self-unref'd — it never keeps the process
|
|
1941
2613
|
* alive and an indeterminate result (no api url / transient failure) leaves the
|
|
1942
2614
|
* last-known values untouched so a blip can't silently flip behavior.
|
|
1943
2615
|
*/
|
|
1944
|
-
const log$
|
|
2616
|
+
const log$10 = logger.child({ module: "feature-flags-poll" });
|
|
1945
2617
|
const FLAG_POLL_INTERVAL_MS = 6e4;
|
|
1946
2618
|
let contextManagement = null;
|
|
1947
|
-
let
|
|
2619
|
+
let closingText = null;
|
|
1948
2620
|
const subscribers = {
|
|
1949
2621
|
contextManagement: /* @__PURE__ */ new Set(),
|
|
1950
|
-
|
|
2622
|
+
closingText: /* @__PURE__ */ new Set()
|
|
1951
2623
|
};
|
|
1952
2624
|
let pollerStarted = false;
|
|
1953
2625
|
let firstPollSettled = false;
|
|
1954
2626
|
let resolveFirstPoll = null;
|
|
1955
|
-
|
|
2627
|
+
new Promise((resolve) => {
|
|
1956
2628
|
resolveFirstPoll = resolve;
|
|
1957
2629
|
});
|
|
1958
2630
|
function markFirstPollSettled() {
|
|
@@ -1962,19 +2634,7 @@ function markFirstPollSettled() {
|
|
|
1962
2634
|
}
|
|
1963
2635
|
/** Last-polled value of a flag, or `null` if not yet resolved. */
|
|
1964
2636
|
function getPolledFlag(name) {
|
|
1965
|
-
return name === "
|
|
1966
|
-
}
|
|
1967
|
-
/**
|
|
1968
|
-
* Await the first poll already kicked by `startFeatureFlagPoller` (never a new
|
|
1969
|
-
* GET). Resolves when that poll settles, immediately if it already has, or
|
|
1970
|
-
* immediately when there's no flag source to poll. Callers on the hot path
|
|
1971
|
-
* should race this against their own short timeout so a slow/failed flag
|
|
1972
|
-
* service cannot delay first-token; a timeout just means the caller reads the
|
|
1973
|
-
* still-cold cache and falls back to its default, exactly as before.
|
|
1974
|
-
*/
|
|
1975
|
-
function awaitFirstFlagPoll() {
|
|
1976
|
-
if (firstPollSettled || !hasFlagSource()) return Promise.resolve();
|
|
1977
|
-
return firstPollPromise;
|
|
2637
|
+
return name === "closingText" ? closingText : contextManagement;
|
|
1978
2638
|
}
|
|
1979
2639
|
/**
|
|
1980
2640
|
* Subscribe to changes of a flag. The callback fires only on a *transition*
|
|
@@ -1987,13 +2647,13 @@ function onFlagChange(name, cb) {
|
|
|
1987
2647
|
}
|
|
1988
2648
|
function apply(name, next) {
|
|
1989
2649
|
if (next === null) return;
|
|
1990
|
-
const prev = name === "
|
|
1991
|
-
if (name === "
|
|
1992
|
-
else
|
|
2650
|
+
const prev = name === "closingText" ? closingText : contextManagement;
|
|
2651
|
+
if (name === "closingText") closingText = next;
|
|
2652
|
+
else contextManagement = next;
|
|
1993
2653
|
if (next !== prev) for (const cb of subscribers[name]) try {
|
|
1994
2654
|
cb(next);
|
|
1995
2655
|
} catch (err) {
|
|
1996
|
-
log$
|
|
2656
|
+
log$10.warn({
|
|
1997
2657
|
err,
|
|
1998
2658
|
flag: name
|
|
1999
2659
|
}, "flag subscriber threw");
|
|
@@ -2004,9 +2664,9 @@ async function pollOnce() {
|
|
|
2004
2664
|
const flags = await fetchHarnessFlags();
|
|
2005
2665
|
if (!flags) return;
|
|
2006
2666
|
apply("contextManagement", flags.contextManagement ?? null);
|
|
2007
|
-
apply("
|
|
2667
|
+
apply("closingText", flags.closingText ?? null);
|
|
2008
2668
|
} catch (err) {
|
|
2009
|
-
log$
|
|
2669
|
+
log$10.debug({ err }, "feature-flag poll threw");
|
|
2010
2670
|
}
|
|
2011
2671
|
}
|
|
2012
2672
|
/**
|
|
@@ -2022,6 +2682,569 @@ function startFeatureFlagPoller() {
|
|
|
2022
2682
|
setInterval(() => void pollOnce(), FLAG_POLL_INTERVAL_MS).unref?.();
|
|
2023
2683
|
}
|
|
2024
2684
|
//#endregion
|
|
2685
|
+
//#region src/closing-text-loop.ts
|
|
2686
|
+
const CLOSING_TEXT_NUDGE = "<system_notification>You ended that turn on a tool call without writing any reply to the user. Write a brief closing message now: what you did and the outcome. Do not call any more tools unless you genuinely still need to.</system_notification>";
|
|
2687
|
+
function isNonEmptyText(block) {
|
|
2688
|
+
return block.type === "text" && typeof block.text === "string" && block.text.trim().length > 0;
|
|
2689
|
+
}
|
|
2690
|
+
/**
|
|
2691
|
+
* True when the assistant produced no visible text since the human's last
|
|
2692
|
+
* message this turn — i.e. the turn ended on tool calls / thinking only.
|
|
2693
|
+
*
|
|
2694
|
+
* We scan from the end back to the most recent `user` message (the human turn
|
|
2695
|
+
* that triggered this run) and inspect every `assistant` message after it.
|
|
2696
|
+
* A string-content assistant message (rare, but pi allows it) counts as text
|
|
2697
|
+
* when non-blank. `toolResult` and custom messages are ignored.
|
|
2698
|
+
*/
|
|
2699
|
+
function turnEndedWithoutVisibleText(messages) {
|
|
2700
|
+
let sawAssistant = false;
|
|
2701
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
2702
|
+
const m = messages[i];
|
|
2703
|
+
if (!m || m.role === "user") break;
|
|
2704
|
+
if (m.role !== "assistant") continue;
|
|
2705
|
+
sawAssistant = true;
|
|
2706
|
+
if (typeof m.content === "string") {
|
|
2707
|
+
if (m.content.trim().length > 0) return false;
|
|
2708
|
+
continue;
|
|
2709
|
+
}
|
|
2710
|
+
if (Array.isArray(m.content) && m.content.some(isNonEmptyText)) return false;
|
|
2711
|
+
}
|
|
2712
|
+
return sawAssistant;
|
|
2713
|
+
}
|
|
2714
|
+
/**
|
|
2715
|
+
* True when the last assistant message this turn ended in a way we must not
|
|
2716
|
+
* nudge: a user abort or a provider error. Both already surface their own
|
|
2717
|
+
* terminal signal to the client; a closing-text prompt on top would be noise.
|
|
2718
|
+
*/
|
|
2719
|
+
function turnEndedAbnormally(messages) {
|
|
2720
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
2721
|
+
const m = messages[i];
|
|
2722
|
+
if (!m || m.role !== "assistant") continue;
|
|
2723
|
+
return m.stopReason === "aborted" || m.stopReason === "error";
|
|
2724
|
+
}
|
|
2725
|
+
return false;
|
|
2726
|
+
}
|
|
2727
|
+
async function runClosingTextLoop({ session, log, enabled = getPolledFlag("closingText") === true }) {
|
|
2728
|
+
if (!enabled) return;
|
|
2729
|
+
if (turnEndedAbnormally(session.messages)) return;
|
|
2730
|
+
if (!turnEndedWithoutVisibleText(session.messages)) return;
|
|
2731
|
+
log.info({ event: "closing_text_continuation_injected" }, "turn ended on a tool call with no visible reply; injecting a closing-text continuation");
|
|
2732
|
+
await session.sendCustomMessage({
|
|
2733
|
+
customType: "closing-text-nudge",
|
|
2734
|
+
content: CLOSING_TEXT_NUDGE,
|
|
2735
|
+
display: false,
|
|
2736
|
+
details: {}
|
|
2737
|
+
}, { triggerTurn: true });
|
|
2738
|
+
}
|
|
2739
|
+
//#endregion
|
|
2740
|
+
//#region src/tracing.ts
|
|
2741
|
+
const SERVICE_NAME = "skydive-agent-harness";
|
|
2742
|
+
const DAEMON_TRACES_URL = "http://localhost:38994/v1/traces";
|
|
2743
|
+
let provider = null;
|
|
2744
|
+
function initTracing() {
|
|
2745
|
+
if (provider) return;
|
|
2746
|
+
propagation.setGlobalPropagator(new W3CTraceContextPropagator());
|
|
2747
|
+
provider = new NodeTracerProvider({ resource: new Resource({ [ATTR_SERVICE_NAME]: SERVICE_NAME }) });
|
|
2748
|
+
const exporter = new OTLPTraceExporter({ url: DAEMON_TRACES_URL });
|
|
2749
|
+
provider.addSpanProcessor(new BatchSpanProcessor(exporter, { scheduledDelayMillis: 1e3 }));
|
|
2750
|
+
provider.register();
|
|
2751
|
+
logger.info({
|
|
2752
|
+
event: "tracing_enabled",
|
|
2753
|
+
endpoint: DAEMON_TRACES_URL
|
|
2754
|
+
}, "OTel tracing initialized — exporting via daemon");
|
|
2755
|
+
const shutdown = async () => {
|
|
2756
|
+
await shutdownTracing();
|
|
2757
|
+
process.exit(0);
|
|
2758
|
+
};
|
|
2759
|
+
process.on("SIGTERM", shutdown);
|
|
2760
|
+
process.on("SIGINT", shutdown);
|
|
2761
|
+
}
|
|
2762
|
+
function getTracer() {
|
|
2763
|
+
return trace.getTracer(SERVICE_NAME);
|
|
2764
|
+
}
|
|
2765
|
+
function extractRemoteContext() {
|
|
2766
|
+
const traceparent = getCurrentTraceparent();
|
|
2767
|
+
if (!traceparent) return ROOT_CONTEXT;
|
|
2768
|
+
const carrier = { traceparent };
|
|
2769
|
+
return propagation.extract(ROOT_CONTEXT, carrier, {
|
|
2770
|
+
get: (c, key) => c[key],
|
|
2771
|
+
keys: (c) => Object.keys(c)
|
|
2772
|
+
});
|
|
2773
|
+
}
|
|
2774
|
+
async function shutdownTracing() {
|
|
2775
|
+
if (provider) await provider.shutdown();
|
|
2776
|
+
}
|
|
2777
|
+
//#endregion
|
|
2778
|
+
//#region src/harness.ts
|
|
2779
|
+
/**
|
|
2780
|
+
* Skydive composition over @skydiveai/pi-server: wires the platform
|
|
2781
|
+
* defaults (tracing to the daemon, tool-update hot-reload hooks, prewarm
|
|
2782
|
+
* paths, header passthrough for proxy routing, Skydive agent-card
|
|
2783
|
+
* branding) into the generic protocol server, and returns mountable
|
|
2784
|
+
* handlers. The agent supplies the pi session factory (their session.ts)
|
|
2785
|
+
* and owns the express app:
|
|
2786
|
+
*
|
|
2787
|
+
* const { platform, protocols } = createHarness({
|
|
2788
|
+
* cwd: process.cwd(),
|
|
2789
|
+
* createSession,
|
|
2790
|
+
* });
|
|
2791
|
+
* app.get('/health', platform.handlers.health);
|
|
2792
|
+
* app.use(platform.handlers.injectEnv);
|
|
2793
|
+
* app.use(platform.handlers.prewarm); // matches /_skydive/prewarm + legacy alias
|
|
2794
|
+
* app.use(protocols.handlers.all);
|
|
2795
|
+
*/
|
|
2796
|
+
const A2A_PATH = "/a2a";
|
|
2797
|
+
const AGENT_CARD_PATH = "/.well-known/agent-card.json";
|
|
2798
|
+
const PREWARM_PATHS = ["/_skydive/prewarm", "/_anyone/prewarm"];
|
|
2799
|
+
const PASSTHROUGH_HEADER_PREFIXES = ["x-anyone-", "x-skydive-"];
|
|
2800
|
+
function createHarness(options) {
|
|
2801
|
+
initTracing();
|
|
2802
|
+
startFeatureFlagPoller();
|
|
2803
|
+
readPlatformVersions().then((runtimeVersions) => logger.info({
|
|
2804
|
+
event: "runtime_versions",
|
|
2805
|
+
runtimeVersions
|
|
2806
|
+
}, "platform runtime versions"));
|
|
2807
|
+
const serverOptions = {
|
|
2808
|
+
...options,
|
|
2809
|
+
onSessionSetup: options.onSessionSetup ?? ((args) => {
|
|
2810
|
+
installToolUpdateAutoStop(args);
|
|
2811
|
+
installIterationCap(args);
|
|
2812
|
+
}),
|
|
2813
|
+
postPrompt: options.postPrompt ?? (async (args) => {
|
|
2814
|
+
await runToolUpdateLoop(args);
|
|
2815
|
+
await runClosingTextLoop(args);
|
|
2816
|
+
}),
|
|
2817
|
+
passthroughHeaderPrefixes: options.passthroughHeaderPrefixes ?? PASSTHROUGH_HEADER_PREFIXES,
|
|
2818
|
+
prewarmPaths: options.prewarmPaths ?? PREWARM_PATHS
|
|
2819
|
+
};
|
|
2820
|
+
const webHandlers = createProtocolHandlers(serverOptions);
|
|
2821
|
+
const cardOverrides = {
|
|
2822
|
+
name: "Skydive Agent",
|
|
2823
|
+
description: "An AI coding agent powered by the Skydive platform.",
|
|
2824
|
+
...options.agentCard
|
|
2825
|
+
};
|
|
2826
|
+
const agentCardWeb = async (request) => {
|
|
2827
|
+
const url = new URL(request.url);
|
|
2828
|
+
if (request.method !== "GET" || url.pathname !== AGENT_CARD_PATH) return null;
|
|
2829
|
+
return Response.json(buildAgentCard(cardOverrides.url ?? `${url.origin}${A2A_PATH}`, cardOverrides));
|
|
2830
|
+
};
|
|
2831
|
+
const a2a = restHandler({
|
|
2832
|
+
requestHandler: new DefaultRequestHandler(buildAgentCard(cardOverrides.url ?? A2A_PATH, cardOverrides), new InMemoryTaskStore(), createAgentExecutor(serverOptions), new DefaultExecutionEventBusManager()),
|
|
2833
|
+
userBuilder: UserBuilder.noAuthentication
|
|
2834
|
+
});
|
|
2835
|
+
return {
|
|
2836
|
+
platform: { handlers: {
|
|
2837
|
+
/**
|
|
2838
|
+
* Platform health plus the agent's `healthMetadata`. Mount above
|
|
2839
|
+
* injectEnv — health must respond immediately for prewarm
|
|
2840
|
+
* stashing and readiness probes, and injectEnv can wait up to
|
|
2841
|
+
* 10s for env vars during boot.
|
|
2842
|
+
*/
|
|
2843
|
+
health: createHealthHandler({ metadata: options.healthMetadata ?? null }),
|
|
2844
|
+
/** Loads platform env (e2b envd / daemon long-poll). */
|
|
2845
|
+
injectEnv: createPlatformEnvMiddleware(),
|
|
2846
|
+
prewarm: webHandlerToMiddleware(webHandlers.prewarm)
|
|
2847
|
+
} },
|
|
2848
|
+
protocols: { handlers: {
|
|
2849
|
+
/** Express-style; mount at /a2a. */
|
|
2850
|
+
a2a,
|
|
2851
|
+
/** GET /.well-known/agent-card.json. */
|
|
2852
|
+
agentCard: webHandlerToMiddleware(agentCardWeb),
|
|
2853
|
+
/** Mirrors each vendor's API shape. */
|
|
2854
|
+
openai: { v1: {
|
|
2855
|
+
chat: { completions: webHandlerToMiddleware(webHandlers.chatCompletions) },
|
|
2856
|
+
responses: webHandlerToMiddleware(webHandlers.responses)
|
|
2857
|
+
} },
|
|
2858
|
+
anthropic: { v1: { messages: webHandlerToMiddleware(webHandlers.messages) } },
|
|
2859
|
+
/**
|
|
2860
|
+
* Everything in one mount: a2a (+ agent card) at their well-known
|
|
2861
|
+
* paths, then chat-completions / anthropic-messages / responses.
|
|
2862
|
+
* Calls next() when nothing matches.
|
|
2863
|
+
*/
|
|
2864
|
+
all: chainMiddleware([mountAt(A2A_PATH, a2a), webHandlerToMiddleware(composeHandlers([
|
|
2865
|
+
agentCardWeb,
|
|
2866
|
+
webHandlers.chatCompletions,
|
|
2867
|
+
webHandlers.messages,
|
|
2868
|
+
webHandlers.responses
|
|
2869
|
+
]))])
|
|
2870
|
+
} }
|
|
2871
|
+
};
|
|
2872
|
+
}
|
|
2873
|
+
/** Effective bash timeout: the model's value when it gave a positive number, else the default. */
|
|
2874
|
+
function resolveBashTimeout(provided) {
|
|
2875
|
+
return typeof provided === "number" && provided > 0 ? provided : 600;
|
|
2876
|
+
}
|
|
2877
|
+
const bashDefaultTimeoutExtension = (pi) => {
|
|
2878
|
+
pi.on("tool_call", async (event) => {
|
|
2879
|
+
if (event.toolName !== "bash") return;
|
|
2880
|
+
event.input.timeout = resolveBashTimeout(event.input.timeout);
|
|
2881
|
+
});
|
|
2882
|
+
};
|
|
2883
|
+
//#endregion
|
|
2884
|
+
//#region src/extensions/resource-pressure-warning.ts
|
|
2885
|
+
/**
|
|
2886
|
+
* Mid-run resource-pressure warning to the agent.
|
|
2887
|
+
*
|
|
2888
|
+
* The sandbox already detects pressure — the boot scripts cap the
|
|
2889
|
+
* user-workload cgroup (memory.high/memory.max) and watchers log warn/crit
|
|
2890
|
+
* edges for memory and disk — but nothing told the *agent*, so a turn burned
|
|
2891
|
+
* straight to the OOM kill (or a full disk) and only learned about it from
|
|
2892
|
+
* the post-mortem notice. This extension closes that gap in-process: while a
|
|
2893
|
+
* turn is active it polls the agent cgroup and the root filesystem and, the
|
|
2894
|
+
* first time usage crosses a warn threshold, folds a system notification into
|
|
2895
|
+
* the open turn so the agent can checkpoint, shed work (constrain
|
|
2896
|
+
* parallelism, kill a background hog, clean scratch space), or request a
|
|
2897
|
+
* bigger tier BEFORE the kill.
|
|
2898
|
+
*
|
|
2899
|
+
* The notification is triggered by the two conditions that actually kill
|
|
2900
|
+
* work — memory near the cgroup hard cap, disk near full — and reports a
|
|
2901
|
+
* snapshot of all the relevant stats (memory, CPU utilization, disk) so the
|
|
2902
|
+
* agent can tell which resource is the problem and how much headroom the
|
|
2903
|
+
* others have.
|
|
2904
|
+
*
|
|
2905
|
+
* Edge-triggered, once per trigger per turn: the fired flags reset on
|
|
2906
|
+
* agent_start, so a turn that rides a threshold gets one warning per
|
|
2907
|
+
* resource, not a stream. Polling only runs while the agent is active — an
|
|
2908
|
+
* idle sandbox's resource usage is not the agent's problem and there is no
|
|
2909
|
+
* open turn to deliver into anyway.
|
|
2910
|
+
*
|
|
2911
|
+
* Best-effort throughout: any read failure (cgroup absent, controller not
|
|
2912
|
+
* delegated, non-cgroup-v2 host, df missing) reads as "no signal" for that
|
|
2913
|
+
* stat and the extension warns on what it can see — it must never break a
|
|
2914
|
+
* turn over an observability feature.
|
|
2915
|
+
*/
|
|
2916
|
+
const execFileAsync = promisify(execFile);
|
|
2917
|
+
const log$9 = logger.child({ module: "resource-pressure-warning" });
|
|
2918
|
+
const POLL_INTERVAL_MS = 1e4;
|
|
2919
|
+
function envOverride(name) {
|
|
2920
|
+
for (const prefix of ["SKYDIVE_", "ANYONE_"]) {
|
|
2921
|
+
const value = process.env[`${prefix}${name}`];
|
|
2922
|
+
if (value != null && value !== "") return value;
|
|
2923
|
+
}
|
|
2924
|
+
return null;
|
|
2925
|
+
}
|
|
2926
|
+
function cgroupDir() {
|
|
2927
|
+
return envOverride("AGENT_CGROUP") ?? "/sys/fs/cgroup/agent";
|
|
2928
|
+
}
|
|
2929
|
+
function diskRoot() {
|
|
2930
|
+
return envOverride("DISK_ROOT") ?? "/";
|
|
2931
|
+
}
|
|
2932
|
+
/**
|
|
2933
|
+
* Read a cgroup v2 scalar file. Returns a number, or null for "max"
|
|
2934
|
+
* (uncapped), an empty/absent file, or any read/parse error — an uncapped or
|
|
2935
|
+
* unreadable limit means there is nothing meaningful to warn against.
|
|
2936
|
+
*/
|
|
2937
|
+
async function readScalar(file) {
|
|
2938
|
+
try {
|
|
2939
|
+
const raw = (await readFile(`${cgroupDir()}/${file}`, "utf8")).trim();
|
|
2940
|
+
if (raw === "" || raw === "max") return null;
|
|
2941
|
+
const n = Number(raw);
|
|
2942
|
+
return Number.isFinite(n) ? n : null;
|
|
2943
|
+
} catch (_error) {
|
|
2944
|
+
return null;
|
|
2945
|
+
}
|
|
2946
|
+
}
|
|
2947
|
+
/**
|
|
2948
|
+
* Read a cgroup v2 "flat keyed" file (one `key value` pair per line, e.g.
|
|
2949
|
+
* cpu.stat) and return the counter for `key`, or null when absent.
|
|
2950
|
+
*/
|
|
2951
|
+
async function readKeyedCounter(file, key) {
|
|
2952
|
+
try {
|
|
2953
|
+
const raw = await readFile(`${cgroupDir()}/${file}`, "utf8");
|
|
2954
|
+
for (const line of raw.split("\n")) {
|
|
2955
|
+
const [k, v] = line.trim().split(/\s+/);
|
|
2956
|
+
if (k === key) {
|
|
2957
|
+
const n = Number(v);
|
|
2958
|
+
return Number.isFinite(n) ? n : null;
|
|
2959
|
+
}
|
|
2960
|
+
}
|
|
2961
|
+
return null;
|
|
2962
|
+
} catch (_error) {
|
|
2963
|
+
return null;
|
|
2964
|
+
}
|
|
2965
|
+
}
|
|
2966
|
+
/**
|
|
2967
|
+
* Live memory usage as an integer percent of the hard cap, or null when
|
|
2968
|
+
* either side is unreadable/uncapped. Exported for tests.
|
|
2969
|
+
*/
|
|
2970
|
+
async function readMemUsePct() {
|
|
2971
|
+
const [current, max] = await Promise.all([readScalar("memory.current"), readScalar("memory.max")]);
|
|
2972
|
+
if (current === null || max === null || max <= 0) return null;
|
|
2973
|
+
return {
|
|
2974
|
+
pct: Math.floor(current / max * 100),
|
|
2975
|
+
currentBytes: current,
|
|
2976
|
+
maxBytes: max
|
|
2977
|
+
};
|
|
2978
|
+
}
|
|
2979
|
+
/**
|
|
2980
|
+
* Root filesystem used% (df -P Capacity column), or null on any failure.
|
|
2981
|
+
* Exported for tests.
|
|
2982
|
+
*/
|
|
2983
|
+
async function readDiskUsePct() {
|
|
2984
|
+
try {
|
|
2985
|
+
const { stdout } = await execFileAsync("df", ["-P", diskRoot()]);
|
|
2986
|
+
const dataRow = stdout.trim().split("\n")[1];
|
|
2987
|
+
if (dataRow == null) return null;
|
|
2988
|
+
const capacity = dataRow.trim().split(/\s+/)[4];
|
|
2989
|
+
if (capacity == null) return null;
|
|
2990
|
+
const pct = Number(capacity.replace("%", ""));
|
|
2991
|
+
return Number.isFinite(pct) ? pct : null;
|
|
2992
|
+
} catch (_error) {
|
|
2993
|
+
return null;
|
|
2994
|
+
}
|
|
2995
|
+
}
|
|
2996
|
+
/**
|
|
2997
|
+
* CPU utilization sampler. cgroup v2 exposes cumulative CPU time
|
|
2998
|
+
* (cpu.stat usage_usec); utilization is the delta between two samples over
|
|
2999
|
+
* the wall time between them, normalized by core count. The first call after
|
|
3000
|
+
* construction has no previous sample and returns null.
|
|
3001
|
+
*/
|
|
3002
|
+
function createCpuSampler() {
|
|
3003
|
+
let prevUsageUsec = null;
|
|
3004
|
+
let prevAtMs = null;
|
|
3005
|
+
return async () => {
|
|
3006
|
+
const usage = await readKeyedCounter("cpu.stat", "usage_usec");
|
|
3007
|
+
const now = Date.now();
|
|
3008
|
+
const prev = prevUsageUsec;
|
|
3009
|
+
const prevAt = prevAtMs;
|
|
3010
|
+
prevUsageUsec = usage;
|
|
3011
|
+
prevAtMs = now;
|
|
3012
|
+
if (usage === null || prev === null || prevAt === null) return null;
|
|
3013
|
+
const wallUsec = (now - prevAt) * 1e3;
|
|
3014
|
+
if (wallUsec <= 0) return null;
|
|
3015
|
+
const cores = availableParallelism();
|
|
3016
|
+
const pct = Math.round((usage - prev) / (wallUsec * cores) * 100);
|
|
3017
|
+
return Math.max(0, Math.min(100, pct));
|
|
3018
|
+
};
|
|
3019
|
+
}
|
|
3020
|
+
function fmtMb(bytes) {
|
|
3021
|
+
return Math.round(bytes / 1024 / 1024);
|
|
3022
|
+
}
|
|
3023
|
+
/** The model-facing warning text. Exported for tests. */
|
|
3024
|
+
function resourcePressureWarningText(trigger, { mem, cpuPct, diskPct }) {
|
|
3025
|
+
const stats = [];
|
|
3026
|
+
if (mem) stats.push(`memory ${mem.pct}% of cap (${fmtMb(mem.currentBytes)}/${fmtMb(mem.maxBytes)} MB)`);
|
|
3027
|
+
if (cpuPct !== null) stats.push(`CPU ${cpuPct}%`);
|
|
3028
|
+
if (diskPct !== null) stats.push(`disk ${diskPct}% full`);
|
|
3029
|
+
const lead = trigger === "memory" ? `Your sandbox is at ${mem?.pct}% of its memory cap. If usage keeps climbing, the kernel will kill the offending process and this turn may die with it.` : `Your sandbox's disk is ${diskPct}% full. If it fills completely, writes will start failing and this turn may die with them.`;
|
|
3030
|
+
const remedy = trigger === "memory" ? "checkpoint in-flight work (commit and push), then reduce the footprint — constrain parallelism, run heavy steps sequentially, or kill background processes you no longer need." : "checkpoint in-flight work (commit and push), then free space — clean build artifacts, caches, and scratch files you no longer need.";
|
|
3031
|
+
return `<system_notification>${lead} Current usage: ${stats.join(", ")}. Act now: ${remedy} If the workload genuinely needs more resources, request a bigger sandbox with \`platform compute request\`. This is an automated resource warning, not a message from the user; continue the task, adjusted.</system_notification>`;
|
|
3032
|
+
}
|
|
3033
|
+
const resourcePressureWarningExtension = (pi) => {
|
|
3034
|
+
let agentActive = false;
|
|
3035
|
+
let warnedMemThisTurn = false;
|
|
3036
|
+
let warnedDiskThisTurn = false;
|
|
3037
|
+
let timer = null;
|
|
3038
|
+
const sampleCpu = createCpuSampler();
|
|
3039
|
+
async function checkOnce() {
|
|
3040
|
+
if (!agentActive || warnedMemThisTurn && warnedDiskThisTurn) return;
|
|
3041
|
+
const [mem, cpuPct, diskPct] = await Promise.all([
|
|
3042
|
+
readMemUsePct(),
|
|
3043
|
+
sampleCpu(),
|
|
3044
|
+
readDiskUsePct()
|
|
3045
|
+
]);
|
|
3046
|
+
let trigger = null;
|
|
3047
|
+
if (!warnedMemThisTurn && mem !== null && mem.pct >= 80) {
|
|
3048
|
+
trigger = "memory";
|
|
3049
|
+
warnedMemThisTurn = true;
|
|
3050
|
+
} else if (!warnedDiskThisTurn && diskPct !== null && diskPct >= 80) {
|
|
3051
|
+
trigger = "disk";
|
|
3052
|
+
warnedDiskThisTurn = true;
|
|
3053
|
+
}
|
|
3054
|
+
if (trigger === null) return;
|
|
3055
|
+
log$9.warn({
|
|
3056
|
+
trigger,
|
|
3057
|
+
mem,
|
|
3058
|
+
cpuPct,
|
|
3059
|
+
diskPct
|
|
3060
|
+
}, "resource pressure warning delivered to agent");
|
|
3061
|
+
await pi.sendMessage({
|
|
3062
|
+
customType: "anyone-resource-pressure-warning",
|
|
3063
|
+
content: resourcePressureWarningText(trigger, {
|
|
3064
|
+
mem,
|
|
3065
|
+
cpuPct,
|
|
3066
|
+
diskPct
|
|
3067
|
+
}),
|
|
3068
|
+
display: false
|
|
3069
|
+
}, {
|
|
3070
|
+
triggerTurn: true,
|
|
3071
|
+
deliverAs: "followUp"
|
|
3072
|
+
});
|
|
3073
|
+
}
|
|
3074
|
+
pi.on("agent_start", async () => {
|
|
3075
|
+
agentActive = true;
|
|
3076
|
+
warnedMemThisTurn = false;
|
|
3077
|
+
warnedDiskThisTurn = false;
|
|
3078
|
+
if (!timer) {
|
|
3079
|
+
timer = setInterval(() => {
|
|
3080
|
+
checkOnce().catch((err) => {
|
|
3081
|
+
log$9.error({ err }, "resource pressure check failed");
|
|
3082
|
+
});
|
|
3083
|
+
}, POLL_INTERVAL_MS);
|
|
3084
|
+
timer.unref?.();
|
|
3085
|
+
}
|
|
3086
|
+
});
|
|
3087
|
+
pi.on("agent_end", async () => {
|
|
3088
|
+
agentActive = false;
|
|
3089
|
+
if (timer) {
|
|
3090
|
+
clearInterval(timer);
|
|
3091
|
+
timer = null;
|
|
3092
|
+
}
|
|
3093
|
+
});
|
|
3094
|
+
};
|
|
3095
|
+
//#endregion
|
|
3096
|
+
//#region src/extensions/disk-guard.ts
|
|
3097
|
+
const log$8 = logger.child({ module: "disk-guard" });
|
|
3098
|
+
/**
|
|
3099
|
+
* In-band bypass. The guard is a safety net, not a jail: when the agent knows
|
|
3100
|
+
* a flagged command is genuinely safe (writing to a different mount, a tiny
|
|
3101
|
+
* bounded download, a delete-then-clone one-liner, an emergency it accepts the
|
|
3102
|
+
* risk on) it can force the command through by appending this marker as a
|
|
3103
|
+
* trailing shell comment. Kept as a comment so it never changes what the
|
|
3104
|
+
* command does, and matched case-insensitively with flexible spacing so the
|
|
3105
|
+
* agent doesn't have to reproduce it byte-for-byte.
|
|
3106
|
+
*/
|
|
3107
|
+
const BYPASS_MARKER = /#\s*disk-guard:\s*allow\b/i;
|
|
3108
|
+
/** The exact marker text the block message tells the agent to append. */
|
|
3109
|
+
const BYPASS_HINT = "# disk-guard: allow";
|
|
3110
|
+
/**
|
|
3111
|
+
* Harness-level kill switch: set DISK_GUARD_DISABLE=1 to turn the guard off
|
|
3112
|
+
* entirely. This is the "I own my harness, let me opt out" knob — an agent
|
|
3113
|
+
* that boots its own harness can disable the guard for its whole process
|
|
3114
|
+
* without a code roll, and it's also the fleet-wide escape hatch if the
|
|
3115
|
+
* classifier ever misfires and blocks real work. The bare name is honored
|
|
3116
|
+
* first; the SKYDIVE_/ANYONE_ prefixes are accepted too for consistency with
|
|
3117
|
+
* the other env overrides. Empty/unset/"0"/"false" leave the guard on.
|
|
3118
|
+
*/
|
|
3119
|
+
function guardDisabledByEnv() {
|
|
3120
|
+
for (const name of [
|
|
3121
|
+
"DISK_GUARD_DISABLE",
|
|
3122
|
+
"SKYDIVE_DISK_GUARD_DISABLE",
|
|
3123
|
+
"ANYONE_DISK_GUARD_DISABLE"
|
|
3124
|
+
]) {
|
|
3125
|
+
const value = process.env[name];
|
|
3126
|
+
if (value != null && value !== "" && value !== "0" && value !== "false") return true;
|
|
3127
|
+
}
|
|
3128
|
+
return false;
|
|
3129
|
+
}
|
|
3130
|
+
/** True when the command carries the in-band bypass marker. */
|
|
3131
|
+
function hasBypassMarker(command) {
|
|
3132
|
+
return BYPASS_MARKER.test(command);
|
|
3133
|
+
}
|
|
3134
|
+
/**
|
|
3135
|
+
* Commands that reclaim space or merely inspect it. If any of these verbs
|
|
3136
|
+
* appears in the command line, we never block — otherwise the guard would trap
|
|
3137
|
+
* the agent by blocking the exact command it needs to dig out. Matched as
|
|
3138
|
+
* whole words so `remove-item` etc. don't accidentally match `rm`.
|
|
3139
|
+
*/
|
|
3140
|
+
const RECLAIM_PATTERNS = [
|
|
3141
|
+
/\brm\b/,
|
|
3142
|
+
/\brmdir\b/,
|
|
3143
|
+
/\bdf\b/,
|
|
3144
|
+
/\bdu\b/,
|
|
3145
|
+
/\bncdu\b/,
|
|
3146
|
+
/\bfind\b[^|]*\s-delete\b/,
|
|
3147
|
+
/\btruncate\b/,
|
|
3148
|
+
/\bgit\s+(gc|prune|clean|worktree\s+remove|worktree\s+prune)\b/,
|
|
3149
|
+
/\b(yarn|npm|pnpm|bun)\s+.*\b(cache\s+clean|cache\s+clear|store\s+prune)\b/,
|
|
3150
|
+
/\bcache\s+(clean|clear|prune)\b/,
|
|
3151
|
+
/\b(docker|podman)\s+.*\bprune\b/,
|
|
3152
|
+
/\bapt(-get)?\s+clean\b/,
|
|
3153
|
+
/\bjournalctl\b[^|]*--vacuum/
|
|
3154
|
+
];
|
|
3155
|
+
/**
|
|
3156
|
+
* File extensions that mean a download is actually LARGE — archives, disk
|
|
3157
|
+
* images, compiled/binary artifacts, model weights, media. A curl/wget is only
|
|
3158
|
+
* gated when it writes one of these; an API/page fetch to a `.json`/`.html`/
|
|
3159
|
+
* `.txt` file is tiny and must not be blocked. Derived from 4,144 real
|
|
3160
|
+
* commands: ~64% of `curl -o` uses were tiny fetches, only ~4% large.
|
|
3161
|
+
*/
|
|
3162
|
+
const BIG_DOWNLOAD_EXT = "(?:tar\\.gz|tgz|tar|zip|iso|gz|bz2|xz|zst|deb|rpm|pkg|dmg|whl|jar|7z|img|mp4|mov|avi|mkv|onnx|gguf|safetensors|bin|node)";
|
|
3163
|
+
/**
|
|
3164
|
+
* Commands that consume a meaningful amount of disk. Kept deliberately tight
|
|
3165
|
+
* and high-precision: validated against 4,144 real commands from the last 7
|
|
3166
|
+
* days, the earlier "writes a file" heuristic flagged 82% of everything (a
|
|
3167
|
+
* `curl -o /tmp/x.json` API call is not a disk event). This set flags ~33%,
|
|
3168
|
+
* almost all genuinely large — real installs, clones, big-archive downloads,
|
|
3169
|
+
* extractions. What was DROPPED and why:
|
|
3170
|
+
* - `git fetch` / `git pull` — incremental on an existing clone, usually tiny.
|
|
3171
|
+
* - `git checkout` — overwhelmingly `git checkout <ref> -- <file>` or a
|
|
3172
|
+
* branch switch, ~zero net growth; the rare full materialization isn't
|
|
3173
|
+
* worth the false-positive rate.
|
|
3174
|
+
* - bare `curl -o` / `wget -o` — see BIG_DOWNLOAD_EXT above.
|
|
3175
|
+
* - loose `… build` — matched `--mode=skip-build`, `oxfmt … build`, prose.
|
|
3176
|
+
* The remaining big-disk op in escher is `git clone` and `git worktree add`
|
|
3177
|
+
* (which is really a checkout), both kept.
|
|
3178
|
+
*/
|
|
3179
|
+
const SPACE_HUNGRY_PATTERNS = [
|
|
3180
|
+
/\bgit\s+clone\b/,
|
|
3181
|
+
/\bgit\s+worktree\s+add\b/,
|
|
3182
|
+
/\b(yarn|npm|pnpm|bun)\s+(install|add|ci)\b/,
|
|
3183
|
+
/\byarn\s*$/,
|
|
3184
|
+
/\byarn\s+--(?!version|help)\S/,
|
|
3185
|
+
/\bpip3?\s+install\b/,
|
|
3186
|
+
/\bapt(-get)?\s+install\b/,
|
|
3187
|
+
/\bnpm\s+pack\b/,
|
|
3188
|
+
/\bdocker\s+(build|pull)\b/,
|
|
3189
|
+
new RegExp(`\\b(?:curl|wget)\\b[^\\n]*\\s-[a-zA-Z]*[oO]\\b[^\\n]*\\.${BIG_DOWNLOAD_EXT}\\b`, "i"),
|
|
3190
|
+
new RegExp(`\\b(?:curl|wget)\\b[^\\n]*\\.${BIG_DOWNLOAD_EXT}\\b[^\\n]*\\s-[a-zA-Z]*[oO]\\b`, "i"),
|
|
3191
|
+
/\btar\s+[^\n|]*x[^\n|]*f/,
|
|
3192
|
+
/\bunzip\b/,
|
|
3193
|
+
/\bdd\b[^\n|]*\bof=/
|
|
3194
|
+
];
|
|
3195
|
+
/**
|
|
3196
|
+
* True when the command reclaims or inspects space — these are always allowed,
|
|
3197
|
+
* even on a 100%-full box, so the agent can dig itself out.
|
|
3198
|
+
*/
|
|
3199
|
+
function isReclaimCommand(command) {
|
|
3200
|
+
return RECLAIM_PATTERNS.some((re) => re.test(command));
|
|
3201
|
+
}
|
|
3202
|
+
/**
|
|
3203
|
+
* True when the command is likely to consume a meaningful amount of disk.
|
|
3204
|
+
* A reclaim/inspect command is never space-hungry — the reclaim check wins so a
|
|
3205
|
+
* `git worktree remove` or a `yarn cache clean` is never mistaken for growth.
|
|
3206
|
+
*/
|
|
3207
|
+
function isSpaceHungryCommand(command) {
|
|
3208
|
+
if (isReclaimCommand(command)) return false;
|
|
3209
|
+
return SPACE_HUNGRY_PATTERNS.some((re) => re.test(command));
|
|
3210
|
+
}
|
|
3211
|
+
/**
|
|
3212
|
+
* The decision, factored out and pure so it's exhaustively testable without a
|
|
3213
|
+
* real filesystem. Block only when we have a disk reading, it's at/above the
|
|
3214
|
+
* critical threshold, the command is space-hungry (and not a reclaim), and the
|
|
3215
|
+
* agent hasn't explicitly opted out with the bypass marker.
|
|
3216
|
+
*/
|
|
3217
|
+
function shouldBlockForDisk(command, diskPct) {
|
|
3218
|
+
if (diskPct === null) return false;
|
|
3219
|
+
if (diskPct < 95) return false;
|
|
3220
|
+
if (hasBypassMarker(command)) return false;
|
|
3221
|
+
return isSpaceHungryCommand(command);
|
|
3222
|
+
}
|
|
3223
|
+
/** The agent-facing explanation returned as the blocked tool result. */
|
|
3224
|
+
function diskBlockReason(command, diskPct) {
|
|
3225
|
+
return `Blocked: the sandbox disk is ${diskPct}% full and this command (\`${command.trim().slice(0, 120)}\`) writes a large amount, so it would fail partway with ENOSPC and leave a corrupt result. Reclaim space FIRST, then retry. Free ONLY what THIS conversation created — scratch/build output you wrote this run, downloads you're done with, and worktrees/branches whose work you've already committed and pushed (\`git worktree remove\`, \`yarn cache clean\`, delete your own scratch). Do NOT blindly wipe /tmp or delete a clone/worktree you don't recognize — other conversations share this box. Check headroom with \`df -h /\` and \`du -sh ~/workspace/* 2>/dev/null\`. If you genuinely can't free enough, stop and tell the user you're blocked on disk rather than retrying the write. If you're certain this command is safe anyway (writes elsewhere, tiny bounded size, delete-then-write), force it through by appending \` ${BYPASS_HINT}\` to the command.`;
|
|
3226
|
+
}
|
|
3227
|
+
const diskGuardExtension = (pi) => {
|
|
3228
|
+
pi.on("tool_call", async (event) => {
|
|
3229
|
+
if (event.toolName !== "bash") return;
|
|
3230
|
+
if (guardDisabledByEnv()) return;
|
|
3231
|
+
const command = event.input.command;
|
|
3232
|
+
if (typeof command !== "string" || command.length === 0) return;
|
|
3233
|
+
if (hasBypassMarker(command)) return;
|
|
3234
|
+
if (!isSpaceHungryCommand(command)) return;
|
|
3235
|
+
const diskPct = await readDiskUsePct();
|
|
3236
|
+
if (!shouldBlockForDisk(command, diskPct)) return;
|
|
3237
|
+
log$8.warn({
|
|
3238
|
+
diskPct,
|
|
3239
|
+
command: command.slice(0, 200)
|
|
3240
|
+
}, "blocked space-hungry bash command on near-full disk");
|
|
3241
|
+
return {
|
|
3242
|
+
block: true,
|
|
3243
|
+
reason: diskBlockReason(command, diskPct)
|
|
3244
|
+
};
|
|
3245
|
+
});
|
|
3246
|
+
};
|
|
3247
|
+
//#endregion
|
|
2025
3248
|
//#region src/extensions/context-management-trim.ts
|
|
2026
3249
|
const CLEARED_PLACEHOLDER = "[old tool result cleared to save context — re-run the tool or re-read the source to recover it]";
|
|
2027
3250
|
/** Rough token estimate (~4 chars/token); good enough for trigger decisions. */
|
|
@@ -2113,7 +3336,7 @@ function transformContextMessages(messages, config, now) {
|
|
|
2113
3336
|
}
|
|
2114
3337
|
//#endregion
|
|
2115
3338
|
//#region src/extensions/context-management.ts
|
|
2116
|
-
const log$
|
|
3339
|
+
const log$7 = logger.child({ module: "context-management-extension" });
|
|
2117
3340
|
function isAnthropicMessagesPayload(payload) {
|
|
2118
3341
|
if (typeof payload !== "object" || payload === null) return false;
|
|
2119
3342
|
const candidate = payload;
|
|
@@ -2178,13 +3401,13 @@ function createContextManagementExtension() {
|
|
|
2178
3401
|
setContextManagementFlagOverride(getPolledFlag("contextManagement"));
|
|
2179
3402
|
onFlagChange("contextManagement", (enabled) => {
|
|
2180
3403
|
setContextManagementFlagOverride(enabled);
|
|
2181
|
-
log$
|
|
3404
|
+
log$7.info({
|
|
2182
3405
|
event: "context_management_flag_update",
|
|
2183
3406
|
enabled
|
|
2184
3407
|
}, "context-management flag updated from platform");
|
|
2185
3408
|
});
|
|
2186
3409
|
startFeatureFlagPoller();
|
|
2187
|
-
log$
|
|
3410
|
+
log$7.info({
|
|
2188
3411
|
event: "context_management_registered",
|
|
2189
3412
|
enabled: initial.enabled,
|
|
2190
3413
|
flagSource: hasFlagSource(),
|
|
@@ -2196,13 +3419,13 @@ function createContextManagementExtension() {
|
|
|
2196
3419
|
const { messages } = event;
|
|
2197
3420
|
try {
|
|
2198
3421
|
const result = transformContextIfEnabled(messages, getContextManagementConfig(), Date.now());
|
|
2199
|
-
if (result.stats && (result.stats.clearedResults > 0 || result.stats.trimmedResults > 0)) log$
|
|
3422
|
+
if (result.stats && (result.stats.clearedResults > 0 || result.stats.trimmedResults > 0)) log$7.info({
|
|
2200
3423
|
event: "context_management_applied",
|
|
2201
3424
|
...result.stats
|
|
2202
3425
|
}, "trimmed/cleared tool output before LLM call");
|
|
2203
3426
|
return { messages: result.messages };
|
|
2204
3427
|
} catch (err) {
|
|
2205
|
-
log$
|
|
3428
|
+
log$7.error({
|
|
2206
3429
|
err,
|
|
2207
3430
|
event: "context_management_transform_failed"
|
|
2208
3431
|
}, "context transform failed; passing messages through unchanged");
|
|
@@ -2214,7 +3437,7 @@ function createContextManagementExtension() {
|
|
|
2214
3437
|
}
|
|
2215
3438
|
//#endregion
|
|
2216
3439
|
//#region src/extensions/current-time.ts
|
|
2217
|
-
const log$
|
|
3440
|
+
const log$6 = logger.child({ module: "current-time-extension" });
|
|
2218
3441
|
const PI_DATE_LINE = /^Current date:.*$/m;
|
|
2219
3442
|
function formatCurrentTimeLine(now) {
|
|
2220
3443
|
return `Current date: ${now.getUTCFullYear()}-${String(now.getUTCMonth() + 1).padStart(2, "0")}-${String(now.getUTCDate()).padStart(2, "0")} (${new Intl.DateTimeFormat("en-US", {
|
|
@@ -2227,7 +3450,7 @@ const currentTimeExtension = (pi) => {
|
|
|
2227
3450
|
const line = formatCurrentTimeLine(/* @__PURE__ */ new Date());
|
|
2228
3451
|
const base = event.systemPrompt;
|
|
2229
3452
|
if (PI_DATE_LINE.test(base)) {
|
|
2230
|
-
log$
|
|
3453
|
+
log$6.info({ event: "pi_date_line_present" }, "pi base prompt carries its own 'Current date:' line again; replacing it in place (pi prompt format may have changed)");
|
|
2231
3454
|
return { systemPrompt: base.replace(PI_DATE_LINE, line) };
|
|
2232
3455
|
}
|
|
2233
3456
|
return { systemPrompt: `${base}\n${line}` };
|
|
@@ -2236,37 +3459,35 @@ const currentTimeExtension = (pi) => {
|
|
|
2236
3459
|
//#endregion
|
|
2237
3460
|
//#region src/memory.ts
|
|
2238
3461
|
/**
|
|
2239
|
-
* In-harness memory
|
|
3462
|
+
* In-harness memory readers.
|
|
2240
3463
|
*
|
|
2241
|
-
* The agent has an agent-level file-based memory at `<cwd>/.memory/`,
|
|
2242
|
-
*
|
|
3464
|
+
* The agent has an agent-level file-based memory at `<cwd>/.memory/`, split
|
|
3465
|
+
* into two halves that are surfaced differently:
|
|
2243
3466
|
*
|
|
2244
|
-
*
|
|
2245
|
-
*
|
|
2246
|
-
*
|
|
2247
|
-
*
|
|
3467
|
+
* 1. Shared knowledge (projects, lessons, external systems) — indexed by a
|
|
3468
|
+
* single hand-maintained `<cwd>/.memory/MEMORY.md` that the *agent* writes
|
|
3469
|
+
* and curates, Claude-Code style: one line per fact pointing at the file
|
|
3470
|
+
* that holds it. `readMemoryIndexFile` just reads that file; the agent owns
|
|
3471
|
+
* its contents. This is the whole index for the shared half — there is no
|
|
3472
|
+
* derived walk and no per-directory `MEMORY.md`.
|
|
2248
3473
|
*
|
|
2249
|
-
*
|
|
2250
|
-
*
|
|
2251
|
-
*
|
|
3474
|
+
* 2. Per-person memory — `.memory/users/<id>-<name>/<topic>.md`. This half is
|
|
3475
|
+
* *derived*, not hand-maintained, because it has to be filtered to the one
|
|
3476
|
+
* person on the current turn (a single hand-written index couldn't be
|
|
3477
|
+
* scoped per-user without leaking one person's notes into another's
|
|
3478
|
+
* conversation). `buildMemoryIndex` walks a single user's directory, reads
|
|
3479
|
+
* only the frontmatter of each `.md` (open fd → read first ~4KB → close, in
|
|
3480
|
+
* parallel), and renders an index. Each `.md`'s frontmatter carries `name`
|
|
3481
|
+
* and `description`; bodies are never read — the agent loads a specific
|
|
3482
|
+
* memory's body on demand via the `read` tool.
|
|
2252
3483
|
*
|
|
2253
|
-
*
|
|
2254
|
-
*
|
|
2255
|
-
*
|
|
2256
|
-
*
|
|
2257
|
-
* ignored. Bodies are never read — the agent loads a specific memory's
|
|
2258
|
-
* body on demand via the `read` tool when the index entry says it's
|
|
2259
|
-
* relevant.
|
|
3484
|
+
* Mtime cache (for the derived per-user half) keyed by cwd — within the
|
|
3485
|
+
* lifetime of a sandbox the cwd is fixed, so this is effectively a single-entry
|
|
3486
|
+
* cache. It invalidates when any `.md` under `users/` is added/modified/
|
|
3487
|
+
* deleted; turns where memory didn't change reuse the cached entries.
|
|
2260
3488
|
*
|
|
2261
|
-
*
|
|
2262
|
-
*
|
|
2263
|
-
* when any `.md` in the tree is added/modified/deleted; turns where
|
|
2264
|
-
* memory didn't change reuse the cached string.
|
|
2265
|
-
*
|
|
2266
|
-
* Frontmatter is parsed as YAML (`yaml` package) and validated with a
|
|
2267
|
-
* zod schema — files that don't match the shape are dropped from the
|
|
2268
|
-
* index. The same schema can be reused at write time if we want to
|
|
2269
|
-
* validate before commit.
|
|
3489
|
+
* Frontmatter is parsed as YAML (`yaml` package) and validated with a zod
|
|
3490
|
+
* schema — files that don't match the shape are dropped from the index.
|
|
2270
3491
|
*/
|
|
2271
3492
|
const FRONTMATTER_READ_BYTES = 4096;
|
|
2272
3493
|
const FrontmatterSchema = z.object({
|
|
@@ -2274,33 +3495,118 @@ const FrontmatterSchema = z.object({
|
|
|
2274
3495
|
description: z.string().min(1)
|
|
2275
3496
|
}).passthrough();
|
|
2276
3497
|
const MEMORY_DIRNAME = ".memory";
|
|
2277
|
-
const
|
|
2278
|
-
|
|
2279
|
-
|
|
2280
|
-
|
|
2281
|
-
|
|
3498
|
+
const MEMORY_INDEX_FILENAME = "MEMORY.md";
|
|
3499
|
+
const USERS_DIRNAME = "users";
|
|
3500
|
+
/**
|
|
3501
|
+
* The shared-knowledge type dirs from the old frontmatter-indexed layout, used
|
|
3502
|
+
* only to seed a `MEMORY.md` for agents created before it existed (see
|
|
3503
|
+
* `seedMemoryIndexFile`). `users/` is deliberately excluded — per-person memory
|
|
3504
|
+
* stays derived and never lands in the shared, un-scoped `MEMORY.md`.
|
|
3505
|
+
*/
|
|
3506
|
+
const LEGACY_SHARED_TYPES = [
|
|
3507
|
+
{
|
|
3508
|
+
dir: "projects",
|
|
3509
|
+
label: "Projects"
|
|
3510
|
+
},
|
|
3511
|
+
{
|
|
3512
|
+
dir: "feedback",
|
|
3513
|
+
label: "Feedback"
|
|
3514
|
+
},
|
|
3515
|
+
{
|
|
3516
|
+
dir: "reference",
|
|
3517
|
+
label: "Reference"
|
|
3518
|
+
}
|
|
2282
3519
|
];
|
|
2283
|
-
|
|
2284
|
-
|
|
2285
|
-
|
|
2286
|
-
|
|
2287
|
-
|
|
2288
|
-
|
|
2289
|
-
|
|
3520
|
+
/**
|
|
3521
|
+
* Soft budget for an injected index block. The index is read and injected into
|
|
3522
|
+
* the system prompt on every turn, so every entry costs context for the rest of
|
|
3523
|
+
* the conversation. Past this size we nudge the agent to consolidate and prune
|
|
3524
|
+
* rather than keep appending. Not a hard cap — nothing is truncated.
|
|
3525
|
+
*/
|
|
3526
|
+
const MEMORY_INDEX_SOFT_BUDGET_CHARS = 2e4;
|
|
3527
|
+
/**
|
|
3528
|
+
* Hard cap on the injected index — double the soft budget. The soft budget only
|
|
3529
|
+
* warns; this actually bounds what we inject so a runaway index can't consume
|
|
3530
|
+
* unbounded context on every turn. Past this, the index is truncated (on a line
|
|
3531
|
+
* boundary) before injection. It's a backstop, not a normal operating point.
|
|
3532
|
+
*/
|
|
3533
|
+
const MEMORY_INDEX_HARD_BUDGET_CHARS = MEMORY_INDEX_SOFT_BUDGET_CHARS * 2;
|
|
3534
|
+
/**
|
|
3535
|
+
* Cheap size summary of a rendered index block, used to surface how much of the
|
|
3536
|
+
* every-turn context budget the index is spending so the agent keeps it lean.
|
|
3537
|
+
* Counts pointer/file lines — both the derived `` - `path` — desc`` form and
|
|
3538
|
+
* the hand-maintained `- [Title](path) — hook` form — not the group headers.
|
|
3539
|
+
*/
|
|
3540
|
+
function summarizeIndex(index) {
|
|
3541
|
+
const entryCount = index.split("\n").filter((line) => /^\s*- (?:`|\[)/.test(line)).length;
|
|
3542
|
+
const charCount = index.length;
|
|
3543
|
+
return {
|
|
3544
|
+
entryCount,
|
|
3545
|
+
charCount,
|
|
3546
|
+
overBudget: charCount > MEMORY_INDEX_SOFT_BUDGET_CHARS
|
|
3547
|
+
};
|
|
3548
|
+
}
|
|
3549
|
+
/**
|
|
3550
|
+
* One-line size note for an index block header, e.g. `12 entries, 3187 chars`.
|
|
3551
|
+
*/
|
|
3552
|
+
function indexSizeNote(index) {
|
|
3553
|
+
const { entryCount, charCount } = summarizeIndex(index);
|
|
3554
|
+
return `${entryCount} ${entryCount === 1 ? "entry" : "entries"}, ${charCount} chars`;
|
|
3555
|
+
}
|
|
3556
|
+
/**
|
|
3557
|
+
* An explicit warning to surface to the agent when an index has grown past its
|
|
3558
|
+
* budget, or `null` when it's within budget. Extensions render this prominently
|
|
3559
|
+
* above the index so the agent prunes before it keeps appending.
|
|
3560
|
+
*/
|
|
3561
|
+
function indexBudgetWarning(index) {
|
|
3562
|
+
const { charCount, overBudget } = summarizeIndex(index);
|
|
3563
|
+
if (!overBudget) return null;
|
|
3564
|
+
return `⚠️ This memory index is ${charCount} chars, over its ${MEMORY_INDEX_SOFT_BUDGET_CHARS}-char budget. It's costing you context on every turn — consolidate duplicate entries and delete stale ones to bring it back under budget before adding anything new.`;
|
|
3565
|
+
}
|
|
3566
|
+
/**
|
|
3567
|
+
* Enforce the hard cap on an index before injection. Under the cap the index is
|
|
3568
|
+
* returned unchanged; over it, the index is truncated on a line boundary and a
|
|
3569
|
+
* notice is appended naming the true size so the agent knows entries are hidden
|
|
3570
|
+
* and must be pruned. This is the actual bound on injected context — callers
|
|
3571
|
+
* still report the true size via {@link indexSizeNote} so nothing is masked.
|
|
3572
|
+
*/
|
|
3573
|
+
function enforceMemoryIndexHardBudget(index) {
|
|
3574
|
+
if (index.length <= 4e4) return index;
|
|
3575
|
+
const clipped = index.slice(0, MEMORY_INDEX_HARD_BUDGET_CHARS);
|
|
3576
|
+
const lastNewline = clipped.lastIndexOf("\n");
|
|
3577
|
+
return `${lastNewline > 0 ? clipped.slice(0, lastNewline) : clipped}\n\n⚠️ Memory index truncated at ${MEMORY_INDEX_HARD_BUDGET_CHARS} chars (it is ${index.length}). Entries past this point are NOT shown. Prune the index now — delete stale entries and consolidate duplicates.`;
|
|
3578
|
+
}
|
|
3579
|
+
/**
|
|
3580
|
+
* Read the agent's hand-maintained shared index at `.memory/MEMORY.md`.
|
|
3581
|
+
* Returns the trimmed contents, or `null` when the file is absent or empty —
|
|
3582
|
+
* the agent owns this file, so we surface exactly what it wrote.
|
|
3583
|
+
*/
|
|
3584
|
+
async function readMemoryIndexFile({ cwd }) {
|
|
3585
|
+
const path = join(cwd, MEMORY_DIRNAME, MEMORY_INDEX_FILENAME);
|
|
3586
|
+
try {
|
|
3587
|
+
const trimmed = (await readFile(path, "utf-8")).trim();
|
|
3588
|
+
return trimmed.length > 0 ? trimmed : null;
|
|
3589
|
+
} catch {
|
|
3590
|
+
return null;
|
|
3591
|
+
}
|
|
3592
|
+
}
|
|
2290
3593
|
const cache = /* @__PURE__ */ new Map();
|
|
2291
3594
|
/**
|
|
3595
|
+
* Build the derived per-person index for a single user.
|
|
3596
|
+
*
|
|
2292
3597
|
* Returns:
|
|
2293
|
-
* - `null` if `.memory/` doesn't exist
|
|
2294
|
-
* - `""` if
|
|
3598
|
+
* - `null` if `.memory/users/` doesn't exist
|
|
3599
|
+
* - `""` if it exists but this user has no memory
|
|
2295
3600
|
* - rendered markdown body (no surrounding header — caller wraps)
|
|
2296
3601
|
*
|
|
2297
|
-
*
|
|
2298
|
-
*
|
|
2299
|
-
*
|
|
3602
|
+
* Scoping by id keeps one person's memory from bleeding into another's
|
|
3603
|
+
* conversation. The mtime-keyed cache stores the raw walked entries (the cost
|
|
3604
|
+
* is the FS walk); filtering by user is cheap and runs per call, so two turns
|
|
3605
|
+
* with different users on the same cwd render correctly from one cached walk.
|
|
2300
3606
|
*/
|
|
2301
|
-
async function buildMemoryIndex({ cwd,
|
|
2302
|
-
const
|
|
2303
|
-
const maxMtimeMs = await maxMtimeAcrossDir(
|
|
3607
|
+
async function buildMemoryIndex({ cwd, userId }) {
|
|
3608
|
+
const usersDirAbs = join(cwd, MEMORY_DIRNAME, USERS_DIRNAME);
|
|
3609
|
+
const maxMtimeMs = await maxMtimeAcrossDir(usersDirAbs);
|
|
2304
3610
|
if (maxMtimeMs === null) {
|
|
2305
3611
|
cache.delete(cwd);
|
|
2306
3612
|
return null;
|
|
@@ -2308,13 +3614,13 @@ async function buildMemoryIndex({ cwd, scope }) {
|
|
|
2308
3614
|
let cached = cache.get(cwd);
|
|
2309
3615
|
if (!cached || cached.builtAtMs < maxMtimeMs) {
|
|
2310
3616
|
cached = {
|
|
2311
|
-
entries: await
|
|
3617
|
+
entries: await collectUserEntries(usersDirAbs, cwd),
|
|
2312
3618
|
builtAtMs: Date.now()
|
|
2313
3619
|
};
|
|
2314
3620
|
cache.set(cwd, cached);
|
|
2315
3621
|
}
|
|
2316
|
-
const visible = cached.entries.filter((entry) =>
|
|
2317
|
-
return visible.length === 0 ? "" :
|
|
3622
|
+
const visible = cached.entries.filter((entry) => entry.subject.startsWith(userId));
|
|
3623
|
+
return visible.length === 0 ? "" : renderUserIndex(visible);
|
|
2318
3624
|
}
|
|
2319
3625
|
async function maxMtimeAcrossDir(dir) {
|
|
2320
3626
|
let dirStat;
|
|
@@ -2362,45 +3668,22 @@ async function listSubdirs(dir) {
|
|
|
2362
3668
|
}
|
|
2363
3669
|
return entries.filter((e) => e.isDirectory()).map((e) => join(dir, e.name));
|
|
2364
3670
|
}
|
|
2365
|
-
async function
|
|
2366
|
-
const
|
|
2367
|
-
await Promise.all(
|
|
2368
|
-
const
|
|
2369
|
-
|
|
2370
|
-
|
|
2371
|
-
await
|
|
2372
|
-
|
|
2373
|
-
|
|
2374
|
-
|
|
2375
|
-
|
|
2376
|
-
|
|
2377
|
-
|
|
2378
|
-
|
|
2379
|
-
|
|
2380
|
-
|
|
2381
|
-
subject,
|
|
2382
|
-
relPath: relative(cwd, file)
|
|
2383
|
-
};
|
|
2384
|
-
}));
|
|
2385
|
-
for (const e of parsed) if (e) collected.push(e);
|
|
2386
|
-
}));
|
|
2387
|
-
} else {
|
|
2388
|
-
const files = await listMdFilesShallow(typeDirAbs);
|
|
2389
|
-
const parsed = await Promise.all(files.map(async (file) => {
|
|
2390
|
-
const fm = await readFrontmatterOnly(file);
|
|
2391
|
-
if (!fm?.name || !fm?.description) return null;
|
|
2392
|
-
return {
|
|
2393
|
-
name: fm.name,
|
|
2394
|
-
description: fm.description,
|
|
2395
|
-
type,
|
|
2396
|
-
subject: null,
|
|
2397
|
-
relPath: relative(cwd, file)
|
|
2398
|
-
};
|
|
2399
|
-
}));
|
|
2400
|
-
for (const e of parsed) if (e) collected.push(e);
|
|
2401
|
-
}
|
|
2402
|
-
}));
|
|
2403
|
-
return collected;
|
|
3671
|
+
async function collectUserEntries(usersDirAbs, cwd) {
|
|
3672
|
+
const subjectDirs = await listSubdirs(usersDirAbs);
|
|
3673
|
+
return (await Promise.all(subjectDirs.map(async (subjectDirAbs) => {
|
|
3674
|
+
const subject = basename(subjectDirAbs);
|
|
3675
|
+
const files = await listMdFilesShallow(subjectDirAbs);
|
|
3676
|
+
return (await Promise.all(files.map(async (file) => {
|
|
3677
|
+
const fm = await readFrontmatterOnly(file);
|
|
3678
|
+
if (!fm?.name || !fm?.description) return null;
|
|
3679
|
+
return {
|
|
3680
|
+
name: fm.name,
|
|
3681
|
+
description: fm.description,
|
|
3682
|
+
subject,
|
|
3683
|
+
relPath: relative(cwd, file)
|
|
3684
|
+
};
|
|
3685
|
+
}))).filter((e) => e !== null);
|
|
3686
|
+
}))).flat();
|
|
2404
3687
|
}
|
|
2405
3688
|
async function readFrontmatterOnly(filePath) {
|
|
2406
3689
|
let fh;
|
|
@@ -2431,77 +3714,189 @@ function parseFrontmatter(text) {
|
|
|
2431
3714
|
const result = FrontmatterSchema.safeParse(parsed);
|
|
2432
3715
|
return result.success ? result.data : null;
|
|
2433
3716
|
}
|
|
2434
|
-
function
|
|
2435
|
-
const
|
|
2436
|
-
|
|
2437
|
-
|
|
2438
|
-
|
|
2439
|
-
|
|
2440
|
-
}
|
|
2441
|
-
|
|
3717
|
+
function renderUserIndex(entries) {
|
|
3718
|
+
const bySubject = /* @__PURE__ */ new Map();
|
|
3719
|
+
for (const e of entries) {
|
|
3720
|
+
const list = bySubject.get(e.subject) ?? [];
|
|
3721
|
+
list.push(e);
|
|
3722
|
+
bySubject.set(e.subject, list);
|
|
3723
|
+
}
|
|
3724
|
+
const lines = ["### Users"];
|
|
3725
|
+
for (const subject of [...bySubject.keys()].sort()) {
|
|
3726
|
+
lines.push(`- **${subject}**`);
|
|
3727
|
+
for (const e of bySubject.get(subject) ?? []) lines.push(` - \`${e.relPath}\` — ${e.description}`);
|
|
3728
|
+
}
|
|
3729
|
+
return lines.join("\n");
|
|
3730
|
+
}
|
|
3731
|
+
/**
|
|
3732
|
+
* One-time migration for agents created before `MEMORY.md` existed. If there is
|
|
3733
|
+
* no hand-maintained `.memory/MEMORY.md` yet but the agent has shared memory
|
|
3734
|
+
* files from the old frontmatter-indexed layout (`projects/`, `feedback/`,
|
|
3735
|
+
* `reference/`), derive a `MEMORY.md` from their frontmatter and write it once.
|
|
3736
|
+
* After that the agent owns the file — this never runs again for that agent and
|
|
3737
|
+
* never clobbers an existing index.
|
|
3738
|
+
*
|
|
3739
|
+
* Returns the seeded contents (also written to disk), or `null` when nothing
|
|
3740
|
+
* was seeded (index already present, or no legacy shared files). A write
|
|
3741
|
+
* failure propagates so the caller can log it; the read path then falls back to
|
|
3742
|
+
* whatever is on disk.
|
|
3743
|
+
*/
|
|
3744
|
+
async function seedMemoryIndexFile({ cwd }) {
|
|
3745
|
+
const memoryDirAbs = join(cwd, MEMORY_DIRNAME);
|
|
3746
|
+
const indexPath = join(memoryDirAbs, MEMORY_INDEX_FILENAME);
|
|
3747
|
+
if (await stat(indexPath).catch(() => null)) return null;
|
|
3748
|
+
const perType = await Promise.all(LEGACY_SHARED_TYPES.map(async ({ dir, label }) => {
|
|
3749
|
+
const typeDirAbs = join(memoryDirAbs, dir);
|
|
3750
|
+
const files = [];
|
|
3751
|
+
await walkMdFiles(typeDirAbs, files);
|
|
3752
|
+
return {
|
|
3753
|
+
label,
|
|
3754
|
+
entries: (await Promise.all(files.map(async (file) => {
|
|
3755
|
+
const fm = await readFrontmatterOnly(file);
|
|
3756
|
+
if (!fm?.name || !fm?.description) return null;
|
|
3757
|
+
return {
|
|
3758
|
+
name: fm.name,
|
|
3759
|
+
description: fm.description,
|
|
3760
|
+
relPath: relative(cwd, file)
|
|
3761
|
+
};
|
|
3762
|
+
}))).filter((e) => !!e)
|
|
3763
|
+
};
|
|
3764
|
+
}));
|
|
3765
|
+
if (perType.every((group) => group.entries.length === 0)) return null;
|
|
3766
|
+
const content = renderSeededIndex(perType);
|
|
3767
|
+
await writeFile(indexPath, `${content}\n`, "utf-8");
|
|
3768
|
+
return content;
|
|
3769
|
+
}
|
|
3770
|
+
function renderSeededIndex(groups) {
|
|
2442
3771
|
const sections = [];
|
|
2443
|
-
for (const
|
|
2444
|
-
|
|
2445
|
-
|
|
2446
|
-
|
|
2447
|
-
|
|
2448
|
-
const bySubject = /* @__PURE__ */ new Map();
|
|
2449
|
-
for (const e of items) {
|
|
2450
|
-
const subject = e.subject ?? "(unknown)";
|
|
2451
|
-
const list = bySubject.get(subject) ?? [];
|
|
2452
|
-
list.push(e);
|
|
2453
|
-
bySubject.set(subject, list);
|
|
2454
|
-
}
|
|
2455
|
-
const subjects = [...bySubject.keys()].sort();
|
|
2456
|
-
for (const subject of subjects) {
|
|
2457
|
-
sections.push(`- **${subject}**`);
|
|
2458
|
-
for (const e of bySubject.get(subject) ?? []) sections.push(` - \`${e.relPath}\` — ${e.description}`);
|
|
2459
|
-
}
|
|
2460
|
-
} else for (const e of items) sections.push(`- \`${e.relPath}\` — ${e.description}`);
|
|
3772
|
+
for (const { label, entries } of groups) {
|
|
3773
|
+
if (entries.length === 0) continue;
|
|
3774
|
+
sections.push(`### ${label}`);
|
|
3775
|
+
const sorted = [...entries].sort((a, b) => a.relPath.localeCompare(b.relPath));
|
|
3776
|
+
for (const e of sorted) sections.push(`- [${e.name}](${e.relPath}) — ${e.description}`);
|
|
2461
3777
|
sections.push("");
|
|
2462
3778
|
}
|
|
2463
3779
|
return sections.join("\n").trimEnd();
|
|
2464
3780
|
}
|
|
3781
|
+
/**
|
|
3782
|
+
* The version at which the hand-maintained `MEMORY.md` layout was introduced.
|
|
3783
|
+
* Used to classify an unversioned `.memory/`: if it already has a `MEMORY.md`
|
|
3784
|
+
* it's on this layout (not the pre-MEMORY.md v1 frontmatter layout), so it
|
|
3785
|
+
* shouldn't be treated as v1 and re-seeded.
|
|
3786
|
+
*/
|
|
3787
|
+
const HAND_MAINTAINED_INDEX_VERSION = 2;
|
|
3788
|
+
const VERSION_FILENAME = ".version";
|
|
3789
|
+
const MEMORY_MIGRATIONS = [{
|
|
3790
|
+
from: 1,
|
|
3791
|
+
to: 2,
|
|
3792
|
+
apply: async ({ cwd }) => {
|
|
3793
|
+
await seedMemoryIndexFile({ cwd });
|
|
3794
|
+
}
|
|
3795
|
+
}];
|
|
3796
|
+
/**
|
|
3797
|
+
* The layout version of an agent's `.memory/`:
|
|
3798
|
+
* - `null` when there's no `.memory/` at all (a fresh agent is current by
|
|
3799
|
+
* construction; nothing to migrate).
|
|
3800
|
+
* - `1` when `.memory/` exists but carries no `.version` marker — i.e. it
|
|
3801
|
+
* predates versioning.
|
|
3802
|
+
* - otherwise the integer in `.memory/.version`.
|
|
3803
|
+
*/
|
|
3804
|
+
async function readMemoryVersion(cwd) {
|
|
3805
|
+
const memoryDirAbs = join(cwd, MEMORY_DIRNAME);
|
|
3806
|
+
if (!(await stat(memoryDirAbs).catch(() => null))?.isDirectory()) return null;
|
|
3807
|
+
const raw = await readFile(join(memoryDirAbs, VERSION_FILENAME), "utf-8").catch(() => null);
|
|
3808
|
+
if (raw === null) return await stat(join(memoryDirAbs, MEMORY_INDEX_FILENAME)).then((s) => s.isFile()).catch(() => false) ? HAND_MAINTAINED_INDEX_VERSION : 1;
|
|
3809
|
+
const parsed = Number.parseInt(raw.trim(), 10);
|
|
3810
|
+
return Number.isInteger(parsed) && parsed > 0 ? parsed : 1;
|
|
3811
|
+
}
|
|
3812
|
+
async function writeMemoryVersion(cwd, version) {
|
|
3813
|
+
await writeFile(join(cwd, MEMORY_DIRNAME, VERSION_FILENAME), `${version}\n`, "utf-8");
|
|
3814
|
+
}
|
|
3815
|
+
/**
|
|
3816
|
+
* Bring an agent's `.memory/` up to `CURRENT_MEMORY_VERSION` by applying the
|
|
3817
|
+
* ordered migrations. Runs on session start. No-op when there's no `.memory/`
|
|
3818
|
+
* yet or it's already current. Migrations must be idempotent, so a lost/unwritten
|
|
3819
|
+
* version marker (the file isn't committed by the harness) only costs a repeated
|
|
3820
|
+
* no-op, never corruption. Returns the `{ from, to }` actually applied, or
|
|
3821
|
+
* `null` when nothing ran.
|
|
3822
|
+
*/
|
|
3823
|
+
async function migrateMemory({ cwd }) {
|
|
3824
|
+
const from = await readMemoryVersion(cwd);
|
|
3825
|
+
if (from === null || from >= 2) return null;
|
|
3826
|
+
let version = from;
|
|
3827
|
+
while (version < 2) {
|
|
3828
|
+
const migration = MEMORY_MIGRATIONS.find((m) => m.from === version);
|
|
3829
|
+
if (!migration) break;
|
|
3830
|
+
await migration.apply({ cwd });
|
|
3831
|
+
version = migration.to;
|
|
3832
|
+
}
|
|
3833
|
+
await writeMemoryVersion(cwd, version);
|
|
3834
|
+
return {
|
|
3835
|
+
from,
|
|
3836
|
+
to: version
|
|
3837
|
+
};
|
|
3838
|
+
}
|
|
2465
3839
|
//#endregion
|
|
2466
3840
|
//#region src/extensions/memory.ts
|
|
2467
|
-
const log$
|
|
3841
|
+
const log$5 = logger.child({ module: "memory-extension" });
|
|
2468
3842
|
/**
|
|
2469
3843
|
* The standing instructions for the memory system. Always injected (even with
|
|
2470
|
-
*
|
|
3844
|
+
* no `MEMORY.md`) so the agent knows it can persist notes and how. `users/` is
|
|
2471
3845
|
* described by the platform memory extension, which is the only thing that can
|
|
2472
3846
|
* scope it to a person — here we just point at it.
|
|
2473
3847
|
*/
|
|
2474
3848
|
function memoryInstructions(cwd) {
|
|
2475
3849
|
return `## Memory across conversations
|
|
2476
3850
|
|
|
2477
|
-
Persistent notes across conversations live at \`${cwd}/.memory/\` — plain markdown files in your repo
|
|
3851
|
+
Persistent notes across conversations live at \`${cwd}/.memory/\` — plain markdown files in your repo, one fact per file. You maintain a hand-written index of them at \`${cwd}/.memory/MEMORY.md\`, and the harness injects that index into your system prompt every turn. **Bodies are NOT auto-loaded** — when an index line looks relevant, use your \`read\` tool to load that specific file.
|
|
2478
3852
|
|
|
2479
3853
|
Memory records **what happened**: facts you learned, events, investigation findings, project and system details worth carrying forward. It is NOT where behavior goes. A standing rule about how you should act — a "from now on, always/never …", a tone or format preference, a workflow convention a user wants you to follow — belongs in \`soul.md\` (see the Persona / Standing instructions section), not here. When a note is really an instruction about your behavior, write it to \`soul.md\`; when it is a fact or a record of something that occurred, write it here.
|
|
2480
3854
|
|
|
2481
|
-
Shared knowledge is laid out as \`projects/<slug>/<topic>.md\` for project and system context, \`feedback/<topic>.md\` for concrete lessons learned from something that happened (the event and what it taught you — not a free-floating rule; the rule itself, if durable, goes in \`soul.md\`), and \`reference/<topic>.md\` for how external systems work. (Notes about a specific person live under \`users/\` and are
|
|
3855
|
+
You own \`MEMORY.md\`. When you learn something durable, write the fact to its own \`.md\` file and add a one-line pointer to \`MEMORY.md\` in the form \`- [Title](relative/path.md) — one-line hook\`, where the hook is what tells future-you when to open the file. \`MEMORY.md\` is an *index*, never a store — put the actual content in the topic file and only a pointer line in \`MEMORY.md\`; do not inline a fact's body into the index even when it seems cheaper. Start each topic file with \`name:\`/\`description:\` frontmatter (the \`description\` is the one-line hook) so the index can be re-seeded, re-derived, or linted from the files themselves. When a fact changes, edit both the file and its line; when it stops being true, delete the file and its line. Shared knowledge is laid out as \`projects/<slug>/<topic>.md\` for project and system context, \`feedback/<topic>.md\` for concrete lessons learned from something that happened (the event and what it taught you — not a free-floating rule; the rule itself, if durable, goes in \`soul.md\`), and \`reference/<topic>.md\` for how external systems work. (Notes about a specific person live under \`users/\` and are indexed for you separately, scoped to whoever you're talking to — don't put people's private notes in the shared \`MEMORY.md\`.)
|
|
3856
|
+
|
|
3857
|
+
Keep \`MEMORY.md\` lean. It's re-injected on *every* turn, so a small, high-signal index is worth far more than an exhaustive one — curate it like a tightly-edited table of contents, not a log:
|
|
3858
|
+
- Be selective. Only record something durable that will matter in a *future* conversation. Don't record what only matters right now, what you can re-derive on demand, or what's already obvious from the repo.
|
|
3859
|
+
- Consolidate before you create. Before adding a line, scan \`MEMORY.md\` for one that already covers the topic; if it exists, \`read\` that file and rewrite it with the new facts merged in rather than adding a near-duplicate. One fact per file, but don't fragment a topic across many thin files.
|
|
3860
|
+
- Prune as you go. Delete lines (and their files) that are wrong, stale, or superseded. The index header reports its size — when it's flagged over budget, consolidate and delete before adding anything new.
|
|
3861
|
+
- Write dates absolute, not relative. Resolve "last week" / "yesterday" to a concrete date when you record it (e.g. "on 7/3 Dhruv told me to …"), so the note still reads correctly in a future conversation.
|
|
3862
|
+
|
|
3863
|
+
Commit and push after editing \`.memory/\` to persist it.`;
|
|
2482
3864
|
}
|
|
2483
3865
|
function composeBlock$1({ cwd, index }) {
|
|
2484
3866
|
const instructions = memoryInstructions(cwd);
|
|
2485
3867
|
if (!index || index.length === 0) return instructions;
|
|
2486
|
-
|
|
3868
|
+
const header = `## Memory index (${indexSizeNote(index)})`;
|
|
3869
|
+
const warning = indexBudgetWarning(index);
|
|
3870
|
+
const rendered = enforceMemoryIndexHardBudget(index);
|
|
3871
|
+
return `${instructions}\n\n${warning ? `${header}\n\n${warning}\n\n${rendered}` : `${header}\n\n${rendered}`}`;
|
|
2487
3872
|
}
|
|
2488
3873
|
const memoryExtension = (pi) => {
|
|
2489
3874
|
let cachedBlock = null;
|
|
2490
3875
|
pi.on("session_start", async (_event, ctx) => {
|
|
2491
3876
|
try {
|
|
2492
|
-
const
|
|
2493
|
-
|
|
2494
|
-
|
|
2495
|
-
|
|
3877
|
+
const migrated = await migrateMemory({ cwd: ctx.cwd });
|
|
3878
|
+
if (migrated !== null) log$5.info({
|
|
3879
|
+
event: "memory_migrated",
|
|
3880
|
+
from: migrated.from,
|
|
3881
|
+
to: migrated.to
|
|
3882
|
+
}, "migrated memory layout to current version");
|
|
3883
|
+
} catch (err) {
|
|
3884
|
+
log$5.warn({
|
|
3885
|
+
err,
|
|
3886
|
+
event: "memory_migration_failed"
|
|
3887
|
+
}, "memory migration failed; continuing with existing index");
|
|
3888
|
+
}
|
|
3889
|
+
try {
|
|
3890
|
+
const index = await readMemoryIndexFile({ cwd: ctx.cwd });
|
|
2496
3891
|
cachedBlock = composeBlock$1({
|
|
2497
3892
|
cwd: ctx.cwd,
|
|
2498
3893
|
index
|
|
2499
3894
|
});
|
|
2500
3895
|
} catch (err) {
|
|
2501
|
-
log$
|
|
3896
|
+
log$5.warn({
|
|
2502
3897
|
err,
|
|
2503
3898
|
event: "memory_index_failed"
|
|
2504
|
-
}, "memory index
|
|
3899
|
+
}, "memory index read failed; injecting instructions only");
|
|
2505
3900
|
cachedBlock = memoryInstructions(ctx.cwd);
|
|
2506
3901
|
}
|
|
2507
3902
|
});
|
|
@@ -2512,7 +3907,7 @@ const memoryExtension = (pi) => {
|
|
|
2512
3907
|
};
|
|
2513
3908
|
//#endregion
|
|
2514
3909
|
//#region src/extensions/platform-memory.ts
|
|
2515
|
-
const log$
|
|
3910
|
+
const log$4 = logger.child({ module: "platform-memory-extension" });
|
|
2516
3911
|
/**
|
|
2517
3912
|
* Resolve the human on this turn via the API, keyed by the message id.
|
|
2518
3913
|
* `/sandbox/channel-context` only returns a sender for a platform-known
|
|
@@ -2525,13 +3920,13 @@ const log$5 = logger.child({ module: "platform-memory-extension" });
|
|
|
2525
3920
|
async function resolveTurnUser(messageId) {
|
|
2526
3921
|
const client = sandboxClient();
|
|
2527
3922
|
if (!client) {
|
|
2528
|
-
log$
|
|
3923
|
+
log$4.debug({ event: "resolve_turn_user_no_api_url" }, "no API url in env; withholding user memory");
|
|
2529
3924
|
return null;
|
|
2530
3925
|
}
|
|
2531
3926
|
try {
|
|
2532
3927
|
const res = await client["channel-context"].$get({ query: { messageId } });
|
|
2533
3928
|
if (!res.ok) {
|
|
2534
|
-
log$
|
|
3929
|
+
log$4.warn({
|
|
2535
3930
|
event: "resolve_turn_user_failed",
|
|
2536
3931
|
status: res.status
|
|
2537
3932
|
}, "channel-context returned non-ok; withholding user memory");
|
|
@@ -2544,7 +3939,7 @@ async function resolveTurnUser(messageId) {
|
|
|
2544
3939
|
displayName: sender.displayName
|
|
2545
3940
|
};
|
|
2546
3941
|
} catch (err) {
|
|
2547
|
-
log$
|
|
3942
|
+
log$4.warn({
|
|
2548
3943
|
err,
|
|
2549
3944
|
event: "resolve_turn_user_failed"
|
|
2550
3945
|
}, "failed to resolve current user; withholding user memory");
|
|
@@ -2560,9 +3955,12 @@ function slugifyName(name) {
|
|
|
2560
3955
|
function composeBlock({ index, user }) {
|
|
2561
3956
|
const instructions = `## Current user memory
|
|
2562
3957
|
|
|
2563
|
-
Notes about the person on this turn — the only \`users/\` memory you can see. Store anything you learn about them under \`${`.memory/users/${user.id}-${slugifyName(user.displayName)}/`}<topic>.md\`, using exactly this directory. Other people's \`users/\` notes are never shown, so never address someone by a name you only find in memory.`;
|
|
3958
|
+
Notes about the person on this turn — the only \`users/\` memory you can see. Store anything you learn about them under \`${`.memory/users/${user.id}-${slugifyName(user.displayName)}/`}<topic>.md\`, using exactly this directory. Other people's \`users/\` notes are never shown, so never address someone by a name you only find in memory. Keep it lean and be selective — this is re-injected every turn; consolidate related facts into one file and delete what's stale rather than piling on near-duplicates.`;
|
|
2564
3959
|
if (!index || index.length === 0) return instructions;
|
|
2565
|
-
|
|
3960
|
+
const warning = indexBudgetWarning(index);
|
|
3961
|
+
const header = `Memory index (${indexSizeNote(index)}):`;
|
|
3962
|
+
const rendered = enforceMemoryIndexHardBudget(index);
|
|
3963
|
+
return `${instructions}\n\n${warning ? `${warning}\n\n${header}\n\n${rendered}` : `${header}\n\n${rendered}`}`;
|
|
2566
3964
|
}
|
|
2567
3965
|
/**
|
|
2568
3966
|
* Build the platform memory extension. `channelContext` is the per-turn ref
|
|
@@ -2583,15 +3981,12 @@ function createPlatformMemoryExtension({ channelContext }) {
|
|
|
2583
3981
|
cachedBlock = composeBlock({
|
|
2584
3982
|
index: await buildMemoryIndex({
|
|
2585
3983
|
cwd: ctx.cwd,
|
|
2586
|
-
|
|
2587
|
-
kind: "user",
|
|
2588
|
-
userId: user.id
|
|
2589
|
-
}
|
|
3984
|
+
userId: user.id
|
|
2590
3985
|
}),
|
|
2591
3986
|
user
|
|
2592
3987
|
});
|
|
2593
3988
|
} catch (err) {
|
|
2594
|
-
log$
|
|
3989
|
+
log$4.warn({
|
|
2595
3990
|
err,
|
|
2596
3991
|
event: "user_memory_index_failed"
|
|
2597
3992
|
}, "user memory index build failed; skipping injection");
|
|
@@ -2606,7 +4001,7 @@ function createPlatformMemoryExtension({ channelContext }) {
|
|
|
2606
4001
|
}
|
|
2607
4002
|
//#endregion
|
|
2608
4003
|
//#region src/extensions/self-trace.ts
|
|
2609
|
-
const log$
|
|
4004
|
+
const log$3 = logger.child({ module: "self-trace-extension" });
|
|
2610
4005
|
/**
|
|
2611
4006
|
* Reports the agent's own execution as OpenTelemetry spans:
|
|
2612
4007
|
* agent.session → agent.run → agent.turn.N → tool.NAME, with token/cost
|
|
@@ -2631,7 +4026,7 @@ const selfTraceExtension = (pi) => {
|
|
|
2631
4026
|
sessionSpan = tracer.startSpan("agent.session", { attributes: { "agent.model": modelId } }, remoteCtx);
|
|
2632
4027
|
sessionCtx = trace.setSpan(remoteCtx, sessionSpan);
|
|
2633
4028
|
const sc = sessionSpan.spanContext();
|
|
2634
|
-
log$
|
|
4029
|
+
log$3.info({
|
|
2635
4030
|
event: "self_trace_session_start",
|
|
2636
4031
|
trace_id: sc.traceId,
|
|
2637
4032
|
span_id: sc.spanId,
|
|
@@ -2743,13 +4138,13 @@ const selfTraceExtension = (pi) => {
|
|
|
2743
4138
|
* Lives in the harness package — soul.md is content from the agent's
|
|
2744
4139
|
* own git repo, not from the platform — so its handling stays here.
|
|
2745
4140
|
*/
|
|
2746
|
-
const log$
|
|
4141
|
+
const log$2 = logger.child({ module: "soul-extension" });
|
|
2747
4142
|
async function readSoul(cwd) {
|
|
2748
4143
|
try {
|
|
2749
4144
|
return (await readFile(join(cwd, "soul.md"), "utf8")).trim() || null;
|
|
2750
4145
|
} catch (err) {
|
|
2751
4146
|
if (err?.code === "ENOENT") return null;
|
|
2752
|
-
log$
|
|
4147
|
+
log$2.warn({
|
|
2753
4148
|
err,
|
|
2754
4149
|
event: "soul_read_failed"
|
|
2755
4150
|
}, "soul.md read failed");
|
|
@@ -2781,8 +4176,7 @@ const soulExtension = (pi) => {
|
|
|
2781
4176
|
};
|
|
2782
4177
|
//#endregion
|
|
2783
4178
|
//#region src/extensions/subagent/index.ts
|
|
2784
|
-
const log$
|
|
2785
|
-
const COLD_START_FLAG_WAIT_MS = 750;
|
|
4179
|
+
const log$1 = logger.child({ module: "subagent-ext" });
|
|
2786
4180
|
const MAX_TASKS = 8;
|
|
2787
4181
|
const TaskItem = Type.Object({
|
|
2788
4182
|
task: Type.String({ description: "The task to delegate to a subagent run." }),
|
|
@@ -2791,8 +4185,14 @@ const TaskItem = Type.Object({
|
|
|
2791
4185
|
maxLength: 120
|
|
2792
4186
|
}),
|
|
2793
4187
|
persona: Type.Optional(Type.String({ description: "Optional extra system prompt / role for this task, applied ON TOP of the child run's own default persona (your full identity and soul are still there underneath). Omit to run with just your default persona." })),
|
|
2794
|
-
model: Type.Optional(Type.String({ description: "Optional model id to run this subagent on. PREFER A LOWER-COST, FASTER MODEL when the task is well-scoped and does not need your full reasoning depth — most delegated subtasks (searching, summarizing, mechanical edits, gathering or reformatting data, running a check) run just as well on a lighter model and cost far less. Reserve a top-tier model for subtasks that genuinely need deep reasoning or careful judgment. Must be a real catalogued model id. Omit to inherit your own model. If you are locked to a Google-compliant model, only compliant models are accepted." }))
|
|
4188
|
+
model: Type.Optional(Type.String({ description: "Optional model id to run this subagent on. PREFER A LOWER-COST, FASTER MODEL when the task is well-scoped and does not need your full reasoning depth — most delegated subtasks (searching, summarizing, mechanical edits, gathering or reformatting data, running a check) run just as well on a lighter model and cost far less. Reserve a top-tier model for subtasks that genuinely need deep reasoning or careful judgment. Must be a real catalogued model id. Omit to inherit your own model. If you are locked to a Google-compliant model, only compliant models are accepted." })),
|
|
4189
|
+
timeoutMinutes: Type.Optional(Type.Integer({
|
|
4190
|
+
description: "Optional wall-clock timeout for this subagent, in minutes. If the run is still going after this long it is ended and you are rewoken with a timeout result, so a hung subagent can never strand you. Omit for the default (30 minutes). Raise it for genuinely long work (a big migration, a large audit); lower it for a quick lookup. Range 1-360.",
|
|
4191
|
+
minimum: 1,
|
|
4192
|
+
maximum: 360
|
|
4193
|
+
}))
|
|
2795
4194
|
});
|
|
4195
|
+
const DEFAULT_SUBAGENT_TIMEOUT_MS = 30 * 6e4;
|
|
2796
4196
|
const SubagentParams = Type.Object({ tasks: Type.Array(TaskItem, {
|
|
2797
4197
|
description: "One or more tasks to delegate. Each spawns an isolated subagent run linked to this conversation; they run in parallel and each rewakes you with its result when it finishes.",
|
|
2798
4198
|
minItems: 1,
|
|
@@ -2807,7 +4207,9 @@ function buildTool(messageId) {
|
|
|
2807
4207
|
"Delegate one or more tasks to subagent runs — fresh isolated copies of yourself, each with its own context window, linked to this conversation.",
|
|
2808
4208
|
"Use it to parallelize independent work, to keep a large or noisy subtask out of your own context, or to run a task under a specialized persona.",
|
|
2809
4209
|
"Fire-and-forget: this returns immediately after queueing. It does NOT wait for results. Each subagent runs on its own and, when it finishes, sends you its result on this thread — so queue the work, then keep going or end your turn. To chain, re-delegate after a result lands.",
|
|
2810
|
-
"Pass tasks: [{ task, title, persona?, model? }]. title is a short 3-6 word name for the task — it is shown to the person in the chat as that subagent's row, so name the work rather than restating the prompt. persona is an optional extra system prompt layered ON TOP of your default persona for that task (it adds to, it does not replace, your identity); omit it to run with just your default persona. model is an optional model id for that task — prefer a lower-cost, faster model for well-scoped subtasks that don't need deep reasoning, and reserve a top-tier model for the ones that do; omit it to inherit your own model."
|
|
4210
|
+
"Pass tasks: [{ task, title, persona?, model? }]. title is a short 3-6 word name for the task — it is shown to the person in the chat as that subagent's row, so name the work rather than restating the prompt. persona is an optional extra system prompt layered ON TOP of your default persona for that task (it adds to, it does not replace, your identity); omit it to run with just your default persona. model is an optional model id for that task — prefer a lower-cost, faster model for well-scoped subtasks that don't need deep reasoning, and reserve a top-tier model for the ones that do; omit it to inherit your own model.",
|
|
4211
|
+
"Peering: each queued task comes back with its own conversation id. A subagent is a real linked conversation, so to see what one is doing RIGHT NOW while it runs — its reasoning, the tools it has called and their results, its progress — read that conversation with `platform conversations show <conversationId>` (you are already authorized; it is your own delegated run). Check in that way instead of waiting blind for the final result. The read reflects the child's persisted state, which lags a few seconds behind live (tool results land as they complete; in-progress reasoning can be up to ~5s stale), so peek between checkpoints rather than polling in a tight loop.",
|
|
4212
|
+
"Steering: to add context, correct course, or answer a question a subagent needs mid-run, post to its conversation with `platform conversations post <conversationId> --message \"...\"`. If the subagent is still running, your message lands as a live steer picked up in that same turn; if it has gone idle, it queues as its next turn. This is the same primitive as any conversation message — there is no separate steer channel."
|
|
2811
4213
|
].join(" "),
|
|
2812
4214
|
promptSnippet: "subagent — delegate tasks to isolated subagent runs; each rewakes you with its result when done",
|
|
2813
4215
|
parameters: SubagentParams,
|
|
@@ -2825,28 +4227,39 @@ function buildTool(messageId) {
|
|
|
2825
4227
|
task: t.task,
|
|
2826
4228
|
title: t.title ?? null,
|
|
2827
4229
|
persona: t.persona ?? null,
|
|
2828
|
-
model: t.model ?? null
|
|
4230
|
+
model: t.model ?? null,
|
|
4231
|
+
timeoutMs: t.timeoutMinutes != null ? t.timeoutMinutes * 6e4 : DEFAULT_SUBAGENT_TIMEOUT_MS
|
|
2829
4232
|
}));
|
|
2830
4233
|
try {
|
|
2831
|
-
const
|
|
4234
|
+
const spawned = await postSubagentSpawn({
|
|
2832
4235
|
messageId,
|
|
2833
4236
|
tasks: spawnTasks
|
|
2834
4237
|
});
|
|
2835
|
-
|
|
4238
|
+
const { taskIds } = spawned;
|
|
4239
|
+
log$1.info({
|
|
2836
4240
|
event: "subagent_spawned",
|
|
2837
4241
|
count: taskIds.length
|
|
2838
4242
|
}, "subagent tasks queued");
|
|
2839
|
-
const
|
|
4243
|
+
const convByTask = new Map(spawned.tasks.map((t) => [t.taskId, t.conversationId]));
|
|
4244
|
+
const lines = taskIds.map((id, i) => {
|
|
4245
|
+
const label = spawnTasks[i]?.title ?? spawnTasks[i]?.task ?? "";
|
|
4246
|
+
const conv = convByTask.get(id);
|
|
4247
|
+
return `- ${id}: ${label}${conv ? ` — conversation ${conv}` : ""}`;
|
|
4248
|
+
}).join("\n");
|
|
4249
|
+
const peerHint = spawned.tasks.length ? "\nEach subagent runs on its own conversation (id shown per task above). To SEE what one is doing while it runs, read it with `platform conversations show <conversationId>`. To STEER one mid-run — add context, correct course, answer a question — post to its conversation with `platform conversations post <conversationId> --message \"...\"`; it lands as a live steer if the subagent is still running, or as its next turn if it has gone idle." : "";
|
|
2840
4250
|
return {
|
|
2841
4251
|
content: [{
|
|
2842
4252
|
type: "text",
|
|
2843
|
-
text: `Queued ${taskIds.length} subagent ${taskIds.length === 1 ? "run" : "runs"}. Each runs on its own and will send you its result on this thread when it finishes — keep working or end your turn meanwhile.\n${lines}`
|
|
4253
|
+
text: `Queued ${taskIds.length} subagent ${taskIds.length === 1 ? "run" : "runs"}. Each runs on its own and will send you its result on this thread when it finishes — keep working or end your turn meanwhile.\n${lines}${peerHint}`
|
|
2844
4254
|
}],
|
|
2845
|
-
details: {
|
|
4255
|
+
details: {
|
|
4256
|
+
taskIds,
|
|
4257
|
+
tasks: spawned.tasks
|
|
4258
|
+
}
|
|
2846
4259
|
};
|
|
2847
4260
|
} catch (err) {
|
|
2848
4261
|
const message = err instanceof Error ? err.message : String(err);
|
|
2849
|
-
log$
|
|
4262
|
+
log$1.warn({
|
|
2850
4263
|
err,
|
|
2851
4264
|
event: "subagent_spawn_failed"
|
|
2852
4265
|
}, "subagent spawn failed");
|
|
@@ -2863,29 +4276,24 @@ function buildTool(messageId) {
|
|
|
2863
4276
|
};
|
|
2864
4277
|
}
|
|
2865
4278
|
/**
|
|
2866
|
-
* Gated on `harness-subagent-enabled`, read from the shared feature-flag poll.
|
|
2867
4279
|
* The factory takes the session's channel context to resolve the originating
|
|
2868
4280
|
* messageId — the api links each spawned run to the conversation that message
|
|
2869
4281
|
* belongs to and rewakes it on completion (nothing about the parent is piped
|
|
2870
|
-
* from the sandbox beyond that id).
|
|
4282
|
+
* from the sandbox beyond that id). The tool is registered unconditionally at
|
|
4283
|
+
* session_start.
|
|
2871
4284
|
*/
|
|
2872
4285
|
function createSubagentExtension({ channelContext }) {
|
|
2873
4286
|
return (pi) => {
|
|
2874
4287
|
const messageId = extractMessageId(channelContext);
|
|
2875
|
-
startFeatureFlagPoller();
|
|
2876
4288
|
let registered = false;
|
|
2877
4289
|
const registerOnce = () => {
|
|
2878
4290
|
if (registered) return;
|
|
2879
4291
|
registered = true;
|
|
2880
4292
|
pi.registerTool(buildTool(messageId));
|
|
2881
|
-
log$
|
|
4293
|
+
log$1.info({ event: "subagent_enabled" }, "subagent tool registered");
|
|
2882
4294
|
};
|
|
2883
|
-
|
|
2884
|
-
|
|
2885
|
-
});
|
|
2886
|
-
pi.on("session_start", async () => {
|
|
2887
|
-
if (getPolledFlag("subagent") === null) await Promise.race([awaitFirstFlagPoll(), new Promise((resolve) => setTimeout(resolve, COLD_START_FLAG_WAIT_MS).unref?.())]);
|
|
2888
|
-
if (getPolledFlag("subagent") === true) registerOnce();
|
|
4295
|
+
pi.on("session_start", () => {
|
|
4296
|
+
registerOnce();
|
|
2889
4297
|
});
|
|
2890
4298
|
};
|
|
2891
4299
|
}
|
|
@@ -2916,214 +4324,6 @@ const toolCallEnvExtension = (pi) => {
|
|
|
2916
4324
|
});
|
|
2917
4325
|
};
|
|
2918
4326
|
//#endregion
|
|
2919
|
-
//#region src/extensions/tool-call-summary.ts
|
|
2920
|
-
const log$1 = logger.child({ module: "tool-call-summary-extension" });
|
|
2921
|
-
/**
|
|
2922
|
-
* The injected parameter name: a namespaced sentinel, so it can never collide
|
|
2923
|
-
* with a real tool argument and is unmistakable in transcripts and logs. The
|
|
2924
|
-
* frontend renderer (ANY-2723) duplicates this literal — keep the two in sync.
|
|
2925
|
-
*/
|
|
2926
|
-
const TOOL_CALL_SUMMARY_FIELD = "__skydive_summary__";
|
|
2927
|
-
/** JSON Schema fragment for the injected parameter. */
|
|
2928
|
-
const SUMMARY_PROPERTY = {
|
|
2929
|
-
type: "string",
|
|
2930
|
-
description: "Required for every tool call. A concise, specific summary (max ~8 words) of what THIS call does and why, written for a person watching the conversation, e.g. \"Searching feedback for billing complaints\" or \"Reading the auth middleware\". Address the user directly in second person: the summary is read by the user, so refer to their things as \"your\", never in third person — \"Reading your emails\", not \"Reading his emails\". Always use the present progressive tense, since it is shown while the call runs: \"Updating your Slack\", never \"Updated your Slack\". Make each summary distinct from your other tool calls; never reuse a generic label like \"Search query\" or \"Running command\"."
|
|
2931
|
-
};
|
|
2932
|
-
const jsonSchemaObjectSchema = z.object({
|
|
2933
|
-
type: z.unknown().optional(),
|
|
2934
|
-
properties: z.record(z.string(), z.unknown()).optional(),
|
|
2935
|
-
required: z.array(z.string()).optional(),
|
|
2936
|
-
additionalProperties: z.unknown().optional()
|
|
2937
|
-
}).passthrough();
|
|
2938
|
-
const toolEntrySchema = z.object({
|
|
2939
|
-
name: z.string().optional(),
|
|
2940
|
-
input_schema: jsonSchemaObjectSchema.optional(),
|
|
2941
|
-
parameters: jsonSchemaObjectSchema.optional(),
|
|
2942
|
-
function: z.object({
|
|
2943
|
-
name: z.string().optional(),
|
|
2944
|
-
parameters: jsonSchemaObjectSchema.optional()
|
|
2945
|
-
}).passthrough().optional()
|
|
2946
|
-
}).passthrough();
|
|
2947
|
-
const payloadWithToolsSchema = z.object({ tools: z.array(z.unknown()) }).passthrough();
|
|
2948
|
-
/**
|
|
2949
|
-
* Add the summary property to one JSON Schema object. Returns the augmented
|
|
2950
|
-
* copy, or `null` when the tool should be left untouched: a strict schema
|
|
2951
|
-
* (`additionalProperties: false`) whose validation would reject the extra
|
|
2952
|
-
* field, or one that already declares a `__skydive_summary__` property of its own.
|
|
2953
|
-
*/
|
|
2954
|
-
function augmentSchema(schema) {
|
|
2955
|
-
if (schema.additionalProperties === false) return null;
|
|
2956
|
-
const properties = schema.properties ?? {};
|
|
2957
|
-
if ("__skydive_summary__" in properties) return null;
|
|
2958
|
-
const required = schema.required ?? [];
|
|
2959
|
-
return {
|
|
2960
|
-
...schema,
|
|
2961
|
-
type: schema.type ?? "object",
|
|
2962
|
-
properties: {
|
|
2963
|
-
[TOOL_CALL_SUMMARY_FIELD]: SUMMARY_PROPERTY,
|
|
2964
|
-
...properties
|
|
2965
|
-
},
|
|
2966
|
-
required: required.includes("__skydive_summary__") ? required : [...required, TOOL_CALL_SUMMARY_FIELD]
|
|
2967
|
-
};
|
|
2968
|
-
}
|
|
2969
|
-
/**
|
|
2970
|
-
* Augment a single tool entry, dispatching on which provider shape it is.
|
|
2971
|
-
* Returns the (possibly rebuilt) entry and whether anything changed. Skipped
|
|
2972
|
-
* tools — wrong shape, strict, or name in `strictToolNames` — return unchanged.
|
|
2973
|
-
*/
|
|
2974
|
-
function augmentToolEntry(entry, strictToolNames) {
|
|
2975
|
-
const parsed = toolEntrySchema.safeParse(entry);
|
|
2976
|
-
if (!parsed.success) return {
|
|
2977
|
-
entry,
|
|
2978
|
-
changed: false
|
|
2979
|
-
};
|
|
2980
|
-
const tool = parsed.data;
|
|
2981
|
-
const name = tool.name ?? tool.function?.name ?? null;
|
|
2982
|
-
if (name !== null && strictToolNames.has(name)) return {
|
|
2983
|
-
entry,
|
|
2984
|
-
changed: false
|
|
2985
|
-
};
|
|
2986
|
-
if (tool.input_schema) {
|
|
2987
|
-
const augmented = augmentSchema(tool.input_schema);
|
|
2988
|
-
if (!augmented) return {
|
|
2989
|
-
entry,
|
|
2990
|
-
changed: false
|
|
2991
|
-
};
|
|
2992
|
-
return {
|
|
2993
|
-
entry: {
|
|
2994
|
-
...tool,
|
|
2995
|
-
input_schema: augmented
|
|
2996
|
-
},
|
|
2997
|
-
changed: true
|
|
2998
|
-
};
|
|
2999
|
-
}
|
|
3000
|
-
if (tool.parameters) {
|
|
3001
|
-
const augmented = augmentSchema(tool.parameters);
|
|
3002
|
-
if (!augmented) return {
|
|
3003
|
-
entry,
|
|
3004
|
-
changed: false
|
|
3005
|
-
};
|
|
3006
|
-
return {
|
|
3007
|
-
entry: {
|
|
3008
|
-
...tool,
|
|
3009
|
-
parameters: augmented
|
|
3010
|
-
},
|
|
3011
|
-
changed: true
|
|
3012
|
-
};
|
|
3013
|
-
}
|
|
3014
|
-
if (tool.function?.parameters) {
|
|
3015
|
-
const augmented = augmentSchema(tool.function.parameters);
|
|
3016
|
-
if (!augmented) return {
|
|
3017
|
-
entry,
|
|
3018
|
-
changed: false
|
|
3019
|
-
};
|
|
3020
|
-
return {
|
|
3021
|
-
entry: {
|
|
3022
|
-
...tool,
|
|
3023
|
-
function: {
|
|
3024
|
-
...tool.function,
|
|
3025
|
-
parameters: augmented
|
|
3026
|
-
}
|
|
3027
|
-
},
|
|
3028
|
-
changed: true
|
|
3029
|
-
};
|
|
3030
|
-
}
|
|
3031
|
-
return {
|
|
3032
|
-
entry,
|
|
3033
|
-
changed: false
|
|
3034
|
-
};
|
|
3035
|
-
}
|
|
3036
|
-
/**
|
|
3037
|
-
* Inject the summary field into every eligible tool in a provider payload.
|
|
3038
|
-
* Returns a new payload when at least one tool was augmented, or `undefined`
|
|
3039
|
-
* to signal "no change" (which keeps the original payload, per the
|
|
3040
|
-
* `before_provider_request` contract).
|
|
3041
|
-
*
|
|
3042
|
-
* @param payload The outgoing provider payload (shape varies by provider).
|
|
3043
|
-
* @param strictToolNames Names of tools whose registered schema is strict and
|
|
3044
|
-
* must be skipped to avoid validation errors.
|
|
3045
|
-
*/
|
|
3046
|
-
function injectToolCallSummary(payload, strictToolNames) {
|
|
3047
|
-
const parsed = payloadWithToolsSchema.safeParse(payload);
|
|
3048
|
-
if (!parsed.success || parsed.data.tools.length === 0) return void 0;
|
|
3049
|
-
let changed = false;
|
|
3050
|
-
const tools = parsed.data.tools.map((entry) => {
|
|
3051
|
-
const result = augmentToolEntry(entry, strictToolNames);
|
|
3052
|
-
if (result.changed) changed = true;
|
|
3053
|
-
return result.entry;
|
|
3054
|
-
});
|
|
3055
|
-
if (!changed) return void 0;
|
|
3056
|
-
return {
|
|
3057
|
-
...parsed.data,
|
|
3058
|
-
tools
|
|
3059
|
-
};
|
|
3060
|
-
}
|
|
3061
|
-
/**
|
|
3062
|
-
* Names of registered tools whose schema sets `additionalProperties: false`.
|
|
3063
|
-
* Pi validates the model's tool args against this registered schema, so the
|
|
3064
|
-
* injected field would make a strict tool's call fail validation — skip them.
|
|
3065
|
-
*/
|
|
3066
|
-
function getStrictToolNames(pi) {
|
|
3067
|
-
const names = /* @__PURE__ */ new Set();
|
|
3068
|
-
for (const tool of pi.getAllTools()) {
|
|
3069
|
-
const parsed = jsonSchemaObjectSchema.safeParse(tool.parameters);
|
|
3070
|
-
if (parsed.success && parsed.data.additionalProperties === false) names.add(tool.name);
|
|
3071
|
-
}
|
|
3072
|
-
return names;
|
|
3073
|
-
}
|
|
3074
|
-
function toolDeclaresSummaryParam(pi, toolName) {
|
|
3075
|
-
const tool = pi.getAllTools().find((candidate) => candidate.name === toolName);
|
|
3076
|
-
if (!tool) return false;
|
|
3077
|
-
const parsed = jsonSchemaObjectSchema.safeParse(tool.parameters);
|
|
3078
|
-
return parsed.success && parsed.data.properties != null && "__skydive_summary__" in parsed.data.properties;
|
|
3079
|
-
}
|
|
3080
|
-
/**
|
|
3081
|
-
* Remove the injected summary from a tool's execution input. No-op when the
|
|
3082
|
-
* field is absent, or when the tool genuinely declares a `__skydive_summary__`
|
|
3083
|
-
* parameter of its own (which we never inject into, so its value is real).
|
|
3084
|
-
* Mutates `input` in place, matching the `tool_call` contract.
|
|
3085
|
-
*
|
|
3086
|
-
* Fails open: this runs on the critical path of tool execution, and the
|
|
3087
|
-
* `getAllTools()` lookup can throw. On any error we leave `input` untouched
|
|
3088
|
-
* (the sentinel may pass through to the tool, but a bug here can never break
|
|
3089
|
-
* tool execution).
|
|
3090
|
-
*/
|
|
3091
|
-
function stripInjectedSummary(pi, toolName, input) {
|
|
3092
|
-
try {
|
|
3093
|
-
if (!("__skydive_summary__" in input)) return;
|
|
3094
|
-
if (toolDeclaresSummaryParam(pi, toolName)) return;
|
|
3095
|
-
delete input[TOOL_CALL_SUMMARY_FIELD];
|
|
3096
|
-
} catch (err) {
|
|
3097
|
-
log$1.error({
|
|
3098
|
-
err,
|
|
3099
|
-
event: "tool_call_summary_strip_failed",
|
|
3100
|
-
toolName
|
|
3101
|
-
}, "tool_call_summary strip failed; leaving tool input untouched");
|
|
3102
|
-
}
|
|
3103
|
-
}
|
|
3104
|
-
/**
|
|
3105
|
-
* Compute the rewritten payload for a `before_provider_request` event, failing
|
|
3106
|
-
* open: on any error the original payload is left untouched so a bug here can
|
|
3107
|
-
* never break an LLM call.
|
|
3108
|
-
*/
|
|
3109
|
-
function buildInjectedPayload(pi, payload) {
|
|
3110
|
-
try {
|
|
3111
|
-
return injectToolCallSummary(payload, getStrictToolNames(pi));
|
|
3112
|
-
} catch (err) {
|
|
3113
|
-
log$1.error({
|
|
3114
|
-
err,
|
|
3115
|
-
event: "tool_call_summary_injection_failed"
|
|
3116
|
-
}, "tool_call_summary injection failed; passing payload through unchanged");
|
|
3117
|
-
return;
|
|
3118
|
-
}
|
|
3119
|
-
}
|
|
3120
|
-
const toolCallSummaryExtension = (pi) => {
|
|
3121
|
-
pi.on("before_provider_request", (event) => buildInjectedPayload(pi, event.payload));
|
|
3122
|
-
pi.on("tool_call", (event) => {
|
|
3123
|
-
stripInjectedSummary(pi, event.toolName, event.input);
|
|
3124
|
-
});
|
|
3125
|
-
};
|
|
3126
|
-
//#endregion
|
|
3127
4327
|
//#region src/extensions/background-tasks.ts
|
|
3128
4328
|
/**
|
|
3129
4329
|
* Background bash tasks as a pi extension.
|
|
@@ -3163,21 +4363,43 @@ const toolCallSummaryExtension = (pi) => {
|
|
|
3163
4363
|
* from the origin messageId in its channel context (`resolveConversationFromApi`)
|
|
3164
4364
|
* — and every run of the same conversation resolves to the same id, keeping the
|
|
3165
4365
|
* shared map correctly scoped across turns. `bg_*`, the completion wake, and the
|
|
3166
|
-
* next-session injection all filter
|
|
3167
|
-
*
|
|
4366
|
+
* next-session injection all filter by a **scope key** — the resolved
|
|
4367
|
+
* conversation id, or, when a session's conversation is unresolvable (a bare
|
|
4368
|
+
* CLI session, or a run whose channel-context ref carries no messageId), a
|
|
4369
|
+
* sentinel unique to that one session instance. Comparing on the raw
|
|
4370
|
+
* `conversationId` would bucket every unresolvable session together under
|
|
4371
|
+
* `null` and leak one's completion wake / status / next-session injection into
|
|
4372
|
+
* another; the sentinel keeps each isolated so an agent never sees or is woken
|
|
4373
|
+
* by a task from a different chat. Only the output log
|
|
3168
4374
|
* spills to disk (/home/user/.anyone/bg-tasks/<id>.log) to avoid buffering a chatty job
|
|
3169
4375
|
* in memory; exit code and run state live on the in-memory task.
|
|
3170
4376
|
*
|
|
3171
|
-
* **
|
|
3172
|
-
*
|
|
3173
|
-
* respawn,
|
|
3174
|
-
*
|
|
3175
|
-
*
|
|
3176
|
-
*
|
|
3177
|
-
*
|
|
3178
|
-
*
|
|
3179
|
-
*
|
|
3180
|
-
*
|
|
4377
|
+
* **Cross-restart survival is OPT-IN, via `bg_run({ resumable: true })`.**
|
|
4378
|
+
* A non-resumable task's state lives only in the running harness process: a
|
|
4379
|
+
* harness restart (crash → supervisord respawn, `platform harness reload`, or
|
|
4380
|
+
* a sandbox recycle on idle timeout / redeploy / template rebuild) drops the
|
|
4381
|
+
* map and pi's exec children are reaped with it, and the task is gone — the
|
|
4382
|
+
* right behavior for a one-shot side-effecting command, which must never
|
|
4383
|
+
* silently re-run.
|
|
4384
|
+
*
|
|
4385
|
+
* A RESUMABLE task additionally checkpoints its *spec* (command, cwd, labels,
|
|
4386
|
+
* origin messageId — not its live output/exit state) to a durable, api-side
|
|
4387
|
+
* journal keyed by conversation (`/sandbox/bg-task-journal`, redis). On the
|
|
4388
|
+
* NEXT run's `session_start` — which usually lands on a *different*,
|
|
4389
|
+
* cold-provisioned sandbox, which is exactly why the journal is api-side and
|
|
4390
|
+
* not on the sandbox disk — the harness lists the journal and relaunches any
|
|
4391
|
+
* spec it isn't already running, keeping the original id and prepending a
|
|
4392
|
+
* `<resumed-after-restart>` banner so the agent knows it re-ran from scratch,
|
|
4393
|
+
* not continued. The spec is dropped from the journal when the task
|
|
4394
|
+
* finishes/kills. Because relaunch RE-EXECUTES the command, resumable is only
|
|
4395
|
+
* for idempotent, long-lived work (pollers, watchers, retry loops); the tool
|
|
4396
|
+
* description enforces this and the default is false.
|
|
4397
|
+
*
|
|
4398
|
+
* The idle-completion wake (a task finishing while the agent is idle) crosses
|
|
4399
|
+
* the sandbox → platform boundary via a fresh run (`bg-task-done`) for both
|
|
4400
|
+
* resumable and non-resumable tasks; the journal is a separate, additive layer
|
|
4401
|
+
* that only handles a task whose harness dies BEFORE it finishes. This stays
|
|
4402
|
+
* distinct from the scheduled-run (cron) system.
|
|
3181
4403
|
*/
|
|
3182
4404
|
const log = logger.child({ module: "background-tasks-ext" });
|
|
3183
4405
|
const ops = createLocalBashOperations();
|
|
@@ -3188,6 +4410,7 @@ const WATCHDOG_INTERVAL_MS = 3e4;
|
|
|
3188
4410
|
const KEEPALIVE_EVERY_MS = 6e4;
|
|
3189
4411
|
const KEEPALIVE_MAX_MS = 3600 * 1e3;
|
|
3190
4412
|
const STALL_HINT_AFTER_MS = 120 * 1e3;
|
|
4413
|
+
const PUBLISH_DEBOUNCE_MS = 300;
|
|
3191
4414
|
const MAX_LOG_BYTES = 100 * 1024 * 1024;
|
|
3192
4415
|
const DEFAULT_TAIL_LINES = 30;
|
|
3193
4416
|
const TAIL_READ_BYTES = 64 * 1024;
|
|
@@ -3195,11 +4418,12 @@ function taskLabel(meta) {
|
|
|
3195
4418
|
return `${meta.id} "${meta.description ?? meta.command.slice(0, 60)}"`;
|
|
3196
4419
|
}
|
|
3197
4420
|
let taskCounter = 0;
|
|
4421
|
+
let sessionScopeCounter = 0;
|
|
3198
4422
|
const tasks = /* @__PURE__ */ new Map();
|
|
3199
4423
|
let watchdogInterval = null;
|
|
3200
4424
|
let lastKeepaliveAt = 0;
|
|
3201
|
-
function
|
|
3202
|
-
return meta.
|
|
4425
|
+
function sameScope(meta, scopeKey) {
|
|
4426
|
+
return meta.scopeKey === scopeKey;
|
|
3203
4427
|
}
|
|
3204
4428
|
function logPath(id) {
|
|
3205
4429
|
return join(tasksDir(), `${id}.log`);
|
|
@@ -3312,6 +4536,10 @@ function createBackgroundTasksExtension({ channelContext }) {
|
|
|
3312
4536
|
return id;
|
|
3313
4537
|
});
|
|
3314
4538
|
}
|
|
4539
|
+
const unresolvedScopeSentinel = `unresolved:${process.pid.toString(36)}:${(sessionScopeCounter += 1).toString(36)}`;
|
|
4540
|
+
function scopeKey() {
|
|
4541
|
+
return conversationId ?? unresolvedScopeSentinel;
|
|
4542
|
+
}
|
|
3315
4543
|
let agentActive = false;
|
|
3316
4544
|
pi.on("agent_start", async () => {
|
|
3317
4545
|
agentActive = true;
|
|
@@ -3339,13 +4567,16 @@ ${recentOutput}
|
|
|
3339
4567
|
</background-task-finished>
|
|
3340
4568
|
Run bg_logs for the full output.
|
|
3341
4569
|
|
|
3342
|
-
This is a background-task completion, not a message from the user.
|
|
4570
|
+
This is a background-task completion, not a message from the user.
|
|
4571
|
+
For a routine or expected completion, output nothing.
|
|
4572
|
+
Only reply if the outcome changes what the user should know or do,
|
|
4573
|
+
or if you were explicitly waiting to report it.`,
|
|
3343
4574
|
display: false
|
|
3344
4575
|
};
|
|
3345
4576
|
}
|
|
3346
4577
|
async function notifyCompletion(meta) {
|
|
3347
4578
|
if (meta.notified) return;
|
|
3348
|
-
if (agentActive &&
|
|
4579
|
+
if (agentActive && sameScope(meta, scopeKey())) {
|
|
3349
4580
|
meta.notified = true;
|
|
3350
4581
|
pi.sendMessage(await taskDoneMessage(meta), {
|
|
3351
4582
|
triggerTurn: true,
|
|
@@ -3376,9 +4607,83 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3376
4607
|
});
|
|
3377
4608
|
} else log.info({ taskId: meta.id }, "bg task completed idle with no origin message; deferring to next session_start");
|
|
3378
4609
|
}
|
|
3379
|
-
|
|
4610
|
+
const SNAPSHOT_TAIL_LINES = 40;
|
|
4611
|
+
let lastPublishedSignature = null;
|
|
4612
|
+
let publishInFlight = null;
|
|
4613
|
+
let publishQueued = false;
|
|
4614
|
+
function snapshotSignature(snapshot) {
|
|
4615
|
+
return JSON.stringify(snapshot.map((t) => ({
|
|
4616
|
+
id: t.id,
|
|
4617
|
+
state: t.state,
|
|
4618
|
+
exitCode: t.exitCode,
|
|
4619
|
+
killedReason: t.killedReason,
|
|
4620
|
+
outputTail: t.outputTail
|
|
4621
|
+
})));
|
|
4622
|
+
}
|
|
4623
|
+
async function doPublishSnapshot() {
|
|
4624
|
+
if (!messageId) return;
|
|
4625
|
+
const mine = [...tasks.values()].filter((t) => sameScope(t, scopeKey()));
|
|
4626
|
+
try {
|
|
4627
|
+
const snapshot = await Promise.all(mine.map(async (t) => ({
|
|
4628
|
+
id: t.id,
|
|
4629
|
+
command: t.command,
|
|
4630
|
+
description: t.description,
|
|
4631
|
+
startedAt: t.startedAt,
|
|
4632
|
+
state: t.running ? "running" : "finished",
|
|
4633
|
+
exitCode: t.exitCode,
|
|
4634
|
+
killedReason: t.killedReason,
|
|
4635
|
+
outputTail: await tailLog(t.id, SNAPSHOT_TAIL_LINES)
|
|
4636
|
+
})));
|
|
4637
|
+
const signature = snapshotSignature(snapshot);
|
|
4638
|
+
if (signature === lastPublishedSignature) return;
|
|
4639
|
+
lastPublishedSignature = signature;
|
|
4640
|
+
postBackgroundTasksSnapshot({
|
|
4641
|
+
messageId,
|
|
4642
|
+
tasks: snapshot
|
|
4643
|
+
});
|
|
4644
|
+
} catch (err) {
|
|
4645
|
+
log.debug({
|
|
4646
|
+
err,
|
|
4647
|
+
event: "bg_tasks_snapshot_build_failed"
|
|
4648
|
+
}, "building bg-tasks snapshot failed");
|
|
4649
|
+
}
|
|
4650
|
+
}
|
|
4651
|
+
async function publishSnapshotNow() {
|
|
4652
|
+
if (publishInFlight) {
|
|
4653
|
+
publishQueued = true;
|
|
4654
|
+
return;
|
|
4655
|
+
}
|
|
4656
|
+
publishInFlight = (async () => {
|
|
4657
|
+
try {
|
|
4658
|
+
do {
|
|
4659
|
+
publishQueued = false;
|
|
4660
|
+
await doPublishSnapshot();
|
|
4661
|
+
} while (publishQueued);
|
|
4662
|
+
} finally {
|
|
4663
|
+
publishInFlight = null;
|
|
4664
|
+
}
|
|
4665
|
+
})();
|
|
4666
|
+
await publishInFlight;
|
|
4667
|
+
}
|
|
4668
|
+
let publishTimer = null;
|
|
4669
|
+
function schedulePublishSnapshot() {
|
|
4670
|
+
if (publishTimer) return;
|
|
4671
|
+
publishTimer = setTimeout(() => {
|
|
4672
|
+
publishTimer = null;
|
|
4673
|
+
publishSnapshotNow();
|
|
4674
|
+
}, PUBLISH_DEBOUNCE_MS);
|
|
4675
|
+
publishTimer.unref?.();
|
|
4676
|
+
}
|
|
4677
|
+
async function flushPublishSnapshot() {
|
|
4678
|
+
if (publishTimer) {
|
|
4679
|
+
clearTimeout(publishTimer);
|
|
4680
|
+
publishTimer = null;
|
|
4681
|
+
}
|
|
4682
|
+
await publishSnapshotNow();
|
|
4683
|
+
}
|
|
4684
|
+
async function launchTask({ command, description, cwd, resumable = false, resumedFromJournal = false, id: providedId, startedAt: providedStartedAt }) {
|
|
3380
4685
|
taskCounter += 1;
|
|
3381
|
-
const id = `bg-${process.pid.toString(36)}-${taskCounter}`;
|
|
4686
|
+
const id = providedId ?? `bg-${process.pid.toString(36)}-${taskCounter}`;
|
|
3382
4687
|
try {
|
|
3383
4688
|
await mkdir(tasksDir(), { recursive: true });
|
|
3384
4689
|
} catch (err) {
|
|
@@ -3394,7 +4699,7 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3394
4699
|
taskId: id
|
|
3395
4700
|
}, "bg task log write failed");
|
|
3396
4701
|
});
|
|
3397
|
-
const startedAt = Date.now();
|
|
4702
|
+
const startedAt = providedStartedAt ?? Date.now();
|
|
3398
4703
|
const meta = {
|
|
3399
4704
|
id,
|
|
3400
4705
|
command: stripPlatformExportsForDisplay(command),
|
|
@@ -3402,6 +4707,7 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3402
4707
|
logBytes: 0,
|
|
3403
4708
|
lastOutputAt: startedAt,
|
|
3404
4709
|
conversationId,
|
|
4710
|
+
scopeKey: scopeKey(),
|
|
3405
4711
|
messageId,
|
|
3406
4712
|
description,
|
|
3407
4713
|
notified: false,
|
|
@@ -3409,9 +4715,23 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3409
4715
|
controller: new AbortController(),
|
|
3410
4716
|
running: true,
|
|
3411
4717
|
exitCode: null,
|
|
3412
|
-
error: null
|
|
4718
|
+
error: null,
|
|
4719
|
+
resumable,
|
|
4720
|
+
cwd,
|
|
4721
|
+
resumedFromJournal
|
|
3413
4722
|
};
|
|
3414
4723
|
tasks.set(id, meta);
|
|
4724
|
+
if (resumable && messageId) putBackgroundTaskJournalSpec({
|
|
4725
|
+
messageId,
|
|
4726
|
+
spec: {
|
|
4727
|
+
id,
|
|
4728
|
+
command,
|
|
4729
|
+
cwd,
|
|
4730
|
+
description,
|
|
4731
|
+
startedAt,
|
|
4732
|
+
messageId
|
|
4733
|
+
}
|
|
4734
|
+
});
|
|
3415
4735
|
ops.exec(command, cwd, {
|
|
3416
4736
|
onData: (chunk) => {
|
|
3417
4737
|
meta.logBytes += chunk.length;
|
|
@@ -3437,6 +4757,11 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3437
4757
|
taskId: id,
|
|
3438
4758
|
exitCode: meta.exitCode
|
|
3439
4759
|
}, "bg task finished");
|
|
4760
|
+
if (meta.resumable && meta.messageId) deleteBackgroundTaskJournalSpec({
|
|
4761
|
+
messageId: meta.messageId,
|
|
4762
|
+
taskId: id
|
|
4763
|
+
});
|
|
4764
|
+
schedulePublishSnapshot();
|
|
3440
4765
|
await notifyCompletion(meta);
|
|
3441
4766
|
});
|
|
3442
4767
|
ensureWatchdog();
|
|
@@ -3444,19 +4769,55 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3444
4769
|
taskId: id,
|
|
3445
4770
|
conversationId
|
|
3446
4771
|
}, "bg task started");
|
|
4772
|
+
schedulePublishSnapshot();
|
|
3447
4773
|
return meta;
|
|
3448
4774
|
}
|
|
3449
4775
|
function knownTaskIds() {
|
|
3450
|
-
return [...tasks.values()].filter((t) =>
|
|
4776
|
+
return [...tasks.values()].filter((t) => sameScope(t, scopeKey())).map((t) => t.id).join(", ") || "(none)";
|
|
4777
|
+
}
|
|
4778
|
+
let journalChecked = false;
|
|
4779
|
+
async function rehydrateJournaledTasks() {
|
|
4780
|
+
if (!messageId || journalChecked) return;
|
|
4781
|
+
journalChecked = true;
|
|
4782
|
+
const specs = await listBackgroundTaskJournalSpecs({ messageId });
|
|
4783
|
+
if (specs.length === 0) return;
|
|
4784
|
+
for (const spec of specs) {
|
|
4785
|
+
const live = tasks.get(spec.id);
|
|
4786
|
+
if (live && sameScope(live, scopeKey())) continue;
|
|
4787
|
+
log.info({
|
|
4788
|
+
taskId: spec.id,
|
|
4789
|
+
conversationId
|
|
4790
|
+
}, "relaunching journaled resumable bg task after restart");
|
|
4791
|
+
const meta = await launchTask({
|
|
4792
|
+
command: spec.command,
|
|
4793
|
+
description: spec.description,
|
|
4794
|
+
cwd: spec.cwd,
|
|
4795
|
+
resumable: true,
|
|
4796
|
+
resumedFromJournal: true,
|
|
4797
|
+
id: spec.id,
|
|
4798
|
+
startedAt: spec.startedAt
|
|
4799
|
+
});
|
|
4800
|
+
try {
|
|
4801
|
+
const stream = createWriteStream(logPath(meta.id), { flags: "a" });
|
|
4802
|
+
stream.write(`<resumed-after-restart>requeued and relaunched on a new sandbox after the previous one was drained; this is a fresh execution of the command from the start, not a continuation</resumed-after-restart>\n`);
|
|
4803
|
+
stream.end();
|
|
4804
|
+
} catch (err) {
|
|
4805
|
+
log.debug({
|
|
4806
|
+
err,
|
|
4807
|
+
taskId: meta.id
|
|
4808
|
+
}, "resume banner write failed");
|
|
4809
|
+
}
|
|
4810
|
+
}
|
|
3451
4811
|
}
|
|
3452
4812
|
pi.on("session_start", async () => {
|
|
3453
|
-
if (tasks.size === 0) return;
|
|
4813
|
+
if (tasks.size === 0 && (!messageId || journalChecked)) return;
|
|
3454
4814
|
await ensureConversationId();
|
|
3455
|
-
|
|
4815
|
+
await rehydrateJournaledTasks();
|
|
4816
|
+
for (const [id, meta] of tasks) if (sameScope(meta, scopeKey()) && !meta.running && meta.notified) {
|
|
3456
4817
|
tasks.delete(id);
|
|
3457
4818
|
await unlink(logPath(id)).catch(() => {});
|
|
3458
4819
|
}
|
|
3459
|
-
const unnotified = [...tasks.values()].filter((t) =>
|
|
4820
|
+
const unnotified = [...tasks.values()].filter((t) => sameScope(t, scopeKey()) && !t.notified && !t.running);
|
|
3460
4821
|
for (const meta of unnotified) {
|
|
3461
4822
|
meta.notified = true;
|
|
3462
4823
|
pi.sendMessage(await taskDoneMessage(meta));
|
|
@@ -3465,7 +4826,9 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3465
4826
|
conversationId,
|
|
3466
4827
|
count: unnotified.length
|
|
3467
4828
|
}, "injected completed bg tasks at session_start");
|
|
3468
|
-
if ([...tasks.values()].some((t) =>
|
|
4829
|
+
if ([...tasks.values()].some((t) => sameScope(t, scopeKey()) && t.running)) ensureWatchdog();
|
|
4830
|
+
lastPublishedSignature = null;
|
|
4831
|
+
await flushPublishSnapshot();
|
|
3469
4832
|
});
|
|
3470
4833
|
function err(text) {
|
|
3471
4834
|
return {
|
|
@@ -3482,11 +4845,11 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3482
4845
|
}
|
|
3483
4846
|
function resolveTask(taskId) {
|
|
3484
4847
|
const exact = tasks.get(taskId);
|
|
3485
|
-
if (exact &&
|
|
4848
|
+
if (exact && sameScope(exact, scopeKey())) return {
|
|
3486
4849
|
error: null,
|
|
3487
4850
|
meta: exact
|
|
3488
4851
|
};
|
|
3489
|
-
const matches = [...tasks.values()].filter((t) =>
|
|
4852
|
+
const matches = [...tasks.values()].filter((t) => sameScope(t, scopeKey()) && t.id.startsWith(taskId));
|
|
3490
4853
|
if (matches.length === 1) return {
|
|
3491
4854
|
error: null,
|
|
3492
4855
|
meta: matches[0]
|
|
@@ -3495,7 +4858,7 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3495
4858
|
return err(`Unknown task ${taskId}. Known tasks: ${knownTaskIds()}`);
|
|
3496
4859
|
}
|
|
3497
4860
|
function listTasks() {
|
|
3498
|
-
const mine = [...tasks.values()].filter((t) =>
|
|
4861
|
+
const mine = [...tasks.values()].filter((t) => sameScope(t, scopeKey()));
|
|
3499
4862
|
if (mine.length === 0) return "No background tasks.";
|
|
3500
4863
|
return mine.map((t) => {
|
|
3501
4864
|
const state = t.running ? "running" : t.exitCode !== null ? `exited ${t.exitCode}${t.killedReason ? ` (killed: ${t.killedReason})` : ""}` : t.killedReason ? `killed: ${t.killedReason}` : "ended";
|
|
@@ -3508,24 +4871,27 @@ This is a background-task completion, not a message from the user. If it needs n
|
|
|
3508
4871
|
const bgRun = {
|
|
3509
4872
|
name: "bg_run",
|
|
3510
4873
|
label: "Run in background",
|
|
3511
|
-
description: "Run a bash command in the background. Returns immediately with a task id; output streams to a log file. You are sent a message when it finishes — keep working or end your turn meanwhile. Use for anything over a couple of minutes (builds, batch jobs, retry loops, downloads). Inspect with bg_status / bg_logs, stop with bg_kill.",
|
|
3512
|
-
promptSnippet: "bg_run — run a long command without blocking; you are notified on completion",
|
|
4874
|
+
description: "Run a bash command in the background. Returns immediately with a task id; output streams to a log file. You are sent a message when it finishes — keep working or end your turn meanwhile. Use for anything over a couple of minutes (builds, batch jobs, retry loops, downloads). Inspect with bg_status / bg_logs, stop with bg_kill. Pass resumable:true ONLY for an idempotent, long-lived command (a poller/watcher/retry loop) that should be requeued and relaunched on your next run if the sandbox goes away before it finishes (a sandbox is drained and replaced with a fresh one, not restarted in place) — the requeue re-runs the command from scratch, so never mark a one-shot side-effecting job (a migration, an apply, a send) resumable.",
|
|
4875
|
+
promptSnippet: "bg_run — run a long command without blocking; you are notified on completion (resumable:true requeues the task onto a fresh sandbox if the current one is drained, idempotent commands only)",
|
|
3513
4876
|
parameters: Type.Object({
|
|
3514
4877
|
command: Type.String({ description: "Bash command to execute" }),
|
|
3515
|
-
description: Type.Optional(Type.String({ description: "Clear, concise description of what this command does in active voice (2-6 words)." }))
|
|
4878
|
+
description: Type.Optional(Type.String({ description: "Clear, concise description of what this command does in active voice (2-6 words)." })),
|
|
4879
|
+
resumable: Type.Optional(Type.Boolean({ description: "When true, this task is requeued and relaunched on your next run if the sandbox it is running on goes away before it finishes (a sandbox is drained and replaced with a fresh one, rather than restarted in place, so an in-flight task would otherwise be lost). Use ONLY for idempotent, long-lived commands (pollers, watchers, retry loops) — the requeue re-runs the command from scratch on the new sandbox, so never set this on a one-shot side-effecting command." }))
|
|
3516
4880
|
}),
|
|
3517
4881
|
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
|
|
3518
4882
|
await ensureConversationId();
|
|
3519
|
-
const { command, description = null } = params;
|
|
4883
|
+
const { command, description = null, resumable = false } = params;
|
|
3520
4884
|
const meta = await launchTask({
|
|
3521
4885
|
command,
|
|
3522
4886
|
description,
|
|
3523
|
-
cwd: ctx.cwd
|
|
4887
|
+
cwd: ctx.cwd,
|
|
4888
|
+
resumable
|
|
3524
4889
|
});
|
|
4890
|
+
const resumeNote = resumable && meta.messageId ? "\nResumable: if this sandbox is drained before the task finishes, it will be requeued and relaunched from the start on your next run." : resumable ? "\nNote: resumable was requested but this session has no durable conversation, so it will NOT be requeued if the sandbox is drained." : "";
|
|
3525
4891
|
return {
|
|
3526
4892
|
content: [{
|
|
3527
4893
|
type: "text",
|
|
3528
|
-
text: `Started background task ${taskLabel(meta)}.\nlog: ${logPath(meta.id)}\nYou will get a message when it finishes. Check on it with bg_status {"taskId":"${meta.id}"}
|
|
4894
|
+
text: `Started background task ${taskLabel(meta)}.\nlog: ${logPath(meta.id)}\nYou will get a message when it finishes. Check on it with bg_status {"taskId":"${meta.id}"}.${resumeNote}`
|
|
3529
4895
|
}],
|
|
3530
4896
|
details: {}
|
|
3531
4897
|
};
|
|
@@ -3645,6 +5011,7 @@ const all = [
|
|
|
3645
5011
|
localToolsExtension,
|
|
3646
5012
|
toolCallEnvExtension,
|
|
3647
5013
|
bashDefaultTimeoutExtension,
|
|
5014
|
+
diskGuardExtension,
|
|
3648
5015
|
toolCallSummaryExtension
|
|
3649
5016
|
];
|
|
3650
5017
|
/**
|
|
@@ -3663,8 +5030,9 @@ function platformExtensions({ sessionId, channelContext }) {
|
|
|
3663
5030
|
selfTraceExtension,
|
|
3664
5031
|
createBackgroundTasksExtension({ channelContext }),
|
|
3665
5032
|
createSubagentExtension({ channelContext }),
|
|
3666
|
-
createContextManagementExtension()
|
|
5033
|
+
createContextManagementExtension(),
|
|
5034
|
+
resourcePressureWarningExtension
|
|
3667
5035
|
];
|
|
3668
5036
|
}
|
|
3669
5037
|
//#endregion
|
|
3670
|
-
export { all, createHarness, createHealthHandler, createPlatformEnvMiddleware, installToolUpdateAutoStop, isPlatformConfigLoaded, loadPlatformConfig, localToolsExtension, mcp_default as mcpExtension, memoryExtension, platformExtensions, runToolUpdateLoop, soulExtension, toolCallEnvExtension };
|
|
5038
|
+
export { all, createHarness, createHealthHandler, createPlatformEnvMiddleware, installToolUpdateAutoStop, isPlatformConfigLoaded, loadPlatformConfig, localToolsExtension, mcp_default as mcpExtension, memoryExtension, platformExtensions, runClosingTextLoop, runToolUpdateLoop, soulExtension, toolCallEnvExtension, turnEndedWithoutVisibleText };
|