@skydiveai/pi-extensions 0.1.0-beta.20 → 0.1.0-beta.2000

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.mjs CHANGED
@@ -2,14 +2,14 @@ import { createRequire } from "node:module";
2
2
  import { DefaultExecutionEventBusManager, DefaultRequestHandler, InMemoryTaskStore } from "@a2a-js/sdk/server";
3
3
  import { UserBuilder, restHandler } from "@a2a-js/sdk/server/express";
4
4
  import { buildAgentCard, chainMiddleware, composeHandlers, createAgentExecutor, createProtocolHandlers, getCurrentTraceparent, logger, mountAt, requestHeaders, requestUrl, webHandlerToMiddleware } from "@skydiveai/pi-server";
5
- import { mkdir, open, readFile, readdir, stat, unlink } from "node:fs/promises";
5
+ import { mkdir, open, readFile, readdir, stat, unlink, writeFile } from "node:fs/promises";
6
6
  import { basename, dirname, join, relative, resolve } from "node:path";
7
7
  import { z } from "zod";
8
8
  import { pathToFileURL } from "node:url";
9
9
  import { CallToolResultSchema } from "@modelcontextprotocol/sdk/types.js";
10
10
  import { Type } from "typebox";
11
11
  import { Client } from "@modelcontextprotocol/sdk/client/index.js";
12
- import { StdioClientTransport } from "@modelcontextprotocol/sdk/client/stdio.js";
12
+ import { StdioClientTransport, getDefaultEnvironment } from "@modelcontextprotocol/sdk/client/stdio.js";
13
13
  import { StreamableHTTPClientTransport, StreamableHTTPError } from "@modelcontextprotocol/sdk/client/streamableHttp.js";
14
14
  import { UnauthorizedError } from "@modelcontextprotocol/sdk/client/auth.js";
15
15
  import { Check, Errors } from "typebox/value";
@@ -19,8 +19,12 @@ import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-http";
19
19
  import { Resource } from "@opentelemetry/resources";
20
20
  import { BatchSpanProcessor, NodeTracerProvider } from "@opentelemetry/sdk-trace-node";
21
21
  import { ATTR_SERVICE_NAME } from "@opentelemetry/semantic-conventions";
22
+ import { execFile } from "node:child_process";
23
+ import { availableParallelism } from "node:os";
24
+ import { promisify } from "node:util";
22
25
  import { hc } from "hono/client";
23
26
  import { parse } from "yaml";
27
+ import { quote } from "shell-quote";
24
28
  import { createWriteStream } from "node:fs";
25
29
  import { finished } from "node:stream/promises";
26
30
  import { createLocalBashOperations } from "@earendil-works/pi-coding-agent";
@@ -255,7 +259,7 @@ function createHealthHandler({ metadata }) {
255
259
  * read on the hot path before every LLM call), it falls back to the default
256
260
  * for that knob and logs once.
257
261
  */
258
- const log$13 = logger.child({ module: "context-management-config" });
262
+ const log$15 = logger.child({ module: "context-management-config" });
259
263
  const DEFAULT_CONTEXT_MANAGEMENT_CONFIG = {
260
264
  enabled: false,
261
265
  perResultMaxBytes: 16 * 1024,
@@ -303,7 +307,7 @@ function resolveContextManagementConfig(env = process.env) {
303
307
  maxModelCallsPerTurn: env.SKYDIVE_CTX_MAX_MODEL_CALLS
304
308
  });
305
309
  if (!parsed.success) {
306
- log$13.warn({
310
+ log$15.warn({
307
311
  event: "context_management_config_invalid",
308
312
  err: parsed.error
309
313
  }, "falling back to default context-management config");
@@ -414,6 +418,275 @@ function installIterationCap({ session, log }, configOverride = null) {
414
418
  */
415
419
  const CAPABILITY_SOUL_NUDGE = "New capability gained — once the current task is done, if this changes what you can do for the user, record it in `soul.md` so it carries into future conversations rather than being rediscovered from scratch (then commit and push).";
416
420
  //#endregion
421
+ //#region src/extensions/tool-call-summary.ts
422
+ const log$14 = logger.child({ module: "tool-call-summary-extension" });
423
+ /**
424
+ * The injected parameter name: a namespaced sentinel, so it can never collide
425
+ * with a real tool argument and is unmistakable in transcripts and logs. The
426
+ * frontend renderer (ANY-2723) duplicates this literal — keep the two in sync.
427
+ */
428
+ const TOOL_CALL_SUMMARY_FIELD = "__skydive_summary__";
429
+ /** JSON Schema fragment for the injected parameter. */
430
+ const SUMMARY_PROPERTY = {
431
+ type: "string",
432
+ description: "Required for every tool call. A concise, specific summary (max ~8 words) of what THIS call does and why, written for a person watching the conversation, e.g. \"Searching feedback for billing complaints\" or \"Reading the auth middleware\". Address the user directly in second person: the summary is read by the user, so refer to their things as \"your\", never in third person — \"Reading your emails\", not \"Reading his emails\". Always use the present progressive tense, since it is shown while the call runs: \"Updating your Slack\", never \"Updated your Slack\". Make each summary distinct from your other tool calls; never reuse a generic label like \"Search query\" or \"Running command\"."
433
+ };
434
+ const jsonSchemaObjectSchema = z.object({
435
+ type: z.unknown().optional(),
436
+ properties: z.record(z.string(), z.unknown()).optional(),
437
+ required: z.array(z.string()).optional(),
438
+ additionalProperties: z.unknown().optional()
439
+ }).passthrough();
440
+ const toolEntrySchema = z.object({
441
+ name: z.string().optional(),
442
+ input_schema: jsonSchemaObjectSchema.optional(),
443
+ parameters: jsonSchemaObjectSchema.optional(),
444
+ function: z.object({
445
+ name: z.string().optional(),
446
+ parameters: jsonSchemaObjectSchema.optional()
447
+ }).passthrough().optional()
448
+ }).passthrough();
449
+ const payloadWithToolsSchema = z.object({ tools: z.array(z.unknown()) }).passthrough();
450
+ /**
451
+ * Add the summary property to one JSON Schema object. Returns the augmented
452
+ * copy, or `null` when the tool should be left untouched: a strict schema
453
+ * (`additionalProperties: false`) whose validation would reject the extra
454
+ * field, or one that already declares a `__skydive_summary__` property of its own.
455
+ */
456
+ function augmentSchema(schema) {
457
+ if (schema.additionalProperties === false) return null;
458
+ const properties = schema.properties ?? {};
459
+ if ("__skydive_summary__" in properties) return null;
460
+ const required = schema.required ?? [];
461
+ return {
462
+ ...schema,
463
+ type: schema.type ?? "object",
464
+ properties: {
465
+ [TOOL_CALL_SUMMARY_FIELD]: SUMMARY_PROPERTY,
466
+ ...properties
467
+ },
468
+ required: required.includes("__skydive_summary__") ? required : [...required, TOOL_CALL_SUMMARY_FIELD]
469
+ };
470
+ }
471
+ /**
472
+ * Augment a single tool entry, dispatching on which provider shape it is.
473
+ * Returns the (possibly rebuilt) entry and whether anything changed. Skipped
474
+ * tools — wrong shape, strict, or name in `strictToolNames` — return unchanged.
475
+ */
476
+ function augmentToolEntry(entry, strictToolNames) {
477
+ const parsed = toolEntrySchema.safeParse(entry);
478
+ if (!parsed.success) return {
479
+ entry,
480
+ changed: false
481
+ };
482
+ const tool = parsed.data;
483
+ const name = tool.name ?? tool.function?.name ?? null;
484
+ if (name !== null && strictToolNames.has(name)) return {
485
+ entry,
486
+ changed: false
487
+ };
488
+ if (tool.input_schema) {
489
+ const augmented = augmentSchema(tool.input_schema);
490
+ if (!augmented) return {
491
+ entry,
492
+ changed: false
493
+ };
494
+ return {
495
+ entry: {
496
+ ...tool,
497
+ input_schema: augmented
498
+ },
499
+ changed: true
500
+ };
501
+ }
502
+ if (tool.parameters) {
503
+ const augmented = augmentSchema(tool.parameters);
504
+ if (!augmented) return {
505
+ entry,
506
+ changed: false
507
+ };
508
+ return {
509
+ entry: {
510
+ ...tool,
511
+ parameters: augmented
512
+ },
513
+ changed: true
514
+ };
515
+ }
516
+ if (tool.function?.parameters) {
517
+ const augmented = augmentSchema(tool.function.parameters);
518
+ if (!augmented) return {
519
+ entry,
520
+ changed: false
521
+ };
522
+ return {
523
+ entry: {
524
+ ...tool,
525
+ function: {
526
+ ...tool.function,
527
+ parameters: augmented
528
+ }
529
+ },
530
+ changed: true
531
+ };
532
+ }
533
+ return {
534
+ entry,
535
+ changed: false
536
+ };
537
+ }
538
+ /**
539
+ * Inject the summary field into every eligible tool in a provider payload.
540
+ * Returns a new payload when at least one tool was augmented, or `undefined`
541
+ * to signal "no change" (which keeps the original payload, per the
542
+ * `before_provider_request` contract).
543
+ *
544
+ * @param payload The outgoing provider payload (shape varies by provider).
545
+ * @param strictToolNames Names of tools whose registered schema is strict and
546
+ * must be skipped to avoid validation errors.
547
+ */
548
+ function injectToolCallSummary(payload, strictToolNames) {
549
+ const parsed = payloadWithToolsSchema.safeParse(payload);
550
+ if (!parsed.success || parsed.data.tools.length === 0) return void 0;
551
+ let changed = false;
552
+ const tools = parsed.data.tools.map((entry) => {
553
+ const result = augmentToolEntry(entry, strictToolNames);
554
+ if (result.changed) changed = true;
555
+ return result.entry;
556
+ });
557
+ if (!changed) return void 0;
558
+ return {
559
+ ...parsed.data,
560
+ tools
561
+ };
562
+ }
563
+ /**
564
+ * Names of registered tools whose schema sets `additionalProperties: false`.
565
+ * Pi validates the model's tool args against this registered schema, so the
566
+ * injected field would make a strict tool's call fail validation — skip them.
567
+ */
568
+ function getStrictToolNames(pi) {
569
+ const names = /* @__PURE__ */ new Set();
570
+ for (const tool of pi.getAllTools()) {
571
+ const parsed = jsonSchemaObjectSchema.safeParse(tool.parameters);
572
+ if (parsed.success && parsed.data.additionalProperties === false) names.add(tool.name);
573
+ }
574
+ return names;
575
+ }
576
+ /**
577
+ * Whether a tool's own schema declares a `__skydive_summary__` property. We
578
+ * never inject into such a tool, so any value it carries is a real argument
579
+ * and must be left alone.
580
+ */
581
+ function schemaDeclaresSummary(parameters) {
582
+ const parsed = jsonSchemaObjectSchema.safeParse(parameters);
583
+ return parsed.success && parsed.data.properties != null && "__skydive_summary__" in parsed.data.properties;
584
+ }
585
+ function toolDeclaresSummaryParam(pi, toolName) {
586
+ const tool = pi.getAllTools().find((candidate) => candidate.name === toolName);
587
+ if (!tool) return false;
588
+ return schemaDeclaresSummary(tool.parameters);
589
+ }
590
+ /** A copy of `args` without the injected summary. Never mutates its input. */
591
+ function withoutInjectedSummary(args) {
592
+ if (args == null || typeof args !== "object" || Array.isArray(args)) return args;
593
+ if (!("__skydive_summary__" in args)) return args;
594
+ const { [TOOL_CALL_SUMMARY_FIELD]: _summary, ...rest } = args;
595
+ return rest;
596
+ }
597
+ /**
598
+ * Register a tool so the injected summary is removed *before* pi validates
599
+ * the model's arguments against the tool's schema.
600
+ *
601
+ * Needed because the `tool_call` strip below runs too late. pi-agent-core's
602
+ * `prepareToolCall` goes: `tool.prepareArguments` → `validateToolArguments` →
603
+ * `beforeToolCall` (which is what dispatches `tool_call`). A tool whose schema
604
+ * sets `additionalProperties: false` therefore rejects the summary and errors
605
+ * the call — client-side, before any request reaches the server — while the
606
+ * strip meant to prevent exactly that sits one step further down.
607
+ *
608
+ * `augmentSchema` already skips strict tools, but that only controls what the
609
+ * model is *told*. It still emits the field on those tools, because every
610
+ * other tool in the list declares it as required for every call. So a server
611
+ * can connect, bind its tools, present a healthy inventory, and have every
612
+ * call fail — and it reads as the server's fault when it is ours.
613
+ *
614
+ * Apply to tools registered from schemas we do not author: MCP servers and
615
+ * local `tools/*.ts`. Tools built here with `Type.Object(...)` do not need it
616
+ * (TypeBox emits no `additionalProperties`, so the field validates fine and
617
+ * the `tool_call` strip removes it in time).
618
+ *
619
+ * Fails open, like the rest of this module: if the strip throws, the original
620
+ * arguments are used rather than failing the call.
621
+ */
622
+ function withSummaryStrippedBeforeValidation(tool) {
623
+ if (schemaDeclaresSummary(tool.parameters)) return tool;
624
+ const toolPrepare = tool.prepareArguments;
625
+ const prepareArguments = ((args) => {
626
+ let stripped = args;
627
+ try {
628
+ stripped = withoutInjectedSummary(args);
629
+ } catch (err) {
630
+ log$14.error({
631
+ err,
632
+ event: "tool_call_summary_prepare_strip_failed",
633
+ toolName: tool.name
634
+ }, "tool_call_summary pre-validation strip failed; leaving arguments untouched");
635
+ }
636
+ return toolPrepare ? toolPrepare(stripped) : stripped;
637
+ });
638
+ return {
639
+ ...tool,
640
+ prepareArguments
641
+ };
642
+ }
643
+ /**
644
+ * Remove the injected summary from a tool's execution input. No-op when the
645
+ * field is absent, or when the tool genuinely declares a `__skydive_summary__`
646
+ * parameter of its own (which we never inject into, so its value is real).
647
+ * Mutates `input` in place, matching the `tool_call` contract.
648
+ *
649
+ * Fails open: this runs on the critical path of tool execution, and the
650
+ * `getAllTools()` lookup can throw. On any error we leave `input` untouched
651
+ * (the sentinel may pass through to the tool, but a bug here can never break
652
+ * tool execution).
653
+ */
654
+ function stripInjectedSummary(pi, toolName, input) {
655
+ try {
656
+ if (!("__skydive_summary__" in input)) return;
657
+ if (toolDeclaresSummaryParam(pi, toolName)) return;
658
+ delete input[TOOL_CALL_SUMMARY_FIELD];
659
+ } catch (err) {
660
+ log$14.error({
661
+ err,
662
+ event: "tool_call_summary_strip_failed",
663
+ toolName
664
+ }, "tool_call_summary strip failed; leaving tool input untouched");
665
+ }
666
+ }
667
+ /**
668
+ * Compute the rewritten payload for a `before_provider_request` event, failing
669
+ * open: on any error the original payload is left untouched so a bug here can
670
+ * never break an LLM call.
671
+ */
672
+ function buildInjectedPayload(pi, payload) {
673
+ try {
674
+ return injectToolCallSummary(payload, getStrictToolNames(pi));
675
+ } catch (err) {
676
+ log$14.error({
677
+ err,
678
+ event: "tool_call_summary_injection_failed"
679
+ }, "tool_call_summary injection failed; passing payload through unchanged");
680
+ return;
681
+ }
682
+ }
683
+ const toolCallSummaryExtension = (pi) => {
684
+ pi.on("before_provider_request", (event) => buildInjectedPayload(pi, event.payload));
685
+ pi.on("tool_call", (event) => {
686
+ stripInjectedSummary(pi, event.toolName, event.input);
687
+ });
688
+ };
689
+ //#endregion
417
690
  //#region src/extensions/local-tools.ts
418
691
  /**
419
692
  * Local-tools adapter as a pi extension. Mirrors the mcp.ts hot-reload pattern
@@ -437,7 +710,7 @@ const CAPABILITY_SOUL_NUDGE = "New capability gained — once the current task i
437
710
  * or `ToolDefinition[]`. Files starting with `_` or `.` are skipped, so
438
711
  * `tools/_example.ts` documents the shape without registering.
439
712
  */
440
- const log$12 = logger.child({ module: "local-tools-extension" });
713
+ const log$13 = logger.child({ module: "local-tools-extension" });
441
714
  const TOOLS_DIRNAME = "tools";
442
715
  const fileState = /* @__PURE__ */ new Map();
443
716
  let pendingLocalToolsUpdate = null;
@@ -564,7 +837,7 @@ async function reconcileLocalTools({ pi, dir }) {
564
837
  action = existing ? "refreshed" : "added";
565
838
  }
566
839
  for (const tool of tools) {
567
- pi.registerTool(withDefaultPromptSnippet(tool));
840
+ pi.registerTool(withSummaryStrippedBeforeValidation(withDefaultPromptSnippet(tool)));
568
841
  summary.totalTools++;
569
842
  }
570
843
  if (action === "added") summary.added.push(file);
@@ -581,7 +854,7 @@ async function reconcileAndQueue({ pi, dir, reason }) {
581
854
  dir
582
855
  });
583
856
  if (reason !== "session_start" && summaryHasChanges$1(summary)) pendingLocalToolsUpdate = summary;
584
- log$12.info({
857
+ log$13.info({
585
858
  event: "local_tools_reconcile",
586
859
  reason,
587
860
  total_tools: summary.totalTools,
@@ -603,7 +876,7 @@ const localToolsExtension = (pi) => {
603
876
  reason: "session_start"
604
877
  });
605
878
  } catch (err) {
606
- log$12.error({
879
+ log$13.error({
607
880
  err,
608
881
  event: "local_tools_reconcile_failed"
609
882
  }, "local tools reconcile failed");
@@ -615,7 +888,7 @@ const localToolsExtension = (pi) => {
615
888
  try {
616
889
  current = await listToolFiles(dir);
617
890
  } catch (err) {
618
- log$12.warn({
891
+ log$13.warn({
619
892
  err,
620
893
  event: "local_tools_listing_failed"
621
894
  }, "tools/ listing failed");
@@ -637,7 +910,7 @@ const localToolsExtension = (pi) => {
637
910
  reason: "auto_reload"
638
911
  });
639
912
  } catch (err) {
640
- log$12.error({
913
+ log$13.error({
641
914
  err,
642
915
  event: "local_tools_auto_reload_failed"
643
916
  }, "auto-reload after tools/ change failed");
@@ -684,6 +957,47 @@ function createStderrBuffer({ maxBytes }) {
684
957
  const DEFAULT_CONNECT_TIMEOUT_MS = 5e3;
685
958
  const STDERR_BUFFER_BYTES = 4096;
686
959
  /**
960
+ * Sandbox egress variables a stdio MCP server needs to make outbound HTTPS
961
+ * calls. The SDK's default child environment is a tiny whitelist (HOME,
962
+ * LOGNAME, PATH, SHELL, TERM, USER on Linux), which drops NODE_EXTRA_CA_CERTS
963
+ * — but the sandbox's egress proxy terminates TLS with a private CA, so a
964
+ * child spawned without it fails every HTTPS request with a bare
965
+ * `fetch failed` (cause: UNABLE_TO_VERIFY_LEAF_SIGNATURE) that looks like the
966
+ * remote service is down. Same story for SSL_CERT_FILE (OpenSSL-based
967
+ * runtimes) and the *_PROXY set. Inherit them from this process so a stdio
968
+ * server gets working egress like every other process in the sandbox.
969
+ */
970
+ const INHERITED_EGRESS_ENV_VARS = [
971
+ "NODE_EXTRA_CA_CERTS",
972
+ "SSL_CERT_FILE",
973
+ "SSL_CERT_DIR",
974
+ "REQUESTS_CA_BUNDLE",
975
+ "CURL_CA_BUNDLE",
976
+ "HTTP_PROXY",
977
+ "HTTPS_PROXY",
978
+ "NO_PROXY",
979
+ "http_proxy",
980
+ "https_proxy",
981
+ "no_proxy"
982
+ ];
983
+ /**
984
+ * Environment for a stdio MCP server child: the SDK's safe defaults, plus the
985
+ * sandbox's egress/TLS variables, plus (last, so it wins) the server's own
986
+ * configured `env`. Building the merge here — instead of only when `env` is
987
+ * unset — also fixes the workaround trap where supplying any `env` in
988
+ * mcp.config.json silently replaced the ENTIRE default set, so a config that
989
+ * added one API key lost PATH/HOME and the server failed to spawn at all.
990
+ */
991
+ function buildStdioEnv(configured, processEnv = process.env) {
992
+ const env = { ...getDefaultEnvironment() };
993
+ for (const key of INHERITED_EGRESS_ENV_VARS) {
994
+ const value = processEnv[key];
995
+ if (value !== void 0) env[key] = value;
996
+ }
997
+ if (configured) Object.assign(env, configured);
998
+ return env;
999
+ }
1000
+ /**
687
1001
  * undici's fetch throws `TypeError: fetch failed` with the actual
688
1002
  * network error hung off `.cause` (e.g. `getaddrinfo ENOTFOUND ...`,
689
1003
  * `ECONNREFUSED`, TLS errors). Surfacing only `err.message` makes
@@ -691,6 +1005,19 @@ const STDERR_BUFFER_BYTES = 4096;
691
1005
  * indistinguishable from any other transport problem. Walk the cause
692
1006
  * chain so the agent sees the real underlying error.
693
1007
  */
1008
+ /**
1009
+ * True when an error from an http MCP transport (connect, listTools, or a tool
1010
+ * call) is an authentication failure. With no authProvider configured the SDK
1011
+ * surfaces a 401 as `StreamableHTTPError(401)`; older paths translate it to
1012
+ * `UnauthorizedError`. A dead/expired OAuth token (the proxy can no longer
1013
+ * mint one) shows up here on the NEXT request against a previously-connected
1014
+ * client — not just at connect — so reconcile must re-classify such a failure
1015
+ * as `pending_auth` instead of a generic `failed`, keeping the "waiting on
1016
+ * auth" report consistent with `platform auth`.
1017
+ */
1018
+ function isUnauthorizedError(err) {
1019
+ return err instanceof UnauthorizedError || err instanceof StreamableHTTPError && err.code === 401;
1020
+ }
694
1021
  function formatError(err) {
695
1022
  if (!(err instanceof Error)) return String(err);
696
1023
  const parts = [err.message];
@@ -702,6 +1029,15 @@ function formatError(err) {
702
1029
  }
703
1030
  return parts.join(": ");
704
1031
  }
1032
+ function probeAlive(pid) {
1033
+ if (pid === null) return null;
1034
+ try {
1035
+ process.kill(pid, 0);
1036
+ return true;
1037
+ } catch {
1038
+ return false;
1039
+ }
1040
+ }
705
1041
  async function connectHttp(_id, config, client) {
706
1042
  const transport = new StreamableHTTPClientTransport(new URL(config.url), { ...config.headers !== null ? { requestInit: { headers: config.headers } } : {} });
707
1043
  try {
@@ -712,12 +1048,12 @@ async function connectHttp(_id, config, client) {
712
1048
  stderr: null
713
1049
  };
714
1050
  } catch (err) {
715
- if (err instanceof UnauthorizedError || err instanceof StreamableHTTPError && err.code === 401) return {
1051
+ if (isUnauthorizedError(err)) return {
716
1052
  status: "pending_auth",
717
1053
  client,
718
1054
  stderr: "",
719
1055
  stderrBuffer: null,
720
- cliHint: `platform auth mcp ${config.url}`
1056
+ cliHint: `platform auth mcp ${config.url} --service "<Product>"`
721
1057
  };
722
1058
  return {
723
1059
  status: "failed",
@@ -736,9 +1072,9 @@ async function connectClient(id, config, opts = {}) {
736
1072
  const params = {
737
1073
  command: config.command,
738
1074
  args: config.args,
739
- stderr: "pipe"
1075
+ stderr: "pipe",
1076
+ env: buildStdioEnv(config.env)
740
1077
  };
741
- if (config.env !== null) params.env = config.env;
742
1078
  if (config.cwd !== null) params.cwd = config.cwd;
743
1079
  const transport = new StdioClientTransport(params);
744
1080
  const stderrBuffer = createStderrBuffer({ maxBytes: STDERR_BUFFER_BYTES });
@@ -750,6 +1086,10 @@ async function connectClient(id, config, opts = {}) {
750
1086
  resolve();
751
1087
  };
752
1088
  });
1089
+ let stdoutMessages = 0;
1090
+ transport.onmessage = () => {
1091
+ if (stdoutMessages < 1e4) stdoutMessages += 1;
1092
+ };
753
1093
  const connectPromise = client.connect(transport);
754
1094
  const TIMEOUT_SENTINEL = Symbol("timeout");
755
1095
  const result = await Promise.race([
@@ -767,12 +1107,20 @@ async function connectClient(id, config, opts = {}) {
767
1107
  error: `server "${id}" exited before completing initialize`,
768
1108
  stderr: stderrBuffer.read()
769
1109
  };
770
- if (result === TIMEOUT_SENTINEL) return {
771
- status: "timeout",
772
- client,
773
- stderr: stderrBuffer.read(),
774
- stderrBuffer
775
- };
1110
+ if (result === TIMEOUT_SENTINEL) {
1111
+ const pid = transport.pid;
1112
+ return {
1113
+ status: "timeout",
1114
+ client,
1115
+ stderr: stderrBuffer.read(),
1116
+ stderrBuffer,
1117
+ diagnostics: {
1118
+ pid,
1119
+ alive: probeAlive(pid),
1120
+ stdoutMessages
1121
+ }
1122
+ };
1123
+ }
776
1124
  try {
777
1125
  await client.close();
778
1126
  } catch {}
@@ -783,6 +1131,126 @@ async function connectClient(id, config, opts = {}) {
783
1131
  };
784
1132
  }
785
1133
  //#endregion
1134
+ //#region src/extensions/mcp/limits.ts
1135
+ /**
1136
+ * Tool-budget guardrails for MCP registration.
1137
+ *
1138
+ * Why this exists: the agent loop sends the *full* tool list — every tool's
1139
+ * name, description, and JSON-schema `parameters` — to the model on every
1140
+ * prompt. MCP servers add tools without bound: a handful of chatty servers (or
1141
+ * one server that exposes 100+ tools, or a few tools with enormous schemas) can
1142
+ * push the registered set past what the model can accept, and the request is
1143
+ * rejected before the turn even runs. Because `reconcile` re-registers from
1144
+ * `mcp.config.json` on *every* turn, an over-limit config bricks the harness on
1145
+ * a loop — the agent can't get a turn to run in order to edit the config back
1146
+ * down. Worse, the person can't tell *why*: tools just stop working.
1147
+ *
1148
+ * What actually overflows the request is *tokens*, not tool count — a few tools
1149
+ * with deeply-nested schemas and long descriptions cost more than a hundred
1150
+ * trivial ones. And how many tokens are safe depends on the *model*: a 200K
1151
+ * context window can afford far more tool surface than a 32K one. So the primary
1152
+ * limiter is a **token budget derived from the active model's context window**,
1153
+ * with a fixed tool-count cap as a coarse secondary guard (and the fallback
1154
+ * when the model — hence its window — isn't known at reconcile time).
1155
+ *
1156
+ * The fix is to make registration bounded and fail-soft. We register in
1157
+ * deterministic config order and stop before we blow the budget, recording how
1158
+ * much we dropped so the agent is *told* it hit the limit and which servers
1159
+ * were truncated. A config that would have bricked the harness now degrades to
1160
+ * "a bounded set of tools plus a loud warning", which the agent can act on by
1161
+ * pruning servers.
1162
+ *
1163
+ * Everything is env-overridable so the ceilings can be tuned per deployment
1164
+ * without a release, but ships with conservative defaults. A cap value of 0 (or
1165
+ * a non-finite / negative override) disables that cap — an explicit escape
1166
+ * hatch, not the default.
1167
+ */
1168
+ const env = process.env;
1169
+ /**
1170
+ * Fraction of the model's context window we're willing to spend on MCP tool
1171
+ * schemas. Tool definitions are sent on every prompt, so they permanently eat
1172
+ * into the window available for the conversation, but a generous tool budget is
1173
+ * worth more than a marginally larger conversation window given how the harness
1174
+ * is used. 0.35 of a 200K window is ~70K tokens of tool schema, comfortably
1175
+ * more than any sane MCP setup; of a 32K window it's ~11.2K, which still forces
1176
+ * truncation before a small model chokes.
1177
+ */
1178
+ const DEFAULT_MCP_TOOL_TOKEN_BUDGET_FRACTION = .35;
1179
+ /**
1180
+ * Floor for the token budget when the model's context window is unknown at
1181
+ * reconcile time (e.g. the model hasn't been resolved yet). Generous enough not
1182
+ * to truncate an ordinary tool set, low enough to still catch a runaway.
1183
+ */
1184
+ const DEFAULT_MCP_TOOL_TOKEN_BUDGET_FLOOR = 16e3;
1185
+ /**
1186
+ * Read a positive-integer cap from an env var, falling back to `fallback`.
1187
+ * A `0` override (or any non-finite / negative value) means "no cap" and is
1188
+ * returned as `Infinity`, so callers can compare against it directly.
1189
+ */
1190
+ function readCap(raw, fallback) {
1191
+ if (raw === void 0 || raw.trim() === "") return fallback;
1192
+ const parsed = Number(raw);
1193
+ if (!Number.isFinite(parsed) || parsed < 0) return Number.POSITIVE_INFINITY;
1194
+ if (parsed === 0) return Number.POSITIVE_INFINITY;
1195
+ return Math.floor(parsed);
1196
+ }
1197
+ /** Read a fraction in (0, 1] from an env var, falling back to `fallback`. */
1198
+ function readFraction(raw, fallback) {
1199
+ if (raw === void 0 || raw.trim() === "") return fallback;
1200
+ const parsed = Number(raw);
1201
+ if (!Number.isFinite(parsed) || parsed <= 0 || parsed > 1) return fallback;
1202
+ return parsed;
1203
+ }
1204
+ /** Read a non-negative integer from an env var, falling back to `fallback`. */
1205
+ function readNonNegativeInt(raw, fallback) {
1206
+ if (raw === void 0 || raw.trim() === "") return fallback;
1207
+ const parsed = Number(raw);
1208
+ if (!Number.isFinite(parsed) || parsed < 0) return fallback;
1209
+ return Math.floor(parsed);
1210
+ }
1211
+ /**
1212
+ * Resolve the active caps from the environment. Read once per reconcile so a
1213
+ * deployment can retune without a restart, cheap enough not to cache.
1214
+ */
1215
+ function resolveMcpToolLimits() {
1216
+ return {
1217
+ maxTotalTools: readCap(env["SKYDIVE_MCP_MAX_TOTAL_TOOLS"], 128),
1218
+ maxToolsPerServer: readCap(env["SKYDIVE_MCP_MAX_TOOLS_PER_SERVER"], 50),
1219
+ tokenBudgetFraction: readFraction(env["SKYDIVE_MCP_TOOL_TOKEN_BUDGET_FRACTION"], DEFAULT_MCP_TOOL_TOKEN_BUDGET_FRACTION),
1220
+ tokenBudgetFloor: readNonNegativeInt(env["SKYDIVE_MCP_TOOL_TOKEN_BUDGET_FLOOR"], DEFAULT_MCP_TOOL_TOKEN_BUDGET_FLOOR)
1221
+ };
1222
+ }
1223
+ /**
1224
+ * The token budget for MCP tool schemas, given the active model's context
1225
+ * window (or undefined when the model isn't known yet).
1226
+ *
1227
+ * With a known window we spend `tokenBudgetFraction` of it, but never less than
1228
+ * the floor — a tiny window shouldn't collapse the budget to near-zero and
1229
+ * strand every tool. With no window we fall back to the floor outright.
1230
+ */
1231
+ function resolveTokenBudget(limits, contextWindow) {
1232
+ if (contextWindow === void 0 || !Number.isFinite(contextWindow)) return limits.tokenBudgetFloor;
1233
+ return Math.max(limits.tokenBudgetFloor, Math.floor(contextWindow * limits.tokenBudgetFraction));
1234
+ }
1235
+ /**
1236
+ * Estimate the tokens an MCP tool's *definition* costs in the request. We
1237
+ * serialize what's actually sent to the model — the tool name, its description,
1238
+ * and its JSON-schema parameters — and apply pi's own ~chars/4 heuristic
1239
+ * (`estimateTokens` in the coding agent uses the same convention; there is no
1240
+ * per-provider tokenizer to lean on, and the provider's real usage is only
1241
+ * known after the response). Rough by design, but it tracks the true cost far
1242
+ * better than a flat per-tool count: a fat schema is charged for its fatness.
1243
+ */
1244
+ function estimateToolTokens(tool) {
1245
+ let chars = tool.name.length + (tool.description?.length ?? 0);
1246
+ if (tool.inputSchema !== void 0 && tool.inputSchema !== null) try {
1247
+ chars += JSON.stringify(tool.inputSchema).length;
1248
+ } catch {
1249
+ chars += 256;
1250
+ }
1251
+ return Math.ceil(chars / 4);
1252
+ }
1253
+ //#endregion
786
1254
  //#region src/extensions/mcp/mcp-config.ts
787
1255
  /**
788
1256
  * mcp.config.json schema + loader. Split out from the extension so it has
@@ -862,7 +1330,7 @@ async function loadMcpConfig(path) {
862
1330
  * Clients are keyed by JSON-stringified config and reused across
863
1331
  * reloads — only changed configs reconnect.
864
1332
  */
865
- const log$11 = logger.child({ module: "mcp-extension" });
1333
+ const log$12 = logger.child({ module: "mcp-extension" });
866
1334
  async function closeConnected(connected) {
867
1335
  try {
868
1336
  await connected.client.close();
@@ -952,7 +1420,7 @@ var McpExtension = class {
952
1420
  const parameters = Type.Unsafe(tool.inputSchema);
953
1421
  const description = tool.description?.trim() ?? "";
954
1422
  const promptSnippet = description.length > 0 ? description : `MCP tool from server "${serverId}".`;
955
- pi.registerTool({
1423
+ pi.registerTool(withSummaryStrippedBeforeValidation({
956
1424
  name,
957
1425
  label: `MCP: ${serverId}/${tool.name}`,
958
1426
  description,
@@ -986,22 +1454,29 @@ var McpExtension = class {
986
1454
  };
987
1455
  }
988
1456
  }
989
- });
1457
+ }));
990
1458
  this.registeredMcpToolNames.add(name);
991
1459
  }
992
- async reconcile({ pi, configPath, connectTimeoutMs }) {
1460
+ async reconcile({ pi, configPath, connectTimeoutMs, contextWindow }) {
993
1461
  let config;
994
1462
  try {
995
1463
  config = await loadMcpConfig(configPath);
996
1464
  } catch (err) {
997
1465
  throw new Error(`Failed to load MCP config: ${err instanceof Error ? err.message : String(err)}`);
998
1466
  }
1467
+ const limits = resolveMcpToolLimits();
1468
+ const tokenBudget = resolveTokenBudget(limits, contextWindow);
999
1469
  const summary = {
1000
1470
  added: [],
1001
1471
  removed: [],
1002
1472
  refreshed: [],
1003
1473
  errors: [],
1004
1474
  totalTools: 0,
1475
+ droppedTools: 0,
1476
+ limits,
1477
+ tokenBudget,
1478
+ tokensUsed: 0,
1479
+ toolCounts: {},
1005
1480
  servers: {}
1006
1481
  };
1007
1482
  const desiredIds = new Set(Object.keys(config.servers));
@@ -1025,14 +1500,46 @@ var McpExtension = class {
1025
1500
  if (outcome.error) summary.errors.push(outcome.error);
1026
1501
  if (outcome.change === "added") summary.added.push(id);
1027
1502
  else if (outcome.change === "refreshed") summary.refreshed.push(id);
1028
- if (outcome.tools) for (const tool of outcome.tools.list) {
1029
- this.registerMcpTool({
1030
- pi,
1031
- serverId: id,
1032
- client: outcome.tools.client,
1033
- tool
1034
- });
1035
- summary.totalTools++;
1503
+ if (outcome.tools) {
1504
+ const advertised = outcome.tools.list.length;
1505
+ const perServerRoom = Math.min(advertised, limits.maxToolsPerServer);
1506
+ let registered = 0;
1507
+ let serverTokens = 0;
1508
+ let droppedReason = null;
1509
+ for (const tool of outcome.tools.list) {
1510
+ if (summary.totalTools >= limits.maxTotalTools) {
1511
+ droppedReason = "total";
1512
+ break;
1513
+ }
1514
+ if (registered >= perServerRoom) {
1515
+ droppedReason = "per_server";
1516
+ break;
1517
+ }
1518
+ const cost = estimateToolTokens(tool);
1519
+ if (summary.tokensUsed + cost > tokenBudget && summary.totalTools > 0) {
1520
+ droppedReason = "tokens";
1521
+ break;
1522
+ }
1523
+ this.registerMcpTool({
1524
+ pi,
1525
+ serverId: id,
1526
+ client: outcome.tools.client,
1527
+ tool
1528
+ });
1529
+ registered++;
1530
+ serverTokens += cost;
1531
+ summary.totalTools++;
1532
+ summary.tokensUsed += cost;
1533
+ }
1534
+ const dropped = advertised - registered;
1535
+ if (dropped > 0) summary.droppedTools += dropped;
1536
+ else droppedReason = null;
1537
+ summary.toolCounts[id] = {
1538
+ advertised,
1539
+ registered,
1540
+ tokens: serverTokens,
1541
+ droppedReason
1542
+ };
1036
1543
  }
1037
1544
  }
1038
1545
  return summary;
@@ -1098,7 +1605,8 @@ var McpExtension = class {
1098
1605
  },
1099
1606
  serverStatus: {
1100
1607
  status: "timeout",
1101
- stderr: result.stderr
1608
+ stderr: result.stderr,
1609
+ diagnostics: result.diagnostics
1102
1610
  },
1103
1611
  change,
1104
1612
  error: null,
@@ -1145,7 +1653,8 @@ var McpExtension = class {
1145
1653
  },
1146
1654
  serverStatus: {
1147
1655
  status: "timeout",
1148
- stderr: retry.stderr
1656
+ stderr: retry.stderr,
1657
+ diagnostics: retry.diagnostics
1149
1658
  },
1150
1659
  change: null,
1151
1660
  error: null,
@@ -1178,7 +1687,8 @@ var McpExtension = class {
1178
1687
  store: connected,
1179
1688
  serverStatus: {
1180
1689
  status: "timeout",
1181
- stderr
1690
+ stderr,
1691
+ diagnostics: null
1182
1692
  },
1183
1693
  change: null,
1184
1694
  error: null,
@@ -1189,6 +1699,64 @@ var McpExtension = class {
1189
1699
  try {
1190
1700
  mcpTools = (await connected.client.listTools()).tools;
1191
1701
  } catch (err) {
1702
+ if (isUnauthorizedError(err) && serverConfig.transport === "http") {
1703
+ await closeConnected(connected);
1704
+ const retry = await connectClient(id, serverConfig, { connectTimeoutMs });
1705
+ if (retry.status === "pending_auth") return {
1706
+ id,
1707
+ store: {
1708
+ client: retry.client,
1709
+ configKey,
1710
+ status: "pending_auth",
1711
+ stderrBuffer: null,
1712
+ cliHint: retry.cliHint
1713
+ },
1714
+ serverStatus: {
1715
+ status: "pending_auth",
1716
+ stderr: "",
1717
+ cliHint: retry.cliHint
1718
+ },
1719
+ change: null,
1720
+ error: null,
1721
+ tools: null
1722
+ };
1723
+ if (retry.status === "failed") return {
1724
+ id,
1725
+ store: null,
1726
+ serverStatus: {
1727
+ status: "failed",
1728
+ error: retry.error,
1729
+ stderr: retry.stderr
1730
+ },
1731
+ change: null,
1732
+ error: {
1733
+ serverId: id,
1734
+ message: retry.error
1735
+ },
1736
+ tools: null
1737
+ };
1738
+ if (retry.status === "connected") {
1739
+ connected = {
1740
+ client: retry.client,
1741
+ configKey,
1742
+ status: "connected",
1743
+ stderrBuffer: retry.stderr,
1744
+ cliHint: null
1745
+ };
1746
+ mcpTools = (await connected.client.listTools()).tools;
1747
+ return {
1748
+ id,
1749
+ store: connected,
1750
+ serverStatus: { status: "connected" },
1751
+ change: action === "reused" ? "refreshed" : action,
1752
+ error: null,
1753
+ tools: {
1754
+ client: connected.client,
1755
+ list: mcpTools
1756
+ }
1757
+ };
1758
+ }
1759
+ }
1192
1760
  const message = err instanceof Error ? err.message : String(err);
1193
1761
  const stderr = connected.stderrBuffer?.read() ?? "";
1194
1762
  return {
@@ -1219,17 +1787,21 @@ var McpExtension = class {
1219
1787
  }
1220
1788
  };
1221
1789
  }
1222
- async reconcileAndRecordMtime({ pi, configPath, reason }) {
1790
+ async reconcileAndRecordMtime({ pi, configPath, reason, contextWindow }) {
1223
1791
  const summary = await this.reconcile({
1224
1792
  pi,
1225
- configPath
1793
+ configPath,
1794
+ contextWindow
1226
1795
  });
1227
1796
  this.lastConfigMtimeMs = await readConfigMtimeMs(configPath);
1228
1797
  if (reason !== "session_start" && summaryHasChanges(summary)) this.pendingMcpUpdate = summary;
1229
- log$11.info({
1798
+ log$12.info({
1230
1799
  event: "mcp_reconcile",
1231
1800
  reason,
1232
1801
  total_tools: summary.totalTools,
1802
+ tokens_used: summary.tokensUsed,
1803
+ token_budget: summary.tokenBudget,
1804
+ dropped_tools: summary.droppedTools,
1233
1805
  added: summary.added,
1234
1806
  removed: summary.removed,
1235
1807
  refreshed: summary.refreshed,
@@ -1246,10 +1818,11 @@ var McpExtension = class {
1246
1818
  await this.reconcileAndRecordMtime({
1247
1819
  pi,
1248
1820
  configPath,
1249
- reason: "session_start"
1821
+ reason: "session_start",
1822
+ contextWindow: ctx.model?.contextWindow
1250
1823
  });
1251
1824
  } catch (err) {
1252
- log$11.error({
1825
+ log$12.error({
1253
1826
  err,
1254
1827
  event: "mcp_reconcile_failed"
1255
1828
  }, "MCP reconcile failed");
@@ -1261,7 +1834,7 @@ var McpExtension = class {
1261
1834
  try {
1262
1835
  mtime = await readConfigMtimeMs(configPath);
1263
1836
  } catch (err) {
1264
- log$11.warn({
1837
+ log$12.warn({
1265
1838
  err,
1266
1839
  event: "mcp_mtime_check_failed"
1267
1840
  }, "mtime check on mcp.config.json failed");
@@ -1272,10 +1845,11 @@ var McpExtension = class {
1272
1845
  await this.reconcileAndRecordMtime({
1273
1846
  pi,
1274
1847
  configPath,
1275
- reason: "auto_reload"
1848
+ reason: "auto_reload",
1849
+ contextWindow: ctx.model?.contextWindow
1276
1850
  });
1277
1851
  } catch (err) {
1278
- log$11.error({
1852
+ log$12.error({
1279
1853
  err,
1280
1854
  event: "mcp_auto_reload_failed"
1281
1855
  }, "auto-reload after mcp.config.json change failed");
@@ -1292,7 +1866,8 @@ var McpExtension = class {
1292
1866
  const summary = await this.reconcileAndRecordMtime({
1293
1867
  pi,
1294
1868
  configPath,
1295
- reason: "tool"
1869
+ reason: "tool",
1870
+ contextWindow: ctx.model?.contextWindow
1296
1871
  });
1297
1872
  return {
1298
1873
  content: [{
@@ -1334,9 +1909,21 @@ function pendingAuthEntries(summary) {
1334
1909
  function timeoutEntries(summary) {
1335
1910
  return Object.entries(summary.servers).flatMap(([id, status]) => status.status === "timeout" ? [{
1336
1911
  id,
1337
- stderr: status.stderr
1912
+ stderr: status.stderr,
1913
+ diagnostics: status.diagnostics
1338
1914
  }] : []);
1339
1915
  }
1916
+ /**
1917
+ * One line describing what the connect probe learned about a timed-out
1918
+ * stdio child, so an empty stderr is no longer a dead end. Empty array
1919
+ * when the entry was re-surfaced without a fresh probe.
1920
+ */
1921
+ function timeoutDiagnosticLines(diagnostics) {
1922
+ if (diagnostics === null) return [];
1923
+ const alive = diagnostics.alive === null ? "unknown (no pid)" : diagnostics.alive ? "alive" : "DEAD";
1924
+ const spoke = diagnostics.stdoutMessages === 0 ? "no JSON-RPC received on stdout (initialize never answered — slow startup, or blocked e.g. on OAuth)" : `${diagnostics.stdoutMessages} JSON-RPC message(s) received on stdout (server spoke but the handshake stalled)`;
1925
+ return [` child process: ${alive}${diagnostics.pid !== null ? ` (pid ${diagnostics.pid})` : ""}; ${spoke}`];
1926
+ }
1340
1927
  function failedEntries(summary) {
1341
1928
  return Object.entries(summary.servers).flatMap(([id, status]) => status.status === "failed" ? [{
1342
1929
  id,
@@ -1344,6 +1931,13 @@ function failedEntries(summary) {
1344
1931
  stderr: status.stderr
1345
1932
  }] : []);
1346
1933
  }
1934
+ function appendTimeoutStderrBlock(lines, stderr) {
1935
+ if (!stderr) {
1936
+ lines.push(" stderr: (empty — the child wrote nothing to stderr)");
1937
+ return;
1938
+ }
1939
+ appendStderrBlock(lines, stderr);
1940
+ }
1347
1941
  function appendStderrBlock(lines, stderr) {
1348
1942
  if (!stderr) return;
1349
1943
  lines.push(" stderr:");
@@ -1351,9 +1945,29 @@ function appendStderrBlock(lines, stderr) {
1351
1945
  for (const line of stderr.trimEnd().split("\n")) lines.push(` ${line}`);
1352
1946
  lines.push(" ---");
1353
1947
  }
1948
+ /**
1949
+ * Human-readable lines describing any budget-driven truncation, shared by
1950
+ * `summaryText` (the reload_mcp tool output) and `formatMcpUpdateMessage` (the
1951
+ * synthetic continuation). Empty when nothing was dropped.
1952
+ */
1953
+ function truncationLines(summary) {
1954
+ if (summary.droppedTools <= 0) return [];
1955
+ const lines = [];
1956
+ const { maxTotalTools, maxToolsPerServer } = summary.limits;
1957
+ const totalCapHint = Number.isFinite(maxTotalTools) ? `${maxTotalTools}` : "unlimited";
1958
+ lines.push(` WARNING: ${summary.droppedTools} MCP tool(s) were NOT registered because a tool budget was hit (token budget: ~${summary.tokenBudget} tokens, used ~${summary.tokensUsed}; total-tool cap: ${totalCapHint}, per-server cap: ${Number.isFinite(maxToolsPerServer) ? maxToolsPerServer : "unlimited"}).`);
1959
+ lines.push(" Tool definitions are sent to the model on every prompt; too many (or too-large) tool schemas push the request over the limit, so they are budgeted against the model context window. Prune servers from mcp.config.json to bring the tool surface down.");
1960
+ for (const [id, count] of Object.entries(summary.toolCounts)) {
1961
+ if (count.droppedReason === null) continue;
1962
+ const why = count.droppedReason === "tokens" ? "token budget exhausted" : count.droppedReason === "total" ? "total-tool cap reached" : "per-server cap";
1963
+ lines.push(` - ${id}: registered ${count.registered}/${count.advertised} tools (~${count.tokens} tokens, ${why}).`);
1964
+ }
1965
+ return lines;
1966
+ }
1354
1967
  function summaryText(summary) {
1355
1968
  const lines = [];
1356
1969
  lines.push(`MCP reconcile complete: ${summary.totalTools} tool(s) live.`);
1970
+ lines.push(...truncationLines(summary));
1357
1971
  if (summary.added.length > 0) lines.push(` Added: ${summary.added.join(", ")}`);
1358
1972
  if (summary.refreshed.length > 0) lines.push(` Refreshed: ${summary.refreshed.join(", ")}`);
1359
1973
  if (summary.removed.length > 0) lines.push(` Removed: ${summary.removed.join(", ")}`);
@@ -1362,11 +1976,12 @@ function summaryText(summary) {
1362
1976
  lines.push(` run \`${cliHint}\` in the shell to authenticate.`);
1363
1977
  appendStderrBlock(lines, stderr);
1364
1978
  }
1365
- for (const { id, stderr } of timeoutEntries(summary)) {
1979
+ for (const { id, stderr, diagnostics } of timeoutEntries(summary)) {
1366
1980
  lines.push(` TIMED OUT ${id}:`);
1367
1981
  lines.push(` bridge spawned but did not complete initialize in time`);
1368
1982
  lines.push(` (usually mcp-remote mid-OAuth — see captured stderr).`);
1369
- appendStderrBlock(lines, stderr);
1983
+ lines.push(...timeoutDiagnosticLines(diagnostics));
1984
+ appendTimeoutStderrBlock(lines, stderr);
1370
1985
  }
1371
1986
  for (const { id, error, stderr } of failedEntries(summary)) {
1372
1987
  lines.push(` FAILED ${id}: ${error}`);
@@ -1379,7 +1994,7 @@ function summaryText(summary) {
1379
1994
  return lines.join("\n");
1380
1995
  }
1381
1996
  function summaryHasChanges(summary) {
1382
- return summary.added.length > 0 || summary.removed.length > 0 || summary.refreshed.length > 0 || summary.errors.length > 0;
1997
+ return summary.added.length > 0 || summary.removed.length > 0 || summary.refreshed.length > 0 || summary.errors.length > 0 || summary.droppedTools > 0;
1383
1998
  }
1384
1999
  /**
1385
2000
  * Format a queued tool-update as a synthetic system-style message for
@@ -1392,6 +2007,10 @@ function formatMcpUpdateMessage(summary) {
1392
2007
  if (summary.added.length > 0) lines.push(`Newly available servers: ${summary.added.join(", ")}`);
1393
2008
  if (summary.refreshed.length > 0) lines.push(`Refreshed servers: ${summary.refreshed.join(", ")}`);
1394
2009
  if (summary.removed.length > 0) lines.push(`Removed servers (and their tools): ${summary.removed.join(", ")}`);
2010
+ if (summary.droppedTools > 0) {
2011
+ lines.push("");
2012
+ lines.push(...truncationLines(summary));
2013
+ }
1395
2014
  const pending = pendingAuthEntries(summary);
1396
2015
  if (pending.length > 0) {
1397
2016
  lines.push("");
@@ -1407,9 +2026,10 @@ function formatMcpUpdateMessage(summary) {
1407
2026
  if (timedOut.length > 0) {
1408
2027
  lines.push("");
1409
2028
  lines.push("Servers that timed out during initialize (bridge is still alive in the background; usually mcp-remote-style stdio bridges mid-OAuth):");
1410
- for (const { id, stderr } of timedOut) {
2029
+ for (const { id, stderr, diagnostics } of timedOut) {
1411
2030
  lines.push(` - ${id}:`);
1412
- appendStderrBlock(lines, stderr);
2031
+ lines.push(...timeoutDiagnosticLines(diagnostics));
2032
+ appendTimeoutStderrBlock(lines, stderr);
1413
2033
  }
1414
2034
  lines.push("If the bridge prints an auth URL in its stderr, share it with the user. Then call `reload_mcp` once they've finished.");
1415
2035
  }
@@ -1638,16 +2258,380 @@ const bashDefaultTimeoutExtension = (pi) => {
1638
2258
  });
1639
2259
  };
1640
2260
  //#endregion
2261
+ //#region src/extensions/resource-pressure-warning.ts
2262
+ /**
2263
+ * Mid-run resource-pressure warning to the agent.
2264
+ *
2265
+ * The sandbox already detects pressure — the boot scripts cap the
2266
+ * user-workload cgroup (memory.high/memory.max) and watchers log warn/crit
2267
+ * edges for memory and disk — but nothing told the *agent*, so a turn burned
2268
+ * straight to the OOM kill (or a full disk) and only learned about it from
2269
+ * the post-mortem notice. This extension closes that gap in-process: while a
2270
+ * turn is active it polls the agent cgroup and the root filesystem and, the
2271
+ * first time usage crosses a warn threshold, folds a system notification into
2272
+ * the open turn so the agent can checkpoint, shed work (constrain
2273
+ * parallelism, kill a background hog, clean scratch space), or request a
2274
+ * bigger tier BEFORE the kill.
2275
+ *
2276
+ * The notification is triggered by the two conditions that actually kill
2277
+ * work — memory near the cgroup hard cap, disk near full — and reports a
2278
+ * snapshot of all the relevant stats (memory, CPU utilization, disk) so the
2279
+ * agent can tell which resource is the problem and how much headroom the
2280
+ * others have.
2281
+ *
2282
+ * Edge-triggered, once per trigger per turn: the fired flags reset on
2283
+ * agent_start, so a turn that rides a threshold gets one warning per
2284
+ * resource, not a stream. Polling only runs while the agent is active — an
2285
+ * idle sandbox's resource usage is not the agent's problem and there is no
2286
+ * open turn to deliver into anyway.
2287
+ *
2288
+ * Best-effort throughout: any read failure (cgroup absent, controller not
2289
+ * delegated, non-cgroup-v2 host, df missing) reads as "no signal" for that
2290
+ * stat and the extension warns on what it can see — it must never break a
2291
+ * turn over an observability feature.
2292
+ */
2293
+ const execFileAsync = promisify(execFile);
2294
+ const log$11 = logger.child({ module: "resource-pressure-warning" });
2295
+ const POLL_INTERVAL_MS = 1e4;
2296
+ function envOverride(name) {
2297
+ for (const prefix of ["SKYDIVE_", "ANYONE_"]) {
2298
+ const value = process.env[`${prefix}${name}`];
2299
+ if (value != null && value !== "") return value;
2300
+ }
2301
+ return null;
2302
+ }
2303
+ function cgroupDir() {
2304
+ return envOverride("AGENT_CGROUP") ?? "/sys/fs/cgroup/agent";
2305
+ }
2306
+ function diskRoot() {
2307
+ return envOverride("DISK_ROOT") ?? "/";
2308
+ }
2309
+ /**
2310
+ * Read a cgroup v2 scalar file. Returns a number, or null for "max"
2311
+ * (uncapped), an empty/absent file, or any read/parse error — an uncapped or
2312
+ * unreadable limit means there is nothing meaningful to warn against.
2313
+ */
2314
+ async function readScalar(file) {
2315
+ try {
2316
+ const raw = (await readFile(`${cgroupDir()}/${file}`, "utf8")).trim();
2317
+ if (raw === "" || raw === "max") return null;
2318
+ const n = Number(raw);
2319
+ return Number.isFinite(n) ? n : null;
2320
+ } catch (_error) {
2321
+ return null;
2322
+ }
2323
+ }
2324
+ /**
2325
+ * Read a cgroup v2 "flat keyed" file (one `key value` pair per line, e.g.
2326
+ * cpu.stat) and return the counter for `key`, or null when absent.
2327
+ */
2328
+ async function readKeyedCounter(file, key) {
2329
+ try {
2330
+ const raw = await readFile(`${cgroupDir()}/${file}`, "utf8");
2331
+ for (const line of raw.split("\n")) {
2332
+ const [k, v] = line.trim().split(/\s+/);
2333
+ if (k === key) {
2334
+ const n = Number(v);
2335
+ return Number.isFinite(n) ? n : null;
2336
+ }
2337
+ }
2338
+ return null;
2339
+ } catch (_error) {
2340
+ return null;
2341
+ }
2342
+ }
2343
+ /**
2344
+ * Live memory usage as an integer percent of the hard cap, or null when
2345
+ * either side is unreadable/uncapped. Exported for tests.
2346
+ */
2347
+ async function readMemUsePct() {
2348
+ const [current, max] = await Promise.all([readScalar("memory.current"), readScalar("memory.max")]);
2349
+ if (current === null || max === null || max <= 0) return null;
2350
+ return {
2351
+ pct: Math.floor(current / max * 100),
2352
+ currentBytes: current,
2353
+ maxBytes: max
2354
+ };
2355
+ }
2356
+ /**
2357
+ * Root filesystem used% (df -P Capacity column), or null on any failure.
2358
+ * Exported for tests.
2359
+ */
2360
+ async function readDiskUsePct() {
2361
+ try {
2362
+ const { stdout } = await execFileAsync("df", ["-P", diskRoot()]);
2363
+ const dataRow = stdout.trim().split("\n")[1];
2364
+ if (dataRow == null) return null;
2365
+ const capacity = dataRow.trim().split(/\s+/)[4];
2366
+ if (capacity == null) return null;
2367
+ const pct = Number(capacity.replace("%", ""));
2368
+ return Number.isFinite(pct) ? pct : null;
2369
+ } catch (_error) {
2370
+ return null;
2371
+ }
2372
+ }
2373
+ /**
2374
+ * CPU utilization sampler. cgroup v2 exposes cumulative CPU time
2375
+ * (cpu.stat usage_usec); utilization is the delta between two samples over
2376
+ * the wall time between them, normalized by core count. The first call after
2377
+ * construction has no previous sample and returns null.
2378
+ */
2379
+ function createCpuSampler() {
2380
+ let prevUsageUsec = null;
2381
+ let prevAtMs = null;
2382
+ return async () => {
2383
+ const usage = await readKeyedCounter("cpu.stat", "usage_usec");
2384
+ const now = Date.now();
2385
+ const prev = prevUsageUsec;
2386
+ const prevAt = prevAtMs;
2387
+ prevUsageUsec = usage;
2388
+ prevAtMs = now;
2389
+ if (usage === null || prev === null || prevAt === null) return null;
2390
+ const wallUsec = (now - prevAt) * 1e3;
2391
+ if (wallUsec <= 0) return null;
2392
+ const cores = availableParallelism();
2393
+ const pct = Math.round((usage - prev) / (wallUsec * cores) * 100);
2394
+ return Math.max(0, Math.min(100, pct));
2395
+ };
2396
+ }
2397
+ function fmtMb(bytes) {
2398
+ return Math.round(bytes / 1024 / 1024);
2399
+ }
2400
+ /** The model-facing warning text. Exported for tests. */
2401
+ function resourcePressureWarningText(trigger, { mem, cpuPct, diskPct }) {
2402
+ const stats = [];
2403
+ if (mem) stats.push(`memory ${mem.pct}% of cap (${fmtMb(mem.currentBytes)}/${fmtMb(mem.maxBytes)} MB)`);
2404
+ if (cpuPct !== null) stats.push(`CPU ${cpuPct}%`);
2405
+ if (diskPct !== null) stats.push(`disk ${diskPct}% full`);
2406
+ const lead = trigger === "memory" ? `Your sandbox is at ${mem?.pct}% of its memory cap. If usage keeps climbing, the kernel will kill the offending process and this turn may die with it.` : `Your sandbox's disk is ${diskPct}% full. If it fills completely, writes will start failing and this turn may die with them.`;
2407
+ const remedy = trigger === "memory" ? "checkpoint in-flight work (commit and push), then reduce the footprint — constrain parallelism, run heavy steps sequentially, or kill background processes you no longer need." : "checkpoint in-flight work (commit and push), then free space — clean build artifacts, caches, and scratch files you no longer need.";
2408
+ return `<system_notification>${lead} Current usage: ${stats.join(", ")}. Act now: ${remedy} If the workload genuinely needs more resources, request a bigger sandbox with \`platform compute request\`. This is an automated resource warning, not a message from the user; continue the task, adjusted.</system_notification>`;
2409
+ }
2410
+ const resourcePressureWarningExtension = (pi) => {
2411
+ let agentActive = false;
2412
+ let warnedMemThisTurn = false;
2413
+ let warnedDiskThisTurn = false;
2414
+ let timer = null;
2415
+ const sampleCpu = createCpuSampler();
2416
+ async function checkOnce() {
2417
+ if (!agentActive || warnedMemThisTurn && warnedDiskThisTurn) return;
2418
+ const [mem, cpuPct, diskPct] = await Promise.all([
2419
+ readMemUsePct(),
2420
+ sampleCpu(),
2421
+ readDiskUsePct()
2422
+ ]);
2423
+ let trigger = null;
2424
+ if (!warnedMemThisTurn && mem !== null && mem.pct >= 80) {
2425
+ trigger = "memory";
2426
+ warnedMemThisTurn = true;
2427
+ } else if (!warnedDiskThisTurn && diskPct !== null && diskPct >= 80) {
2428
+ trigger = "disk";
2429
+ warnedDiskThisTurn = true;
2430
+ }
2431
+ if (trigger === null) return;
2432
+ log$11.warn({
2433
+ trigger,
2434
+ mem,
2435
+ cpuPct,
2436
+ diskPct
2437
+ }, "resource pressure warning delivered to agent");
2438
+ await pi.sendMessage({
2439
+ customType: "anyone-resource-pressure-warning",
2440
+ content: resourcePressureWarningText(trigger, {
2441
+ mem,
2442
+ cpuPct,
2443
+ diskPct
2444
+ }),
2445
+ display: false
2446
+ }, {
2447
+ triggerTurn: true,
2448
+ deliverAs: "followUp"
2449
+ });
2450
+ }
2451
+ pi.on("agent_start", async () => {
2452
+ agentActive = true;
2453
+ warnedMemThisTurn = false;
2454
+ warnedDiskThisTurn = false;
2455
+ if (!timer) {
2456
+ timer = setInterval(() => {
2457
+ checkOnce().catch((err) => {
2458
+ log$11.error({ err }, "resource pressure check failed");
2459
+ });
2460
+ }, POLL_INTERVAL_MS);
2461
+ timer.unref?.();
2462
+ }
2463
+ });
2464
+ pi.on("agent_end", async () => {
2465
+ agentActive = false;
2466
+ if (timer) {
2467
+ clearInterval(timer);
2468
+ timer = null;
2469
+ }
2470
+ });
2471
+ };
2472
+ //#endregion
2473
+ //#region src/extensions/disk-guard.ts
2474
+ const log$10 = logger.child({ module: "disk-guard" });
2475
+ /**
2476
+ * In-band bypass. The guard is a safety net, not a jail: when the agent knows
2477
+ * a flagged command is genuinely safe (writing to a different mount, a tiny
2478
+ * bounded download, a delete-then-clone one-liner, an emergency it accepts the
2479
+ * risk on) it can force the command through by appending this marker as a
2480
+ * trailing shell comment. Kept as a comment so it never changes what the
2481
+ * command does, and matched case-insensitively with flexible spacing so the
2482
+ * agent doesn't have to reproduce it byte-for-byte.
2483
+ */
2484
+ const BYPASS_MARKER = /#\s*disk-guard:\s*allow\b/i;
2485
+ /** The exact marker text the block message tells the agent to append. */
2486
+ const BYPASS_HINT = "# disk-guard: allow";
2487
+ /**
2488
+ * Harness-level kill switch: set DISK_GUARD_DISABLE=1 to turn the guard off
2489
+ * entirely. This is the "I own my harness, let me opt out" knob — an agent
2490
+ * that boots its own harness can disable the guard for its whole process
2491
+ * without a code roll, and it's also the fleet-wide escape hatch if the
2492
+ * classifier ever misfires and blocks real work. The bare name is honored
2493
+ * first; the SKYDIVE_/ANYONE_ prefixes are accepted too for consistency with
2494
+ * the other env overrides. Empty/unset/"0"/"false" leave the guard on.
2495
+ */
2496
+ function guardDisabledByEnv() {
2497
+ for (const name of [
2498
+ "DISK_GUARD_DISABLE",
2499
+ "SKYDIVE_DISK_GUARD_DISABLE",
2500
+ "ANYONE_DISK_GUARD_DISABLE"
2501
+ ]) {
2502
+ const value = process.env[name];
2503
+ if (value != null && value !== "" && value !== "0" && value !== "false") return true;
2504
+ }
2505
+ return false;
2506
+ }
2507
+ /** True when the command carries the in-band bypass marker. */
2508
+ function hasBypassMarker(command) {
2509
+ return BYPASS_MARKER.test(command);
2510
+ }
2511
+ /**
2512
+ * Commands that reclaim space or merely inspect it. If any of these verbs
2513
+ * appears in the command line, we never block — otherwise the guard would trap
2514
+ * the agent by blocking the exact command it needs to dig out. Matched as
2515
+ * whole words so `remove-item` etc. don't accidentally match `rm`.
2516
+ */
2517
+ const RECLAIM_PATTERNS = [
2518
+ /\brm\b/,
2519
+ /\brmdir\b/,
2520
+ /\bdf\b/,
2521
+ /\bdu\b/,
2522
+ /\bncdu\b/,
2523
+ /\bfind\b[^|]*\s-delete\b/,
2524
+ /\btruncate\b/,
2525
+ /\bgit\s+(gc|prune|clean|worktree\s+remove|worktree\s+prune)\b/,
2526
+ /\b(yarn|npm|pnpm|bun)\s+.*\b(cache\s+clean|cache\s+clear|store\s+prune)\b/,
2527
+ /\bcache\s+(clean|clear|prune)\b/,
2528
+ /\b(docker|podman)\s+.*\bprune\b/,
2529
+ /\bapt(-get)?\s+clean\b/,
2530
+ /\bjournalctl\b[^|]*--vacuum/
2531
+ ];
2532
+ /**
2533
+ * File extensions that mean a download is actually LARGE — archives, disk
2534
+ * images, compiled/binary artifacts, model weights, media. A curl/wget is only
2535
+ * gated when it writes one of these; an API/page fetch to a `.json`/`.html`/
2536
+ * `.txt` file is tiny and must not be blocked. Derived from 4,144 real
2537
+ * commands: ~64% of `curl -o` uses were tiny fetches, only ~4% large.
2538
+ */
2539
+ const BIG_DOWNLOAD_EXT = "(?:tar\\.gz|tgz|tar|zip|iso|gz|bz2|xz|zst|deb|rpm|pkg|dmg|whl|jar|7z|img|mp4|mov|avi|mkv|onnx|gguf|safetensors|bin|node)";
2540
+ /**
2541
+ * Commands that consume a meaningful amount of disk. Kept deliberately tight
2542
+ * and high-precision: validated against 4,144 real commands from the last 7
2543
+ * days, the earlier "writes a file" heuristic flagged 82% of everything (a
2544
+ * `curl -o /tmp/x.json` API call is not a disk event). This set flags ~33%,
2545
+ * almost all genuinely large — real installs, clones, big-archive downloads,
2546
+ * extractions. What was DROPPED and why:
2547
+ * - `git fetch` / `git pull` — incremental on an existing clone, usually tiny.
2548
+ * - `git checkout` — overwhelmingly `git checkout <ref> -- <file>` or a
2549
+ * branch switch, ~zero net growth; the rare full materialization isn't
2550
+ * worth the false-positive rate.
2551
+ * - bare `curl -o` / `wget -o` — see BIG_DOWNLOAD_EXT above.
2552
+ * - loose `… build` — matched `--mode=skip-build`, `oxfmt … build`, prose.
2553
+ * The remaining big-disk op in escher is `git clone` and `git worktree add`
2554
+ * (which is really a checkout), both kept.
2555
+ */
2556
+ const SPACE_HUNGRY_PATTERNS = [
2557
+ /\bgit\s+clone\b/,
2558
+ /\bgit\s+worktree\s+add\b/,
2559
+ /\b(yarn|npm|pnpm|bun)\s+(install|add|ci)\b/,
2560
+ /\byarn\s*$/,
2561
+ /\byarn\s+--(?!version|help)\S/,
2562
+ /\bpip3?\s+install\b/,
2563
+ /\bapt(-get)?\s+install\b/,
2564
+ /\bnpm\s+pack\b/,
2565
+ /\bdocker\s+(build|pull)\b/,
2566
+ new RegExp(`\\b(?:curl|wget)\\b[^\\n]*\\s-[a-zA-Z]*[oO]\\b[^\\n]*\\.${BIG_DOWNLOAD_EXT}\\b`, "i"),
2567
+ new RegExp(`\\b(?:curl|wget)\\b[^\\n]*\\.${BIG_DOWNLOAD_EXT}\\b[^\\n]*\\s-[a-zA-Z]*[oO]\\b`, "i"),
2568
+ /\btar\s+[^\n|]*x[^\n|]*f/,
2569
+ /\bunzip\b/,
2570
+ /\bdd\b[^\n|]*\bof=/
2571
+ ];
2572
+ /**
2573
+ * True when the command reclaims or inspects space — these are always allowed,
2574
+ * even on a 100%-full box, so the agent can dig itself out.
2575
+ */
2576
+ function isReclaimCommand(command) {
2577
+ return RECLAIM_PATTERNS.some((re) => re.test(command));
2578
+ }
2579
+ /**
2580
+ * True when the command is likely to consume a meaningful amount of disk.
2581
+ * A reclaim/inspect command is never space-hungry — the reclaim check wins so a
2582
+ * `git worktree remove` or a `yarn cache clean` is never mistaken for growth.
2583
+ */
2584
+ function isSpaceHungryCommand(command) {
2585
+ if (isReclaimCommand(command)) return false;
2586
+ return SPACE_HUNGRY_PATTERNS.some((re) => re.test(command));
2587
+ }
2588
+ /**
2589
+ * The decision, factored out and pure so it's exhaustively testable without a
2590
+ * real filesystem. Block only when we have a disk reading, it's at/above the
2591
+ * critical threshold, the command is space-hungry (and not a reclaim), and the
2592
+ * agent hasn't explicitly opted out with the bypass marker.
2593
+ */
2594
+ function shouldBlockForDisk(command, diskPct) {
2595
+ if (diskPct === null) return false;
2596
+ if (diskPct < 95) return false;
2597
+ if (hasBypassMarker(command)) return false;
2598
+ return isSpaceHungryCommand(command);
2599
+ }
2600
+ /** The agent-facing explanation returned as the blocked tool result. */
2601
+ function diskBlockReason(command, diskPct) {
2602
+ return `Blocked: the sandbox disk is ${diskPct}% full and this command (\`${command.trim().slice(0, 120)}\`) writes a large amount, so it would fail partway with ENOSPC and leave a corrupt result. Reclaim space FIRST, then retry. Free ONLY what THIS conversation created — scratch/build output you wrote this run, downloads you're done with, and worktrees/branches whose work you've already committed and pushed (\`git worktree remove\`, \`yarn cache clean\`, delete your own scratch). Do NOT blindly wipe /tmp or delete a clone/worktree you don't recognize — other conversations share this box. Check headroom with \`df -h /\` and \`du -sh ~/workspace/* 2>/dev/null\`. If you genuinely can't free enough, stop and tell the user you're blocked on disk rather than retrying the write. If you're certain this command is safe anyway (writes elsewhere, tiny bounded size, delete-then-write), force it through by appending \` ${BYPASS_HINT}\` to the command.`;
2603
+ }
2604
+ const diskGuardExtension = (pi) => {
2605
+ pi.on("tool_call", async (event) => {
2606
+ if (event.toolName !== "bash") return;
2607
+ if (guardDisabledByEnv()) return;
2608
+ const command = event.input.command;
2609
+ if (typeof command !== "string" || command.length === 0) return;
2610
+ if (hasBypassMarker(command)) return;
2611
+ if (!isSpaceHungryCommand(command)) return;
2612
+ const diskPct = await readDiskUsePct();
2613
+ if (!shouldBlockForDisk(command, diskPct)) return;
2614
+ log$10.warn({
2615
+ diskPct,
2616
+ command: command.slice(0, 200)
2617
+ }, "blocked space-hungry bash command on near-full disk");
2618
+ return {
2619
+ block: true,
2620
+ reason: diskBlockReason(command, diskPct)
2621
+ };
2622
+ });
2623
+ };
2624
+ //#endregion
1641
2625
  //#region src/channel-context-ref.ts
1642
2626
  /**
1643
- * The worker injects only a reference — `{ channel, messageId }` — into the
1644
- * sandbox env (`SKYDIVE_CHANNEL_CONTEXT`) rather than the full context.
2627
+ * The worker injects a small reference — `{ channel, messageId, runId }` —
2628
+ * into the sandbox env (`SKYDIVE_CHANNEL_CONTEXT`) rather than the full context.
1645
2629
  *
1646
2630
  * The canonical `ChannelContextRef` type + `parseChannelContextRef` live in
1647
2631
  * `@createinc/anyone-channels`, but the harness (`@skydiveai/*`) keeps zero
1648
2632
  * `@createinc/*` dependencies — importing that package would pull the whole
1649
- * platform channel stack (Slack/email/Linq SDKs, messaging) in just to read
1650
- * two fields. So we validate the (stable) shape locally instead.
2633
+ * platform channel stack (Slack/email/Linq SDKs, messaging) just to read one
2634
+ * field. So we validate the field this consumer needs locally instead.
1651
2635
  */
1652
2636
  const channelContextRefSchema = z.object({ messageId: z.string().nullable() });
1653
2637
  /**
@@ -1679,20 +2663,84 @@ function apiBaseUrl() {
1679
2663
  }
1680
2664
  //#endregion
1681
2665
  //#region src/extensions/platform.ts
2666
+ /**
2667
+ * Platform extension — bridges the agent harness to the Skydive platform daemon.
2668
+ *
2669
+ * Responsibilities:
2670
+ * - Heartbeat: periodic POST to the API so the sandbox manager knows the
2671
+ * agent is alive. Throttled to once per minute, triggered by tool events.
2672
+ * - Session tracking: registers the session with the daemon on start,
2673
+ * streams tool_call / tool_result events so the daemon can track which
2674
+ * session is actively executing, and signals session end on agent_end.
2675
+ * - Channel context: passes the SKYDIVE_CHANNEL_CONTEXT (containing the
2676
+ * messageId) to the daemon so file writes can be attributed to the
2677
+ * correct conversation.
2678
+ *
2679
+ * All daemon POSTs are fire-and-forget — failures are logged but never
2680
+ * block the agent. The daemon may not be running (e.g. local dev without
2681
+ * a sandbox), and that's fine.
2682
+ */
1682
2683
  const HEARTBEAT_THROTTLE_MS = 6e4;
1683
2684
  const TOOL_HEARTBEAT_INTERVAL_MS = 5e3;
1684
2685
  const MAX_TOOL_HEARTBEATS = 1440 * 60 * 1e3 / TOOL_HEARTBEAT_INTERVAL_MS;
1685
2686
  const DAEMON_URL = "http://localhost:38994";
1686
- const log$10 = logger.child({ module: "platform-ext" });
2687
+ const log$9 = logger.child({ module: "platform-ext" });
1687
2688
  function sandboxClient() {
1688
2689
  const apiUrl = apiBaseUrl();
1689
2690
  if (!apiUrl) return null;
1690
2691
  return hc(`${apiUrl}/api/v1/sandbox`);
1691
2692
  }
1692
2693
  /**
1693
- * Fetch every harness feature flag in one GET (`{ contextManagement, subagent,
1694
- * ... }` — see apps/anyone/api/src/routes/sandbox-feature-flags.ts). Returns
1695
- * null when indeterminate (no api url, or the request failed) so the shared
2694
+ * Is this box still an unclaimed warm-pool sandbox? (ANY-6000, the
2695
+ * feature-flags half of the ANY-5184 pool 403 wave.)
2696
+ *
2697
+ * `GET /sandbox/feature-flags` is agent-only, so the shared poller's request
2698
+ * from a pool box can only 403 — a guaranteed-failing GET every 60s for the
2699
+ * life of the pool phase. The discriminator is the sandbox token's `type`
2700
+ * claim, read UNVERIFIED (this box never holds the signing secret): not an
2701
+ * authorization decision, only "should I bother calling?", and the api still
2702
+ * authorizes every request.
2703
+ *
2704
+ * Read per call from the daemon's persisted env file, NOT process.env:
2705
+ * claiming a pool box rebinds the token in place (the daemon rewrites this
2706
+ * file) while the harness's process.env keeps the boot snapshot, so a
2707
+ * process-env gate would leave a claimed box permanently skipping — trading a
2708
+ * wasted request for silently frozen flags, which is strictly worse. "Cannot
2709
+ * tell" (no file, no token, unparseable payload) reports false so the poll
2710
+ * proceeds.
2711
+ */
2712
+ const daemonEnvIdentitySchema = z.object({
2713
+ ANYONE_SANDBOX_TOKEN: z.string().optional(),
2714
+ SKYDIVE_SANDBOX_TOKEN: z.string().optional()
2715
+ }).passthrough();
2716
+ const tokenTypeSchema = z.object({ type: z.string() }).passthrough();
2717
+ async function isPoolIdentity() {
2718
+ try {
2719
+ const override = process.env.ANYONE_DAEMON_ENV_CACHE;
2720
+ const candidates = override ? [override] : ["/run/anyone-system/daemon-env.json", "/tmp/.anyone/daemon-env.json"];
2721
+ let raw = null;
2722
+ for (const file of candidates) {
2723
+ raw = await readFile(file, "utf8").catch(() => null);
2724
+ if (raw !== null) break;
2725
+ }
2726
+ if (raw === null) return false;
2727
+ const env = daemonEnvIdentitySchema.safeParse(JSON.parse(raw));
2728
+ if (!env.success) return false;
2729
+ const token = env.data.ANYONE_SANDBOX_TOKEN ?? env.data.SKYDIVE_SANDBOX_TOKEN;
2730
+ if (typeof token !== "string" || token === "") return false;
2731
+ const payload = token.split(".")[1];
2732
+ if (!payload) return false;
2733
+ const claims = tokenTypeSchema.safeParse(JSON.parse(Buffer.from(payload, "base64url").toString("utf8")));
2734
+ return claims.success && claims.data.type === "onboarding-pool";
2735
+ } catch (_err) {
2736
+ return false;
2737
+ }
2738
+ }
2739
+ /**
2740
+ * Fetch every harness feature flag in one GET (`{ contextManagement, ... }`
2741
+ * — see apps/anyone/api/src/routes/sandbox-feature-flags.ts). Returns
2742
+ * null when indeterminate (no api url, the request failed, or the box is an
2743
+ * unclaimed pool sandbox whose token the route would 403) so the shared
1696
2744
  * poller keeps the last-known values rather than flipping on a transient error.
1697
2745
  * This is the single fetch behind `feature-flags-poll.ts`; extensions read the
1698
2746
  * polled values there instead of issuing their own GET.
@@ -1700,22 +2748,19 @@ function sandboxClient() {
1700
2748
  async function fetchHarnessFlags() {
1701
2749
  const client = sandboxClient();
1702
2750
  if (!client) return null;
2751
+ if (await isPoolIdentity()) return null;
1703
2752
  try {
1704
2753
  const res = await client["feature-flags"].$get();
1705
2754
  if (!res.ok) {
1706
- log$10.debug({
2755
+ log$9.debug({
1707
2756
  status: res.status,
1708
2757
  event: "feature_flags_fetch_failed"
1709
2758
  }, "feature-flags fetch failed");
1710
2759
  return null;
1711
2760
  }
1712
- const body = await res.json();
1713
- return {
1714
- contextManagement: body.contextManagement ?? null,
1715
- subagent: body.subagent ?? null
1716
- };
2761
+ return { contextManagement: (await res.json()).contextManagement ?? null };
1717
2762
  } catch (err) {
1718
- log$10.debug({
2763
+ log$9.debug({
1719
2764
  err,
1720
2765
  event: "feature_flags_fetch_error"
1721
2766
  }, "feature-flags request errored");
@@ -1726,7 +2771,7 @@ function postHeartbeat({ messageId }) {
1726
2771
  const client = sandboxClient();
1727
2772
  if (!client) return;
1728
2773
  client.heartbeat.$post({ json: { messageId } }).catch((err) => {
1729
- log$10.debug({
2774
+ log$9.debug({
1730
2775
  err,
1731
2776
  event: "heartbeat_failed"
1732
2777
  }, "heartbeat failed");
@@ -1738,7 +2783,7 @@ async function resolveConversationFromApi(messageId) {
1738
2783
  try {
1739
2784
  const res = await client["message-conversation"].$get({ query: { messageId } });
1740
2785
  if (!res.ok) {
1741
- log$10.warn({
2786
+ log$9.warn({
1742
2787
  status: res.status,
1743
2788
  messageId,
1744
2789
  event: "resolve_conversation_failed"
@@ -1747,7 +2792,7 @@ async function resolveConversationFromApi(messageId) {
1747
2792
  }
1748
2793
  return (await res.json()).conversationId ?? null;
1749
2794
  } catch (err) {
1750
- log$10.warn({
2795
+ log$9.warn({
1751
2796
  err,
1752
2797
  messageId,
1753
2798
  event: "resolve_conversation_error"
@@ -1764,6 +2809,71 @@ async function postBackgroundTaskDone({ messageId, content }) {
1764
2809
  } });
1765
2810
  if (!res.ok) throw new Error(`bg-task-done POST failed: ${res.status}`);
1766
2811
  }
2812
+ async function putBackgroundTaskJournalSpec({ messageId, spec }) {
2813
+ const client = sandboxClient();
2814
+ if (!client) return;
2815
+ try {
2816
+ const res = await client["bg-task-journal"].$put({ json: {
2817
+ messageId,
2818
+ spec
2819
+ } });
2820
+ if (!res.ok) log$9.warn({
2821
+ status: res.status,
2822
+ taskId: spec.id,
2823
+ event: "bg_journal_put_failed"
2824
+ }, "bg-task journal PUT failed");
2825
+ } catch (err) {
2826
+ log$9.warn({
2827
+ err,
2828
+ taskId: spec.id,
2829
+ event: "bg_journal_put_failed"
2830
+ }, "bg-task journal PUT threw");
2831
+ }
2832
+ }
2833
+ async function deleteBackgroundTaskJournalSpec({ messageId, taskId }) {
2834
+ const client = sandboxClient();
2835
+ if (!client) return;
2836
+ try {
2837
+ await client["bg-task-journal"].$delete({ json: {
2838
+ messageId,
2839
+ taskId
2840
+ } });
2841
+ } catch (err) {
2842
+ log$9.debug({
2843
+ err,
2844
+ taskId,
2845
+ event: "bg_journal_delete_failed"
2846
+ }, "bg-task journal DELETE failed");
2847
+ }
2848
+ }
2849
+ async function listBackgroundTaskJournalSpecs({ messageId }) {
2850
+ const client = sandboxClient();
2851
+ if (!client) return [];
2852
+ try {
2853
+ const res = await client["bg-task-journal"].$get({ query: { messageId } });
2854
+ if (!res.ok) return [];
2855
+ return (await res.json()).specs ?? [];
2856
+ } catch (err) {
2857
+ log$9.debug({
2858
+ err,
2859
+ event: "bg_journal_list_failed"
2860
+ }, "bg-task journal GET failed");
2861
+ return [];
2862
+ }
2863
+ }
2864
+ function postBackgroundTasksSnapshot({ messageId, tasks }) {
2865
+ const client = sandboxClient();
2866
+ if (!client || !messageId) return;
2867
+ client["bg-tasks"].$post({ json: {
2868
+ messageId,
2869
+ tasks
2870
+ } }).catch((err) => {
2871
+ log$9.debug({
2872
+ err,
2873
+ event: "bg_tasks_snapshot_failed"
2874
+ }, "bg-tasks snapshot publish failed");
2875
+ });
2876
+ }
1767
2877
  async function postSubagentSpawn({ messageId, tasks }) {
1768
2878
  const client = sandboxClient();
1769
2879
  if (!client) throw new Error("no api url for subagent-spawn");
@@ -1771,8 +2881,19 @@ async function postSubagentSpawn({ messageId, tasks }) {
1771
2881
  messageId,
1772
2882
  tasks
1773
2883
  } });
1774
- if (!res.ok) throw new Error(`subagent-spawn POST failed: ${res.status}`);
1775
- return { taskIds: (await res.json()).taskIds };
2884
+ if (!res.ok) {
2885
+ let detail = "";
2886
+ try {
2887
+ const errBody = await res.json();
2888
+ if (errBody && typeof errBody.error === "string") detail = `: ${errBody.error}`;
2889
+ } catch {}
2890
+ throw new Error(`subagent-spawn POST failed (${res.status})${detail}`);
2891
+ }
2892
+ const body = await res.json();
2893
+ return {
2894
+ taskIds: body.taskIds,
2895
+ tasks: body.tasks ?? []
2896
+ };
1776
2897
  }
1777
2898
  function createHeartbeatThrottle({ messageId }) {
1778
2899
  let lastAt = 0;
@@ -1821,7 +2942,7 @@ function createToolHeartbeat({ messageId }) {
1821
2942
  }
1822
2943
  heartbeatCount++;
1823
2944
  if (heartbeatCount > MAX_TOOL_HEARTBEATS) {
1824
- log$10.warn({
2945
+ log$9.warn({
1825
2946
  heartbeatCount,
1826
2947
  activeToolCalls: [...activeToolCalls]
1827
2948
  }, "tool heartbeat max reached, stopping");
@@ -1852,7 +2973,7 @@ function postToDaemon(path, body) {
1852
2973
  headers: { "content-type": "application/json" },
1853
2974
  body: JSON.stringify(body)
1854
2975
  }).catch((err) => {
1855
- log$10.debug({
2976
+ log$9.debug({
1856
2977
  err,
1857
2978
  path,
1858
2979
  event: "daemon_post_failed"
@@ -1861,7 +2982,7 @@ function postToDaemon(path, body) {
1861
2982
  }
1862
2983
  function createPlatformExtensions({ sessionId, channelContext }) {
1863
2984
  return (pi) => {
1864
- log$10.info({
2985
+ log$9.info({
1865
2986
  sessionId,
1866
2987
  hasChannelContext: Boolean(channelContext)
1867
2988
  }, "platform extension initialized");
@@ -1910,7 +3031,7 @@ function createPlatformExtensions({ sessionId, channelContext }) {
1910
3031
  });
1911
3032
  });
1912
3033
  pi.on("agent_end", () => {
1913
- log$10.info({ sessionId }, "session ending");
3034
+ log$9.info({ sessionId }, "session ending");
1914
3035
  postToDaemon("/session/end", { sessionId });
1915
3036
  });
1916
3037
  };
@@ -1921,37 +3042,39 @@ function createPlatformExtensions({ sessionId, channelContext }) {
1921
3042
  * Shared harness feature-flag poll.
1922
3043
  *
1923
3044
  * The api exposes one `/feature-flags` GET that returns every harness flag in a
1924
- * single response (`{ contextManagement, subagent, commandFlags }` — see
3045
+ * single response (`{ contextManagement, commandFlags }` — see
1925
3046
  * apps/anyone/api/src/routes/sandbox-feature-flags.ts). Rather than each
1926
3047
  * extension issuing its own GET — and, worse, a *blocking* GET on the
1927
3048
  * pre-first-token `session_start` path — a single background poller fetches
1928
3049
  * that response once per interval and fans the values out to every subscriber.
1929
3050
  *
1930
- * Why one poller: the subagent extension gates its tool registration on the
1931
- * `subagent` flag. If it awaited a fresh GET inside `session_start` the tool
1932
- * schema (part of the prefill) couldn't be finalized until a serial
1933
- * sandbox→api round-trip settled, adding a net-new pre-token network hop on
1934
- * every session, flag on or off. Reading the last-polled value instead keeps
1935
- * the hot path allocation-only. A cold cache reads as `null` (fail-open to
1936
- * unregistered); a newly-flipped flag takes effect on the next poll, matching
1937
- * how context-management already treats its flag.
3051
+ * Why one poller: context-management consumes the `contextManagement` flag
3052
+ * without a blocking GET on the pre-first-token `session_start` path. Reading
3053
+ * the last-polled value keeps the hot path allocation-only; a cold cache reads
3054
+ * as `null` and a newly-flipped flag takes effect on the next poll.
1938
3055
  *
1939
3056
  * The poll is fire-and-forget and self-unref'd — it never keeps the process
1940
3057
  * alive and an indeterminate result (no api url / transient failure) leaves the
1941
3058
  * last-known values untouched so a blip can't silently flip behavior.
1942
3059
  */
1943
- const log$9 = logger.child({ module: "feature-flags-poll" });
3060
+ const log$8 = logger.child({ module: "feature-flags-poll" });
1944
3061
  const FLAG_POLL_INTERVAL_MS = 6e4;
1945
3062
  let contextManagement = null;
1946
- let subagent = null;
1947
- const subscribers = {
1948
- contextManagement: /* @__PURE__ */ new Set(),
1949
- subagent: /* @__PURE__ */ new Set()
1950
- };
3063
+ const subscribers = { contextManagement: /* @__PURE__ */ new Set() };
1951
3064
  let pollerStarted = false;
3065
+ let firstPollSettled = false;
3066
+ let resolveFirstPoll = null;
3067
+ new Promise((resolve) => {
3068
+ resolveFirstPoll = resolve;
3069
+ });
3070
+ function markFirstPollSettled() {
3071
+ if (firstPollSettled) return;
3072
+ firstPollSettled = true;
3073
+ resolveFirstPoll?.();
3074
+ }
1952
3075
  /** Last-polled value of a flag, or `null` if not yet resolved. */
1953
- function getPolledFlag(name) {
1954
- return name === "contextManagement" ? contextManagement : subagent;
3076
+ function getPolledFlag(_name) {
3077
+ return contextManagement;
1955
3078
  }
1956
3079
  /**
1957
3080
  * Subscribe to changes of a flag. The callback fires only on a *transition*
@@ -1964,23 +3087,25 @@ function onFlagChange(name, cb) {
1964
3087
  }
1965
3088
  function apply(name, next) {
1966
3089
  if (next === null) return;
1967
- const prev = name === "contextManagement" ? contextManagement : subagent;
1968
- if (name === "contextManagement") contextManagement = next;
1969
- else subagent = next;
3090
+ const prev = contextManagement;
3091
+ contextManagement = next;
1970
3092
  if (next !== prev) for (const cb of subscribers[name]) try {
1971
3093
  cb(next);
1972
3094
  } catch (err) {
1973
- log$9.warn({
3095
+ log$8.warn({
1974
3096
  err,
1975
3097
  flag: name
1976
3098
  }, "flag subscriber threw");
1977
3099
  }
1978
3100
  }
1979
3101
  async function pollOnce() {
1980
- const flags = await fetchHarnessFlags();
1981
- if (!flags) return;
1982
- apply("contextManagement", flags.contextManagement ?? null);
1983
- apply("subagent", flags.subagent ?? null);
3102
+ try {
3103
+ const flags = await fetchHarnessFlags();
3104
+ if (!flags) return;
3105
+ apply("contextManagement", flags.contextManagement ?? null);
3106
+ } catch (err) {
3107
+ log$8.debug({ err }, "feature-flag poll threw");
3108
+ }
1984
3109
  }
1985
3110
  /**
1986
3111
  * Start the shared background poll (idempotent). No-op when there's no
@@ -1991,7 +3116,7 @@ async function pollOnce() {
1991
3116
  function startFeatureFlagPoller() {
1992
3117
  if (pollerStarted || !hasFlagSource()) return;
1993
3118
  pollerStarted = true;
1994
- pollOnce();
3119
+ pollOnce().finally(markFirstPollSettled);
1995
3120
  setInterval(() => void pollOnce(), FLAG_POLL_INTERVAL_MS).unref?.();
1996
3121
  }
1997
3122
  //#endregion
@@ -2086,7 +3211,7 @@ function transformContextMessages(messages, config, now) {
2086
3211
  }
2087
3212
  //#endregion
2088
3213
  //#region src/extensions/context-management.ts
2089
- const log$8 = logger.child({ module: "context-management-extension" });
3214
+ const log$7 = logger.child({ module: "context-management-extension" });
2090
3215
  function isAnthropicMessagesPayload(payload) {
2091
3216
  if (typeof payload !== "object" || payload === null) return false;
2092
3217
  const candidate = payload;
@@ -2151,13 +3276,13 @@ function createContextManagementExtension() {
2151
3276
  setContextManagementFlagOverride(getPolledFlag("contextManagement"));
2152
3277
  onFlagChange("contextManagement", (enabled) => {
2153
3278
  setContextManagementFlagOverride(enabled);
2154
- log$8.info({
3279
+ log$7.info({
2155
3280
  event: "context_management_flag_update",
2156
3281
  enabled
2157
3282
  }, "context-management flag updated from platform");
2158
3283
  });
2159
3284
  startFeatureFlagPoller();
2160
- log$8.info({
3285
+ log$7.info({
2161
3286
  event: "context_management_registered",
2162
3287
  enabled: initial.enabled,
2163
3288
  flagSource: hasFlagSource(),
@@ -2169,13 +3294,13 @@ function createContextManagementExtension() {
2169
3294
  const { messages } = event;
2170
3295
  try {
2171
3296
  const result = transformContextIfEnabled(messages, getContextManagementConfig(), Date.now());
2172
- if (result.stats && (result.stats.clearedResults > 0 || result.stats.trimmedResults > 0)) log$8.info({
3297
+ if (result.stats && (result.stats.clearedResults > 0 || result.stats.trimmedResults > 0)) log$7.info({
2173
3298
  event: "context_management_applied",
2174
3299
  ...result.stats
2175
3300
  }, "trimmed/cleared tool output before LLM call");
2176
3301
  return { messages: result.messages };
2177
3302
  } catch (err) {
2178
- log$8.error({
3303
+ log$7.error({
2179
3304
  err,
2180
3305
  event: "context_management_transform_failed"
2181
3306
  }, "context transform failed; passing messages through unchanged");
@@ -2187,7 +3312,7 @@ function createContextManagementExtension() {
2187
3312
  }
2188
3313
  //#endregion
2189
3314
  //#region src/extensions/current-time.ts
2190
- const log$7 = logger.child({ module: "current-time-extension" });
3315
+ const log$6 = logger.child({ module: "current-time-extension" });
2191
3316
  const PI_DATE_LINE = /^Current date:.*$/m;
2192
3317
  function formatCurrentTimeLine(now) {
2193
3318
  return `Current date: ${now.getUTCFullYear()}-${String(now.getUTCMonth() + 1).padStart(2, "0")}-${String(now.getUTCDate()).padStart(2, "0")} (${new Intl.DateTimeFormat("en-US", {
@@ -2200,7 +3325,7 @@ const currentTimeExtension = (pi) => {
2200
3325
  const line = formatCurrentTimeLine(/* @__PURE__ */ new Date());
2201
3326
  const base = event.systemPrompt;
2202
3327
  if (PI_DATE_LINE.test(base)) {
2203
- log$7.info({ event: "pi_date_line_present" }, "pi base prompt carries its own 'Current date:' line again; replacing it in place (pi prompt format may have changed)");
3328
+ log$6.info({ event: "pi_date_line_present" }, "pi base prompt carries its own 'Current date:' line again; replacing it in place (pi prompt format may have changed)");
2204
3329
  return { systemPrompt: base.replace(PI_DATE_LINE, line) };
2205
3330
  }
2206
3331
  return { systemPrompt: `${base}\n${line}` };
@@ -2209,37 +3334,35 @@ const currentTimeExtension = (pi) => {
2209
3334
  //#endregion
2210
3335
  //#region src/memory.ts
2211
3336
  /**
2212
- * In-harness memory index builder.
2213
- *
2214
- * The agent has an agent-level file-based memory at `<cwd>/.memory/`,
2215
- * organized by directory:
3337
+ * In-harness memory readers.
2216
3338
  *
2217
- * .memory/users/<id>-<name>/<topic>.md
2218
- * .memory/projects/<project_slug>/<topic>.md
2219
- * .memory/feedback/<topic>.md
2220
- * .memory/reference/<topic>.md
3339
+ * The agent has an agent-level file-based memory at `<cwd>/.memory/`, split
3340
+ * into two halves that are surfaced differently:
2221
3341
  *
2222
- * The path encodes type and subject (for `users/`, the subject is the
2223
- * person's stable id with a readable name suffix); each `.md` file's
2224
- * frontmatter only carries `name` and `description`.
3342
+ * 1. Shared knowledge (projects, lessons, external systems) — indexed by a
3343
+ * single hand-maintained `<cwd>/.memory/MEMORY.md` that the *agent* writes
3344
+ * and curates, Claude-Code style: one line per fact pointing at the file
3345
+ * that holds it. `readMemoryIndexFile` just reads that file; the agent owns
3346
+ * its contents. This is the whole index for the shared half — there is no
3347
+ * derived walk and no per-directory `MEMORY.md`.
2225
3348
  *
2226
- * `buildMemoryIndex` walks `.memory/` by type directory, reads only the
2227
- * frontmatter of each `.md` (open fd → read first ~4KB → close, in
2228
- * parallel), and renders a markdown index grouped by type and (where
2229
- * applicable) by subject. Files outside the four type directories are
2230
- * ignored. Bodies are never read — the agent loads a specific memory's
2231
- * body on demand via the `read` tool when the index entry says it's
2232
- * relevant.
3349
+ * 2. Per-person memory — `.memory/users/<id>-<name>/<topic>.md`. This half is
3350
+ * *derived*, not hand-maintained, because it has to be filtered to the one
3351
+ * person on the current turn (a single hand-written index couldn't be
3352
+ * scoped per-user without leaking one person's notes into another's
3353
+ * conversation). `buildMemoryIndex` walks a single user's directory, reads
3354
+ * only the frontmatter of each `.md` (open fd → read first ~4KB → close, in
3355
+ * parallel), and renders an index. Each `.md`'s frontmatter carries `name`
3356
+ * and `description`; bodies are never read — the agent loads a specific
3357
+ * memory's body on demand via the `read` tool.
2233
3358
  *
2234
- * Mtime cache keyed by cwd — within the lifetime of a sandbox the cwd
2235
- * is fixed, so this is effectively a single-entry cache. Cache invalidates
2236
- * when any `.md` in the tree is added/modified/deleted; turns where
2237
- * memory didn't change reuse the cached string.
3359
+ * Mtime cache (for the derived per-user half) keyed by cwd — within the
3360
+ * lifetime of a sandbox the cwd is fixed, so this is effectively a single-entry
3361
+ * cache. It invalidates when any `.md` under `users/` is added/modified/
3362
+ * deleted; turns where memory didn't change reuse the cached entries.
2238
3363
  *
2239
- * Frontmatter is parsed as YAML (`yaml` package) and validated with a
2240
- * zod schema — files that don't match the shape are dropped from the
2241
- * index. The same schema can be reused at write time if we want to
2242
- * validate before commit.
3364
+ * Frontmatter is parsed as YAML (`yaml` package) and validated with a zod
3365
+ * schema — files that don't match the shape are dropped from the index.
2243
3366
  */
2244
3367
  const FRONTMATTER_READ_BYTES = 4096;
2245
3368
  const FrontmatterSchema = z.object({
@@ -2247,33 +3370,118 @@ const FrontmatterSchema = z.object({
2247
3370
  description: z.string().min(1)
2248
3371
  }).passthrough();
2249
3372
  const MEMORY_DIRNAME = ".memory";
2250
- const TYPE_DIRS = [
2251
- "users",
2252
- "projects",
2253
- "feedback",
2254
- "reference"
3373
+ const MEMORY_INDEX_FILENAME = "MEMORY.md";
3374
+ const USERS_DIRNAME = "users";
3375
+ /**
3376
+ * The shared-knowledge type dirs from the old frontmatter-indexed layout, used
3377
+ * only to seed a `MEMORY.md` for agents created before it existed (see
3378
+ * `seedMemoryIndexFile`). `users/` is deliberately excluded — per-person memory
3379
+ * stays derived and never lands in the shared, un-scoped `MEMORY.md`.
3380
+ */
3381
+ const LEGACY_SHARED_TYPES = [
3382
+ {
3383
+ dir: "projects",
3384
+ label: "Projects"
3385
+ },
3386
+ {
3387
+ dir: "feedback",
3388
+ label: "Feedback"
3389
+ },
3390
+ {
3391
+ dir: "reference",
3392
+ label: "Reference"
3393
+ }
2255
3394
  ];
2256
- const TYPES_WITH_SUBJECT = new Set(["users", "projects"]);
2257
- const TYPE_LABELS = {
2258
- users: "Users",
2259
- projects: "Projects",
2260
- feedback: "Feedback",
2261
- reference: "Reference"
2262
- };
3395
+ /**
3396
+ * Soft budget for an injected index block. The index is read and injected into
3397
+ * the system prompt on every turn, so every entry costs context for the rest of
3398
+ * the conversation. Past this size we nudge the agent to consolidate and prune
3399
+ * rather than keep appending. Not a hard cap — nothing is truncated.
3400
+ */
3401
+ const MEMORY_INDEX_SOFT_BUDGET_CHARS = 2e4;
3402
+ /**
3403
+ * Hard cap on the injected index — double the soft budget. The soft budget only
3404
+ * warns; this actually bounds what we inject so a runaway index can't consume
3405
+ * unbounded context on every turn. Past this, the index is truncated (on a line
3406
+ * boundary) before injection. It's a backstop, not a normal operating point.
3407
+ */
3408
+ const MEMORY_INDEX_HARD_BUDGET_CHARS = MEMORY_INDEX_SOFT_BUDGET_CHARS * 2;
3409
+ /**
3410
+ * Cheap size summary of a rendered index block, used to surface how much of the
3411
+ * every-turn context budget the index is spending so the agent keeps it lean.
3412
+ * Counts pointer/file lines — both the derived `` - `path` — desc`` form and
3413
+ * the hand-maintained `- [Title](path) — hook` form — not the group headers.
3414
+ */
3415
+ function summarizeIndex(index) {
3416
+ const entryCount = index.split("\n").filter((line) => /^\s*- (?:`|\[)/.test(line)).length;
3417
+ const charCount = index.length;
3418
+ return {
3419
+ entryCount,
3420
+ charCount,
3421
+ overBudget: charCount > MEMORY_INDEX_SOFT_BUDGET_CHARS
3422
+ };
3423
+ }
3424
+ /**
3425
+ * One-line size note for an index block header, e.g. `12 entries, 3187 chars`.
3426
+ */
3427
+ function indexSizeNote(index) {
3428
+ const { entryCount, charCount } = summarizeIndex(index);
3429
+ return `${entryCount} ${entryCount === 1 ? "entry" : "entries"}, ${charCount} chars`;
3430
+ }
3431
+ /**
3432
+ * An explicit warning to surface to the agent when an index has grown past its
3433
+ * budget, or `null` when it's within budget. Extensions render this prominently
3434
+ * above the index so the agent prunes before it keeps appending.
3435
+ */
3436
+ function indexBudgetWarning(index) {
3437
+ const { charCount, overBudget } = summarizeIndex(index);
3438
+ if (!overBudget) return null;
3439
+ return `⚠️ This memory index is ${charCount} chars, over its ${MEMORY_INDEX_SOFT_BUDGET_CHARS}-char budget. It's costing you context on every turn — consolidate duplicate entries and delete stale ones to bring it back under budget before adding anything new.`;
3440
+ }
3441
+ /**
3442
+ * Enforce the hard cap on an index before injection. Under the cap the index is
3443
+ * returned unchanged; over it, the index is truncated on a line boundary and a
3444
+ * notice is appended naming the true size so the agent knows entries are hidden
3445
+ * and must be pruned. This is the actual bound on injected context — callers
3446
+ * still report the true size via {@link indexSizeNote} so nothing is masked.
3447
+ */
3448
+ function enforceMemoryIndexHardBudget(index) {
3449
+ if (index.length <= 4e4) return index;
3450
+ const clipped = index.slice(0, MEMORY_INDEX_HARD_BUDGET_CHARS);
3451
+ const lastNewline = clipped.lastIndexOf("\n");
3452
+ return `${lastNewline > 0 ? clipped.slice(0, lastNewline) : clipped}\n\n⚠️ Memory index truncated at ${MEMORY_INDEX_HARD_BUDGET_CHARS} chars (it is ${index.length}). Entries past this point are NOT shown. Prune the index now — delete stale entries and consolidate duplicates.`;
3453
+ }
3454
+ /**
3455
+ * Read the agent's hand-maintained shared index at `.memory/MEMORY.md`.
3456
+ * Returns the trimmed contents, or `null` when the file is absent or empty —
3457
+ * the agent owns this file, so we surface exactly what it wrote.
3458
+ */
3459
+ async function readMemoryIndexFile({ cwd }) {
3460
+ const path = join(cwd, MEMORY_DIRNAME, MEMORY_INDEX_FILENAME);
3461
+ try {
3462
+ const trimmed = (await readFile(path, "utf-8")).trim();
3463
+ return trimmed.length > 0 ? trimmed : null;
3464
+ } catch {
3465
+ return null;
3466
+ }
3467
+ }
2263
3468
  const cache = /* @__PURE__ */ new Map();
2264
3469
  /**
3470
+ * Build the derived per-person index for a single user.
3471
+ *
2265
3472
  * Returns:
2266
- * - `null` if `.memory/` doesn't exist
2267
- * - `""` if the dir exists but contains nothing in the requested scope
3473
+ * - `null` if `.memory/users/` doesn't exist
3474
+ * - `""` if it exists but this user has no memory
2268
3475
  * - rendered markdown body (no surrounding header — caller wraps)
2269
3476
  *
2270
- * The mtime-keyed cache stores the raw walked entries (the cost is the FS
2271
- * walk); filtering by scope is cheap and runs per call, so two turns with
2272
- * different scopes on the same cwd render correctly from one cached walk.
3477
+ * Scoping by id keeps one person's memory from bleeding into another's
3478
+ * conversation. The mtime-keyed cache stores the raw walked entries (the cost
3479
+ * is the FS walk); filtering by user is cheap and runs per call, so two turns
3480
+ * with different users on the same cwd render correctly from one cached walk.
2273
3481
  */
2274
- async function buildMemoryIndex({ cwd, scope }) {
2275
- const memoryDirAbs = join(cwd, MEMORY_DIRNAME);
2276
- const maxMtimeMs = await maxMtimeAcrossDir(memoryDirAbs);
3482
+ async function buildMemoryIndex({ cwd, userId }) {
3483
+ const usersDirAbs = join(cwd, MEMORY_DIRNAME, USERS_DIRNAME);
3484
+ const maxMtimeMs = await maxMtimeAcrossDir(usersDirAbs);
2277
3485
  if (maxMtimeMs === null) {
2278
3486
  cache.delete(cwd);
2279
3487
  return null;
@@ -2281,13 +3489,13 @@ async function buildMemoryIndex({ cwd, scope }) {
2281
3489
  let cached = cache.get(cwd);
2282
3490
  if (!cached || cached.builtAtMs < maxMtimeMs) {
2283
3491
  cached = {
2284
- entries: await collectEntries(memoryDirAbs, cwd),
3492
+ entries: await collectUserEntries(usersDirAbs, cwd),
2285
3493
  builtAtMs: Date.now()
2286
3494
  };
2287
3495
  cache.set(cwd, cached);
2288
3496
  }
2289
- const visible = cached.entries.filter((entry) => scope.kind === "user" ? entry.type === "users" && entry.subject?.startsWith(scope.userId) === true : entry.type !== "users");
2290
- return visible.length === 0 ? "" : renderIndex(visible);
3497
+ const visible = cached.entries.filter((entry) => entry.subject.startsWith(userId));
3498
+ return visible.length === 0 ? "" : renderUserIndex(visible);
2291
3499
  }
2292
3500
  async function maxMtimeAcrossDir(dir) {
2293
3501
  let dirStat;
@@ -2335,45 +3543,22 @@ async function listSubdirs(dir) {
2335
3543
  }
2336
3544
  return entries.filter((e) => e.isDirectory()).map((e) => join(dir, e.name));
2337
3545
  }
2338
- async function collectEntries(rootDirAbs, cwd) {
2339
- const collected = [];
2340
- await Promise.all(TYPE_DIRS.map(async (type) => {
2341
- const typeDirAbs = join(rootDirAbs, type);
2342
- if (TYPES_WITH_SUBJECT.has(type)) {
2343
- const subjectDirs = await listSubdirs(typeDirAbs);
2344
- await Promise.all(subjectDirs.map(async (subjectDirAbs) => {
2345
- const subject = basename(subjectDirAbs);
2346
- const files = await listMdFilesShallow(subjectDirAbs);
2347
- const parsed = await Promise.all(files.map(async (file) => {
2348
- const fm = await readFrontmatterOnly(file);
2349
- if (!fm?.name || !fm?.description) return null;
2350
- return {
2351
- name: fm.name,
2352
- description: fm.description,
2353
- type,
2354
- subject,
2355
- relPath: relative(cwd, file)
2356
- };
2357
- }));
2358
- for (const e of parsed) if (e) collected.push(e);
2359
- }));
2360
- } else {
2361
- const files = await listMdFilesShallow(typeDirAbs);
2362
- const parsed = await Promise.all(files.map(async (file) => {
2363
- const fm = await readFrontmatterOnly(file);
2364
- if (!fm?.name || !fm?.description) return null;
2365
- return {
2366
- name: fm.name,
2367
- description: fm.description,
2368
- type,
2369
- subject: null,
2370
- relPath: relative(cwd, file)
2371
- };
2372
- }));
2373
- for (const e of parsed) if (e) collected.push(e);
2374
- }
2375
- }));
2376
- return collected;
3546
+ async function collectUserEntries(usersDirAbs, cwd) {
3547
+ const subjectDirs = await listSubdirs(usersDirAbs);
3548
+ return (await Promise.all(subjectDirs.map(async (subjectDirAbs) => {
3549
+ const subject = basename(subjectDirAbs);
3550
+ const files = await listMdFilesShallow(subjectDirAbs);
3551
+ return (await Promise.all(files.map(async (file) => {
3552
+ const fm = await readFrontmatterOnly(file);
3553
+ if (!fm?.name || !fm?.description) return null;
3554
+ return {
3555
+ name: fm.name,
3556
+ description: fm.description,
3557
+ subject,
3558
+ relPath: relative(cwd, file)
3559
+ };
3560
+ }))).filter((e) => e !== null);
3561
+ }))).flat();
2377
3562
  }
2378
3563
  async function readFrontmatterOnly(filePath) {
2379
3564
  let fh;
@@ -2404,77 +3589,189 @@ function parseFrontmatter(text) {
2404
3589
  const result = FrontmatterSchema.safeParse(parsed);
2405
3590
  return result.success ? result.data : null;
2406
3591
  }
2407
- function renderIndex(entries) {
2408
- const byType = {
2409
- users: [],
2410
- projects: [],
2411
- feedback: [],
2412
- reference: []
2413
- };
2414
- for (const e of entries) byType[e.type].push(e);
3592
+ function renderUserIndex(entries) {
3593
+ const bySubject = /* @__PURE__ */ new Map();
3594
+ for (const e of entries) {
3595
+ const list = bySubject.get(e.subject) ?? [];
3596
+ list.push(e);
3597
+ bySubject.set(e.subject, list);
3598
+ }
3599
+ const lines = ["### Users"];
3600
+ for (const subject of [...bySubject.keys()].sort()) {
3601
+ lines.push(`- **${subject}**`);
3602
+ for (const e of bySubject.get(subject) ?? []) lines.push(` - \`${e.relPath}\` — ${e.description}`);
3603
+ }
3604
+ return lines.join("\n");
3605
+ }
3606
+ /**
3607
+ * One-time migration for agents created before `MEMORY.md` existed. If there is
3608
+ * no hand-maintained `.memory/MEMORY.md` yet but the agent has shared memory
3609
+ * files from the old frontmatter-indexed layout (`projects/`, `feedback/`,
3610
+ * `reference/`), derive a `MEMORY.md` from their frontmatter and write it once.
3611
+ * After that the agent owns the file — this never runs again for that agent and
3612
+ * never clobbers an existing index.
3613
+ *
3614
+ * Returns the seeded contents (also written to disk), or `null` when nothing
3615
+ * was seeded (index already present, or no legacy shared files). A write
3616
+ * failure propagates so the caller can log it; the read path then falls back to
3617
+ * whatever is on disk.
3618
+ */
3619
+ async function seedMemoryIndexFile({ cwd }) {
3620
+ const memoryDirAbs = join(cwd, MEMORY_DIRNAME);
3621
+ const indexPath = join(memoryDirAbs, MEMORY_INDEX_FILENAME);
3622
+ if (await stat(indexPath).catch(() => null)) return null;
3623
+ const perType = await Promise.all(LEGACY_SHARED_TYPES.map(async ({ dir, label }) => {
3624
+ const typeDirAbs = join(memoryDirAbs, dir);
3625
+ const files = [];
3626
+ await walkMdFiles(typeDirAbs, files);
3627
+ return {
3628
+ label,
3629
+ entries: (await Promise.all(files.map(async (file) => {
3630
+ const fm = await readFrontmatterOnly(file);
3631
+ if (!fm?.name || !fm?.description) return null;
3632
+ return {
3633
+ name: fm.name,
3634
+ description: fm.description,
3635
+ relPath: relative(cwd, file)
3636
+ };
3637
+ }))).filter((e) => !!e)
3638
+ };
3639
+ }));
3640
+ if (perType.every((group) => group.entries.length === 0)) return null;
3641
+ const content = renderSeededIndex(perType);
3642
+ await writeFile(indexPath, `${content}\n`, "utf-8");
3643
+ return content;
3644
+ }
3645
+ function renderSeededIndex(groups) {
2415
3646
  const sections = [];
2416
- for (const type of TYPE_DIRS) {
2417
- const items = byType[type];
2418
- if (items.length === 0) continue;
2419
- sections.push(`### ${TYPE_LABELS[type]}`);
2420
- if (TYPES_WITH_SUBJECT.has(type)) {
2421
- const bySubject = /* @__PURE__ */ new Map();
2422
- for (const e of items) {
2423
- const subject = e.subject ?? "(unknown)";
2424
- const list = bySubject.get(subject) ?? [];
2425
- list.push(e);
2426
- bySubject.set(subject, list);
2427
- }
2428
- const subjects = [...bySubject.keys()].sort();
2429
- for (const subject of subjects) {
2430
- sections.push(`- **${subject}**`);
2431
- for (const e of bySubject.get(subject) ?? []) sections.push(` - \`${e.relPath}\` — ${e.description}`);
2432
- }
2433
- } else for (const e of items) sections.push(`- \`${e.relPath}\` — ${e.description}`);
3647
+ for (const { label, entries } of groups) {
3648
+ if (entries.length === 0) continue;
3649
+ sections.push(`### ${label}`);
3650
+ const sorted = [...entries].sort((a, b) => a.relPath.localeCompare(b.relPath));
3651
+ for (const e of sorted) sections.push(`- [${e.name}](${e.relPath}) — ${e.description}`);
2434
3652
  sections.push("");
2435
3653
  }
2436
3654
  return sections.join("\n").trimEnd();
2437
3655
  }
3656
+ /**
3657
+ * The version at which the hand-maintained `MEMORY.md` layout was introduced.
3658
+ * Used to classify an unversioned `.memory/`: if it already has a `MEMORY.md`
3659
+ * it's on this layout (not the pre-MEMORY.md v1 frontmatter layout), so it
3660
+ * shouldn't be treated as v1 and re-seeded.
3661
+ */
3662
+ const HAND_MAINTAINED_INDEX_VERSION = 2;
3663
+ const VERSION_FILENAME = ".version";
3664
+ const MEMORY_MIGRATIONS = [{
3665
+ from: 1,
3666
+ to: 2,
3667
+ apply: async ({ cwd }) => {
3668
+ await seedMemoryIndexFile({ cwd });
3669
+ }
3670
+ }];
3671
+ /**
3672
+ * The layout version of an agent's `.memory/`:
3673
+ * - `null` when there's no `.memory/` at all (a fresh agent is current by
3674
+ * construction; nothing to migrate).
3675
+ * - `1` when `.memory/` exists but carries no `.version` marker — i.e. it
3676
+ * predates versioning.
3677
+ * - otherwise the integer in `.memory/.version`.
3678
+ */
3679
+ async function readMemoryVersion(cwd) {
3680
+ const memoryDirAbs = join(cwd, MEMORY_DIRNAME);
3681
+ if (!(await stat(memoryDirAbs).catch(() => null))?.isDirectory()) return null;
3682
+ const raw = await readFile(join(memoryDirAbs, VERSION_FILENAME), "utf-8").catch(() => null);
3683
+ if (raw === null) return await stat(join(memoryDirAbs, MEMORY_INDEX_FILENAME)).then((s) => s.isFile()).catch(() => false) ? HAND_MAINTAINED_INDEX_VERSION : 1;
3684
+ const parsed = Number.parseInt(raw.trim(), 10);
3685
+ return Number.isInteger(parsed) && parsed > 0 ? parsed : 1;
3686
+ }
3687
+ async function writeMemoryVersion(cwd, version) {
3688
+ await writeFile(join(cwd, MEMORY_DIRNAME, VERSION_FILENAME), `${version}\n`, "utf-8");
3689
+ }
3690
+ /**
3691
+ * Bring an agent's `.memory/` up to `CURRENT_MEMORY_VERSION` by applying the
3692
+ * ordered migrations. Runs on session start. No-op when there's no `.memory/`
3693
+ * yet or it's already current. Migrations must be idempotent, so a lost/unwritten
3694
+ * version marker (the file isn't committed by the harness) only costs a repeated
3695
+ * no-op, never corruption. Returns the `{ from, to }` actually applied, or
3696
+ * `null` when nothing ran.
3697
+ */
3698
+ async function migrateMemory({ cwd }) {
3699
+ const from = await readMemoryVersion(cwd);
3700
+ if (from === null || from >= 2) return null;
3701
+ let version = from;
3702
+ while (version < 2) {
3703
+ const migration = MEMORY_MIGRATIONS.find((m) => m.from === version);
3704
+ if (!migration) break;
3705
+ await migration.apply({ cwd });
3706
+ version = migration.to;
3707
+ }
3708
+ await writeMemoryVersion(cwd, version);
3709
+ return {
3710
+ from,
3711
+ to: version
3712
+ };
3713
+ }
2438
3714
  //#endregion
2439
3715
  //#region src/extensions/memory.ts
2440
- const log$6 = logger.child({ module: "memory-extension" });
3716
+ const log$5 = logger.child({ module: "memory-extension" });
2441
3717
  /**
2442
3718
  * The standing instructions for the memory system. Always injected (even with
2443
- * an empty `.memory/`) so the agent knows it can persist notes. `users/` is
3719
+ * no `MEMORY.md`) so the agent knows it can persist notes and how. `users/` is
2444
3720
  * described by the platform memory extension, which is the only thing that can
2445
3721
  * scope it to a person — here we just point at it.
2446
3722
  */
2447
3723
  function memoryInstructions(cwd) {
2448
3724
  return `## Memory across conversations
2449
3725
 
2450
- Persistent notes across conversations live at \`${cwd}/.memory/\` — plain markdown files in your repo. The harness builds and injects an **index** of these files (paths + one-line descriptions) into your system prompt every turn; **bodies are NOT auto-loaded** — when an index entry looks relevant, use your \`read\` tool to load that specific file.
3726
+ Persistent notes across conversations live at \`${cwd}/.memory/\` — plain markdown files in your repo, one fact per file. You maintain a hand-written index of them at \`${cwd}/.memory/MEMORY.md\`, and the harness injects that index into your system prompt every turn. **Bodies are NOT auto-loaded** — when an index line looks relevant, use your \`read\` tool to load that specific file.
2451
3727
 
2452
3728
  Memory records **what happened**: facts you learned, events, investigation findings, project and system details worth carrying forward. It is NOT where behavior goes. A standing rule about how you should act — a "from now on, always/never …", a tone or format preference, a workflow convention a user wants you to follow — belongs in \`soul.md\` (see the Persona / Standing instructions section), not here. When a note is really an instruction about your behavior, write it to \`soul.md\`; when it is a fact or a record of something that occurred, write it here.
2453
3729
 
2454
- Shared knowledge is laid out as \`projects/<slug>/<topic>.md\` for project and system context, \`feedback/<topic>.md\` for concrete lessons learned from something that happened (the event and what it taught you — not a free-floating rule; the rule itself, if durable, goes in \`soul.md\`), and \`reference/<topic>.md\` for how external systems work. (Notes about a specific person live under \`users/\` and are shown separately, scoped to whoever you're talking to.) Each file's frontmatter declares \`name\` and \`description\` (the description is what shows up in the index, so make it a one-line behavior-triggering hook). Commit and push after writing to persist it.`;
3730
+ You own \`MEMORY.md\`. When you learn something durable, write the fact to its own \`.md\` file and add a one-line pointer to \`MEMORY.md\` in the form \`- [Title](relative/path.md) — one-line hook\`, where the hook is what tells future-you when to open the file. \`MEMORY.md\` is an *index*, never a store — put the actual content in the topic file and only a pointer line in \`MEMORY.md\`; do not inline a fact's body into the index even when it seems cheaper. Start each topic file with \`name:\`/\`description:\` frontmatter (the \`description\` is the one-line hook) so the index can be re-seeded, re-derived, or linted from the files themselves. When a fact changes, edit both the file and its line; when it stops being true, delete the file and its line. Shared knowledge is laid out as \`projects/<slug>/<topic>.md\` for project and system context, \`feedback/<topic>.md\` for concrete lessons learned from something that happened (the event and what it taught you — not a free-floating rule; the rule itself, if durable, goes in \`soul.md\`), and \`reference/<topic>.md\` for how external systems work. (Notes about a specific person live under \`users/\` and are indexed for you separately, scoped to whoever you're talking to — don't put people's private notes in the shared \`MEMORY.md\`.)
3731
+
3732
+ Keep \`MEMORY.md\` lean. It's re-injected on *every* turn, so a small, high-signal index is worth far more than an exhaustive one — curate it like a tightly-edited table of contents, not a log:
3733
+ - Be selective. Only record something durable that will matter in a *future* conversation. Don't record what only matters right now, what you can re-derive on demand, or what's already obvious from the repo.
3734
+ - Consolidate before you create. Before adding a line, scan \`MEMORY.md\` for one that already covers the topic; if it exists, \`read\` that file and rewrite it with the new facts merged in rather than adding a near-duplicate. One fact per file, but don't fragment a topic across many thin files.
3735
+ - Prune as you go. Delete lines (and their files) that are wrong, stale, or superseded. The index header reports its size — when it's flagged over budget, consolidate and delete before adding anything new.
3736
+ - Write dates absolute, not relative. Resolve "last week" / "yesterday" to a concrete date when you record it (e.g. "on 7/3 Dhruv told me to …"), so the note still reads correctly in a future conversation.
3737
+
3738
+ Commit and push after editing \`.memory/\` to persist it.`;
2455
3739
  }
2456
3740
  function composeBlock$1({ cwd, index }) {
2457
3741
  const instructions = memoryInstructions(cwd);
2458
3742
  if (!index || index.length === 0) return instructions;
2459
- return `${instructions}\n\n## Memory index\n\n${index}`;
3743
+ const header = `## Memory index (${indexSizeNote(index)})`;
3744
+ const warning = indexBudgetWarning(index);
3745
+ const rendered = enforceMemoryIndexHardBudget(index);
3746
+ return `${instructions}\n\n${warning ? `${header}\n\n${warning}\n\n${rendered}` : `${header}\n\n${rendered}`}`;
2460
3747
  }
2461
3748
  const memoryExtension = (pi) => {
2462
3749
  let cachedBlock = null;
2463
3750
  pi.on("session_start", async (_event, ctx) => {
2464
3751
  try {
2465
- const index = await buildMemoryIndex({
2466
- cwd: ctx.cwd,
2467
- scope: { kind: "shared" }
2468
- });
3752
+ const migrated = await migrateMemory({ cwd: ctx.cwd });
3753
+ if (migrated !== null) log$5.info({
3754
+ event: "memory_migrated",
3755
+ from: migrated.from,
3756
+ to: migrated.to
3757
+ }, "migrated memory layout to current version");
3758
+ } catch (err) {
3759
+ log$5.warn({
3760
+ err,
3761
+ event: "memory_migration_failed"
3762
+ }, "memory migration failed; continuing with existing index");
3763
+ }
3764
+ try {
3765
+ const index = await readMemoryIndexFile({ cwd: ctx.cwd });
2469
3766
  cachedBlock = composeBlock$1({
2470
3767
  cwd: ctx.cwd,
2471
3768
  index
2472
3769
  });
2473
3770
  } catch (err) {
2474
- log$6.warn({
3771
+ log$5.warn({
2475
3772
  err,
2476
3773
  event: "memory_index_failed"
2477
- }, "memory index build failed; injecting instructions only");
3774
+ }, "memory index read failed; injecting instructions only");
2478
3775
  cachedBlock = memoryInstructions(ctx.cwd);
2479
3776
  }
2480
3777
  });
@@ -2485,7 +3782,7 @@ const memoryExtension = (pi) => {
2485
3782
  };
2486
3783
  //#endregion
2487
3784
  //#region src/extensions/platform-memory.ts
2488
- const log$5 = logger.child({ module: "platform-memory-extension" });
3785
+ const log$4 = logger.child({ module: "platform-memory-extension" });
2489
3786
  /**
2490
3787
  * Resolve the human on this turn via the API, keyed by the message id.
2491
3788
  * `/sandbox/channel-context` only returns a sender for a platform-known
@@ -2498,13 +3795,13 @@ const log$5 = logger.child({ module: "platform-memory-extension" });
2498
3795
  async function resolveTurnUser(messageId) {
2499
3796
  const client = sandboxClient();
2500
3797
  if (!client) {
2501
- log$5.debug({ event: "resolve_turn_user_no_api_url" }, "no API url in env; withholding user memory");
3798
+ log$4.debug({ event: "resolve_turn_user_no_api_url" }, "no API url in env; withholding user memory");
2502
3799
  return null;
2503
3800
  }
2504
3801
  try {
2505
3802
  const res = await client["channel-context"].$get({ query: { messageId } });
2506
3803
  if (!res.ok) {
2507
- log$5.warn({
3804
+ log$4.warn({
2508
3805
  event: "resolve_turn_user_failed",
2509
3806
  status: res.status
2510
3807
  }, "channel-context returned non-ok; withholding user memory");
@@ -2517,7 +3814,7 @@ async function resolveTurnUser(messageId) {
2517
3814
  displayName: sender.displayName
2518
3815
  };
2519
3816
  } catch (err) {
2520
- log$5.warn({
3817
+ log$4.warn({
2521
3818
  err,
2522
3819
  event: "resolve_turn_user_failed"
2523
3820
  }, "failed to resolve current user; withholding user memory");
@@ -2533,9 +3830,12 @@ function slugifyName(name) {
2533
3830
  function composeBlock({ index, user }) {
2534
3831
  const instructions = `## Current user memory
2535
3832
 
2536
- Notes about the person on this turn — the only \`users/\` memory you can see. Store anything you learn about them under \`${`.memory/users/${user.id}-${slugifyName(user.displayName)}/`}<topic>.md\`, using exactly this directory. Other people's \`users/\` notes are never shown, so never address someone by a name you only find in memory.`;
3833
+ Notes about the person on this turn — the only \`users/\` memory you can see. Store anything you learn about them under \`${`.memory/users/${user.id}-${slugifyName(user.displayName)}/`}<topic>.md\`, using exactly this directory. Other people's \`users/\` notes are never shown, so never address someone by a name you only find in memory. Keep it lean and be selective — this is re-injected every turn; consolidate related facts into one file and delete what's stale rather than piling on near-duplicates.`;
2537
3834
  if (!index || index.length === 0) return instructions;
2538
- return `${instructions}\n\n${index}`;
3835
+ const warning = indexBudgetWarning(index);
3836
+ const header = `Memory index (${indexSizeNote(index)}):`;
3837
+ const rendered = enforceMemoryIndexHardBudget(index);
3838
+ return `${instructions}\n\n${warning ? `${warning}\n\n${header}\n\n${rendered}` : `${header}\n\n${rendered}`}`;
2539
3839
  }
2540
3840
  /**
2541
3841
  * Build the platform memory extension. `channelContext` is the per-turn ref
@@ -2556,15 +3856,12 @@ function createPlatformMemoryExtension({ channelContext }) {
2556
3856
  cachedBlock = composeBlock({
2557
3857
  index: await buildMemoryIndex({
2558
3858
  cwd: ctx.cwd,
2559
- scope: {
2560
- kind: "user",
2561
- userId: user.id
2562
- }
3859
+ userId: user.id
2563
3860
  }),
2564
3861
  user
2565
3862
  });
2566
3863
  } catch (err) {
2567
- log$5.warn({
3864
+ log$4.warn({
2568
3865
  err,
2569
3866
  event: "user_memory_index_failed"
2570
3867
  }, "user memory index build failed; skipping injection");
@@ -2579,7 +3876,7 @@ function createPlatformMemoryExtension({ channelContext }) {
2579
3876
  }
2580
3877
  //#endregion
2581
3878
  //#region src/extensions/self-trace.ts
2582
- const log$4 = logger.child({ module: "self-trace-extension" });
3879
+ const log$3 = logger.child({ module: "self-trace-extension" });
2583
3880
  /**
2584
3881
  * Reports the agent's own execution as OpenTelemetry spans:
2585
3882
  * agent.session → agent.run → agent.turn.N → tool.NAME, with token/cost
@@ -2604,7 +3901,7 @@ const selfTraceExtension = (pi) => {
2604
3901
  sessionSpan = tracer.startSpan("agent.session", { attributes: { "agent.model": modelId } }, remoteCtx);
2605
3902
  sessionCtx = trace.setSpan(remoteCtx, sessionSpan);
2606
3903
  const sc = sessionSpan.spanContext();
2607
- log$4.info({
3904
+ log$3.info({
2608
3905
  event: "self_trace_session_start",
2609
3906
  trace_id: sc.traceId,
2610
3907
  span_id: sc.spanId,
@@ -2716,13 +4013,13 @@ const selfTraceExtension = (pi) => {
2716
4013
  * Lives in the harness package — soul.md is content from the agent's
2717
4014
  * own git repo, not from the platform — so its handling stays here.
2718
4015
  */
2719
- const log$3 = logger.child({ module: "soul-extension" });
4016
+ const log$2 = logger.child({ module: "soul-extension" });
2720
4017
  async function readSoul(cwd) {
2721
4018
  try {
2722
4019
  return (await readFile(join(cwd, "soul.md"), "utf8")).trim() || null;
2723
4020
  } catch (err) {
2724
4021
  if (err?.code === "ENOENT") return null;
2725
- log$3.warn({
4022
+ log$2.warn({
2726
4023
  err,
2727
4024
  event: "soul_read_failed"
2728
4025
  }, "soul.md read failed");
@@ -2736,7 +4033,7 @@ function soulSection(cwd, soul) {
2736
4033
 
2737
4034
  **\`soul.md\` is where behavior lives.** Any standing instruction about how you should act — a rule a user wants you to follow going forward, a tone or format preference, a workflow convention, a "from now on, always/never …" — belongs here, not in \`.memory/\`. Memory records *what happened* (facts, events, findings); soul defines *how you behave*. When a user gives you a durable behavioral rule, write it to \`soul.md\`. If you find behavioral rules that ended up in \`.memory/\`, treat that as misfiled and move them here.
2738
4035
 
2739
- Keep it current. When you gain a durable new capability — a tool you build, a skill or integration you set up, a service you connect — or a user hands you a lasting behavioral rule, record it in \`soul.md\` so a future conversation knows it's part of you rather than rediscovering it from scratch. Edit it (then \`git add soul.md && git commit && git push\`) to redefine yourself; picked up on the next message.
4036
+ Keep it current. When you gain a durable new capability — a tool you build, a skill or integration you set up, a service you connect, a secret or auth credential you wire in — or a user hands you a lasting behavioral rule, record it in \`soul.md\` so a future conversation knows it's part of you rather than rediscovering it from scratch. Do this the moment you gain the capability, and for a credential that means the moment it verifies with a real call, not after a human points out that you forgot. Connecting a capability is itself a durable change worth recording, not merely a step toward the task in front of you. Edit \`soul.md\` (then \`git add soul.md && git commit && git push\`) to redefine yourself; picked up on the next message.
2740
4037
 
2741
4038
  ${soul ? soul : "_(empty — write to `soul.md` to define your persona)_"}`;
2742
4039
  }
@@ -2754,13 +4051,23 @@ const soulExtension = (pi) => {
2754
4051
  };
2755
4052
  //#endregion
2756
4053
  //#region src/extensions/subagent/index.ts
2757
- const log$2 = logger.child({ module: "subagent-ext" });
4054
+ const log$1 = logger.child({ module: "subagent-ext" });
2758
4055
  const MAX_TASKS = 8;
2759
4056
  const TaskItem = Type.Object({
2760
4057
  task: Type.String({ description: "The task to delegate to a subagent run." }),
4058
+ title: Type.String({
4059
+ description: "A SHORT name for this task — 3-6 words, sentence case, no trailing period. This is what the person in the chat sees as the row for this subagent, so name the work, do not restate the prompt. Good: \"Audit the billing gate\", \"Compare competitor pricing\", \"Draft the migration\". Bad: \"You are looking at apps/anyone/web and should check every component…\".",
4060
+ maxLength: 120
4061
+ }),
2761
4062
  persona: Type.Optional(Type.String({ description: "Optional extra system prompt / role for this task, applied ON TOP of the child run's own default persona (your full identity and soul are still there underneath). Omit to run with just your default persona." })),
2762
- model: Type.Optional(Type.String({ description: "Optional model id to run this subagent on (e.g. \"anthropic/claude-opus-4-8\"). Must be a real catalogued model. Omit to run on your own model. If you are locked to a Google-compliant model, only compliant models are accepted." }))
4063
+ model: Type.Optional(Type.String({ description: "Optional model id to run this subagent on. PREFER A LOWER-COST, FASTER MODEL when the task is well-scoped and does not need your full reasoning depth — most delegated subtasks (searching, summarizing, mechanical edits, gathering or reformatting data, running a check) run just as well on a lighter model and cost far less. Reserve a top-tier model for subtasks that genuinely need deep reasoning or careful judgment. Must be a real catalogued model id. Omit to inherit your own model. If you are locked to a Google-compliant model, only compliant models are accepted." })),
4064
+ timeoutMinutes: Type.Optional(Type.Integer({
4065
+ description: "Optional wall-clock timeout for this subagent, in minutes. If the run is still going after this long it is ended and you are rewoken with a timeout result, so a hung subagent can never strand you. Omit for the default (30 minutes). Raise it for genuinely long work (a big migration, a large audit); lower it for a quick lookup. Range 1-360.",
4066
+ minimum: 1,
4067
+ maximum: 360
4068
+ }))
2763
4069
  });
4070
+ const DEFAULT_SUBAGENT_TIMEOUT_MS = 30 * 6e4;
2764
4071
  const SubagentParams = Type.Object({ tasks: Type.Array(TaskItem, {
2765
4072
  description: "One or more tasks to delegate. Each spawns an isolated subagent run linked to this conversation; they run in parallel and each rewakes you with its result when it finishes.",
2766
4073
  minItems: 1,
@@ -2775,7 +4082,9 @@ function buildTool(messageId) {
2775
4082
  "Delegate one or more tasks to subagent runs — fresh isolated copies of yourself, each with its own context window, linked to this conversation.",
2776
4083
  "Use it to parallelize independent work, to keep a large or noisy subtask out of your own context, or to run a task under a specialized persona.",
2777
4084
  "Fire-and-forget: this returns immediately after queueing. It does NOT wait for results. Each subagent runs on its own and, when it finishes, sends you its result on this thread — so queue the work, then keep going or end your turn. To chain, re-delegate after a result lands.",
2778
- "Pass tasks: [{ task, persona?, model? }]. persona is an optional extra system prompt layered ON TOP of your default persona for that task (it adds to, it does not replace, your identity); omit it to run with just your default persona. model is an optional model id for that task; omit it to run on your own model."
4085
+ "Pass tasks: [{ task, title, persona?, model? }]. title is a short 3-6 word name for the task — it is shown to the person in the chat as that subagent's row, so name the work rather than restating the prompt. persona is an optional extra system prompt layered ON TOP of your default persona for that task (it adds to, it does not replace, your identity); omit it to run with just your default persona. model is an optional model id for that task — prefer a lower-cost, faster model for well-scoped subtasks that don't need deep reasoning, and reserve a top-tier model for the ones that do; omit it to inherit your own model.",
4086
+ "Peering: each queued task comes back with its own conversation id. A subagent is a real linked conversation, so to see what one is doing RIGHT NOW while it runs — its reasoning, the tools it has called and their results, its progress — read that conversation with `platform conversations show <conversationId>` (you are already authorized; it is your own delegated run). Check in that way instead of waiting blind for the final result. The read reflects the child's persisted state, which lags a few seconds behind live (tool results land as they complete; in-progress reasoning can be up to ~5s stale), so peek between checkpoints rather than polling in a tight loop.",
4087
+ "Steering: to add context, correct course, or answer a question a subagent needs mid-run, post to its conversation with `platform conversations post <conversationId> --message \"...\"`. If the subagent is still running, your message lands as a live steer picked up in that same turn; if it has gone idle, it queues as its next turn. This is the same primitive as any conversation message — there is no separate steer channel."
2779
4088
  ].join(" "),
2780
4089
  promptSnippet: "subagent — delegate tasks to isolated subagent runs; each rewakes you with its result when done",
2781
4090
  parameters: SubagentParams,
@@ -2791,29 +4100,41 @@ function buildTool(messageId) {
2791
4100
  };
2792
4101
  const spawnTasks = tasks.map((t) => ({
2793
4102
  task: t.task,
4103
+ title: t.title ?? null,
2794
4104
  persona: t.persona ?? null,
2795
- model: t.model ?? null
4105
+ model: t.model ?? null,
4106
+ timeoutMs: t.timeoutMinutes != null ? t.timeoutMinutes * 6e4 : DEFAULT_SUBAGENT_TIMEOUT_MS
2796
4107
  }));
2797
4108
  try {
2798
- const { taskIds } = await postSubagentSpawn({
4109
+ const spawned = await postSubagentSpawn({
2799
4110
  messageId,
2800
4111
  tasks: spawnTasks
2801
4112
  });
2802
- log$2.info({
4113
+ const { taskIds } = spawned;
4114
+ log$1.info({
2803
4115
  event: "subagent_spawned",
2804
4116
  count: taskIds.length
2805
4117
  }, "subagent tasks queued");
2806
- const lines = taskIds.map((id, i) => `- ${id}: ${spawnTasks[i]?.task ?? ""}`).join("\n");
4118
+ const convByTask = new Map(spawned.tasks.map((t) => [t.taskId, t.conversationId]));
4119
+ const lines = taskIds.map((id, i) => {
4120
+ const label = spawnTasks[i]?.title ?? spawnTasks[i]?.task ?? "";
4121
+ const conv = convByTask.get(id);
4122
+ return `- ${id}: ${label}${conv ? ` — conversation ${conv}` : ""}`;
4123
+ }).join("\n");
4124
+ const peerHint = spawned.tasks.length ? "\nEach subagent runs on its own conversation (id shown per task above). To SEE what one is doing while it runs, read it with `platform conversations show <conversationId>`. To STEER one mid-run — add context, correct course, answer a question — post to its conversation with `platform conversations post <conversationId> --message \"...\"`; it lands as a live steer if the subagent is still running, or as its next turn if it has gone idle." : "";
2807
4125
  return {
2808
4126
  content: [{
2809
4127
  type: "text",
2810
- text: `Queued ${taskIds.length} subagent ${taskIds.length === 1 ? "run" : "runs"}. Each runs on its own and will send you its result on this thread when it finishes — keep working or end your turn meanwhile.\n${lines}`
4128
+ text: `Queued ${taskIds.length} subagent ${taskIds.length === 1 ? "run" : "runs"}. Each runs on its own and will send you its result on this thread when it finishes — keep working or end your turn meanwhile.\n${lines}${peerHint}`
2811
4129
  }],
2812
- details: { taskIds }
4130
+ details: {
4131
+ taskIds,
4132
+ tasks: spawned.tasks
4133
+ }
2813
4134
  };
2814
4135
  } catch (err) {
2815
4136
  const message = err instanceof Error ? err.message : String(err);
2816
- log$2.warn({
4137
+ log$1.warn({
2817
4138
  err,
2818
4139
  event: "subagent_spawn_failed"
2819
4140
  }, "subagent spawn failed");
@@ -2830,31 +4151,37 @@ function buildTool(messageId) {
2830
4151
  };
2831
4152
  }
2832
4153
  /**
2833
- * Gated on `harness-subagent-enabled`, read from the shared feature-flag poll.
2834
4154
  * The factory takes the session's channel context to resolve the originating
2835
4155
  * messageId — the api links each spawned run to the conversation that message
2836
4156
  * belongs to and rewakes it on completion (nothing about the parent is piped
2837
- * from the sandbox beyond that id).
4157
+ * from the sandbox beyond that id). The tool is registered unconditionally at
4158
+ * session_start.
2838
4159
  */
2839
4160
  function createSubagentExtension({ channelContext }) {
2840
4161
  return (pi) => {
2841
4162
  const messageId = extractMessageId(channelContext);
2842
- startFeatureFlagPoller();
2843
- pi.on("session_start", async () => {
2844
- if (getPolledFlag("subagent") === true) {
2845
- pi.registerTool(buildTool(messageId));
2846
- log$2.info({ event: "subagent_enabled" }, "subagent tool registered");
2847
- }
4163
+ let registered = false;
4164
+ const registerOnce = () => {
4165
+ if (registered) return;
4166
+ registered = true;
4167
+ pi.registerTool(buildTool(messageId));
4168
+ log$1.info({ event: "subagent_enabled" }, "subagent tool registered");
4169
+ };
4170
+ pi.on("session_start", () => {
4171
+ registerOnce();
2848
4172
  });
2849
4173
  };
2850
4174
  }
2851
4175
  //#endregion
2852
4176
  //#region src/extensions/tool-call-env.ts
2853
4177
  const TOOL_CALL_ID_VAR = "TOOL_CALL_ID";
4178
+ function shellQuoteValue(value) {
4179
+ return quote([value]);
4180
+ }
2854
4181
  function withToolCallId({ command, toolCallId }) {
2855
- return `export ${TOOL_CALL_ID_VAR}=${toolCallId}; ${command}`;
4182
+ return `export ${TOOL_CALL_ID_VAR}=${shellQuoteValue(toolCallId)}; ${command}`;
2856
4183
  }
2857
- const PLATFORM_EXPORT = new RegExp(`^\\s*export\\s+(?:${TOOL_CALL_ID_VAR}|ANYONE_\\w+|SKYDIVE_\\w+)=(?:"(?:\\\\.|[^"])*"|'[^']*'|[^;\\s]*)\\s*;\\s*`);
4184
+ const PLATFORM_EXPORT = new RegExp(`^\\s*export\\s+(?:${TOOL_CALL_ID_VAR}|ANYONE_\\w+|SKYDIVE_\\w+)=(?:"(?:\\\\.|[^"])*"|'(?:'\\\\''|[^'])*'|[^;\\s]*)\\s*;\\s*`);
2858
4185
  function stripPlatformExportsForDisplay(command) {
2859
4186
  let c = command;
2860
4187
  let m;
@@ -2872,214 +4199,6 @@ const toolCallEnvExtension = (pi) => {
2872
4199
  });
2873
4200
  };
2874
4201
  //#endregion
2875
- //#region src/extensions/tool-call-summary.ts
2876
- const log$1 = logger.child({ module: "tool-call-summary-extension" });
2877
- /**
2878
- * The injected parameter name: a namespaced sentinel, so it can never collide
2879
- * with a real tool argument and is unmistakable in transcripts and logs. The
2880
- * frontend renderer (ANY-2723) duplicates this literal — keep the two in sync.
2881
- */
2882
- const TOOL_CALL_SUMMARY_FIELD = "__skydive_summary__";
2883
- /** JSON Schema fragment for the injected parameter. */
2884
- const SUMMARY_PROPERTY = {
2885
- type: "string",
2886
- description: "Required for every tool call. A concise, specific summary (max ~8 words) of what THIS call does and why, written for a person watching the conversation, e.g. \"Searching feedback for billing complaints\" or \"Reading the auth middleware\". Address the user directly in second person: the summary is read by the user, so refer to their things as \"your\", never in third person — \"Reading your emails\", not \"Reading his emails\". Always use the present progressive tense, since it is shown while the call runs: \"Updating your Slack\", never \"Updated your Slack\". Make each summary distinct from your other tool calls; never reuse a generic label like \"Search query\" or \"Running command\"."
2887
- };
2888
- const jsonSchemaObjectSchema = z.object({
2889
- type: z.unknown().optional(),
2890
- properties: z.record(z.string(), z.unknown()).optional(),
2891
- required: z.array(z.string()).optional(),
2892
- additionalProperties: z.unknown().optional()
2893
- }).passthrough();
2894
- const toolEntrySchema = z.object({
2895
- name: z.string().optional(),
2896
- input_schema: jsonSchemaObjectSchema.optional(),
2897
- parameters: jsonSchemaObjectSchema.optional(),
2898
- function: z.object({
2899
- name: z.string().optional(),
2900
- parameters: jsonSchemaObjectSchema.optional()
2901
- }).passthrough().optional()
2902
- }).passthrough();
2903
- const payloadWithToolsSchema = z.object({ tools: z.array(z.unknown()) }).passthrough();
2904
- /**
2905
- * Add the summary property to one JSON Schema object. Returns the augmented
2906
- * copy, or `null` when the tool should be left untouched: a strict schema
2907
- * (`additionalProperties: false`) whose validation would reject the extra
2908
- * field, or one that already declares a `__skydive_summary__` property of its own.
2909
- */
2910
- function augmentSchema(schema) {
2911
- if (schema.additionalProperties === false) return null;
2912
- const properties = schema.properties ?? {};
2913
- if ("__skydive_summary__" in properties) return null;
2914
- const required = schema.required ?? [];
2915
- return {
2916
- ...schema,
2917
- type: schema.type ?? "object",
2918
- properties: {
2919
- [TOOL_CALL_SUMMARY_FIELD]: SUMMARY_PROPERTY,
2920
- ...properties
2921
- },
2922
- required: required.includes("__skydive_summary__") ? required : [...required, TOOL_CALL_SUMMARY_FIELD]
2923
- };
2924
- }
2925
- /**
2926
- * Augment a single tool entry, dispatching on which provider shape it is.
2927
- * Returns the (possibly rebuilt) entry and whether anything changed. Skipped
2928
- * tools — wrong shape, strict, or name in `strictToolNames` — return unchanged.
2929
- */
2930
- function augmentToolEntry(entry, strictToolNames) {
2931
- const parsed = toolEntrySchema.safeParse(entry);
2932
- if (!parsed.success) return {
2933
- entry,
2934
- changed: false
2935
- };
2936
- const tool = parsed.data;
2937
- const name = tool.name ?? tool.function?.name ?? null;
2938
- if (name !== null && strictToolNames.has(name)) return {
2939
- entry,
2940
- changed: false
2941
- };
2942
- if (tool.input_schema) {
2943
- const augmented = augmentSchema(tool.input_schema);
2944
- if (!augmented) return {
2945
- entry,
2946
- changed: false
2947
- };
2948
- return {
2949
- entry: {
2950
- ...tool,
2951
- input_schema: augmented
2952
- },
2953
- changed: true
2954
- };
2955
- }
2956
- if (tool.parameters) {
2957
- const augmented = augmentSchema(tool.parameters);
2958
- if (!augmented) return {
2959
- entry,
2960
- changed: false
2961
- };
2962
- return {
2963
- entry: {
2964
- ...tool,
2965
- parameters: augmented
2966
- },
2967
- changed: true
2968
- };
2969
- }
2970
- if (tool.function?.parameters) {
2971
- const augmented = augmentSchema(tool.function.parameters);
2972
- if (!augmented) return {
2973
- entry,
2974
- changed: false
2975
- };
2976
- return {
2977
- entry: {
2978
- ...tool,
2979
- function: {
2980
- ...tool.function,
2981
- parameters: augmented
2982
- }
2983
- },
2984
- changed: true
2985
- };
2986
- }
2987
- return {
2988
- entry,
2989
- changed: false
2990
- };
2991
- }
2992
- /**
2993
- * Inject the summary field into every eligible tool in a provider payload.
2994
- * Returns a new payload when at least one tool was augmented, or `undefined`
2995
- * to signal "no change" (which keeps the original payload, per the
2996
- * `before_provider_request` contract).
2997
- *
2998
- * @param payload The outgoing provider payload (shape varies by provider).
2999
- * @param strictToolNames Names of tools whose registered schema is strict and
3000
- * must be skipped to avoid validation errors.
3001
- */
3002
- function injectToolCallSummary(payload, strictToolNames) {
3003
- const parsed = payloadWithToolsSchema.safeParse(payload);
3004
- if (!parsed.success || parsed.data.tools.length === 0) return void 0;
3005
- let changed = false;
3006
- const tools = parsed.data.tools.map((entry) => {
3007
- const result = augmentToolEntry(entry, strictToolNames);
3008
- if (result.changed) changed = true;
3009
- return result.entry;
3010
- });
3011
- if (!changed) return void 0;
3012
- return {
3013
- ...parsed.data,
3014
- tools
3015
- };
3016
- }
3017
- /**
3018
- * Names of registered tools whose schema sets `additionalProperties: false`.
3019
- * Pi validates the model's tool args against this registered schema, so the
3020
- * injected field would make a strict tool's call fail validation — skip them.
3021
- */
3022
- function getStrictToolNames(pi) {
3023
- const names = /* @__PURE__ */ new Set();
3024
- for (const tool of pi.getAllTools()) {
3025
- const parsed = jsonSchemaObjectSchema.safeParse(tool.parameters);
3026
- if (parsed.success && parsed.data.additionalProperties === false) names.add(tool.name);
3027
- }
3028
- return names;
3029
- }
3030
- function toolDeclaresSummaryParam(pi, toolName) {
3031
- const tool = pi.getAllTools().find((candidate) => candidate.name === toolName);
3032
- if (!tool) return false;
3033
- const parsed = jsonSchemaObjectSchema.safeParse(tool.parameters);
3034
- return parsed.success && parsed.data.properties != null && "__skydive_summary__" in parsed.data.properties;
3035
- }
3036
- /**
3037
- * Remove the injected summary from a tool's execution input. No-op when the
3038
- * field is absent, or when the tool genuinely declares a `__skydive_summary__`
3039
- * parameter of its own (which we never inject into, so its value is real).
3040
- * Mutates `input` in place, matching the `tool_call` contract.
3041
- *
3042
- * Fails open: this runs on the critical path of tool execution, and the
3043
- * `getAllTools()` lookup can throw. On any error we leave `input` untouched
3044
- * (the sentinel may pass through to the tool, but a bug here can never break
3045
- * tool execution).
3046
- */
3047
- function stripInjectedSummary(pi, toolName, input) {
3048
- try {
3049
- if (!("__skydive_summary__" in input)) return;
3050
- if (toolDeclaresSummaryParam(pi, toolName)) return;
3051
- delete input[TOOL_CALL_SUMMARY_FIELD];
3052
- } catch (err) {
3053
- log$1.error({
3054
- err,
3055
- event: "tool_call_summary_strip_failed",
3056
- toolName
3057
- }, "tool_call_summary strip failed; leaving tool input untouched");
3058
- }
3059
- }
3060
- /**
3061
- * Compute the rewritten payload for a `before_provider_request` event, failing
3062
- * open: on any error the original payload is left untouched so a bug here can
3063
- * never break an LLM call.
3064
- */
3065
- function buildInjectedPayload(pi, payload) {
3066
- try {
3067
- return injectToolCallSummary(payload, getStrictToolNames(pi));
3068
- } catch (err) {
3069
- log$1.error({
3070
- err,
3071
- event: "tool_call_summary_injection_failed"
3072
- }, "tool_call_summary injection failed; passing payload through unchanged");
3073
- return;
3074
- }
3075
- }
3076
- const toolCallSummaryExtension = (pi) => {
3077
- pi.on("before_provider_request", (event) => buildInjectedPayload(pi, event.payload));
3078
- pi.on("tool_call", (event) => {
3079
- stripInjectedSummary(pi, event.toolName, event.input);
3080
- });
3081
- };
3082
- //#endregion
3083
4202
  //#region src/extensions/background-tasks.ts
3084
4203
  /**
3085
4204
  * Background bash tasks as a pi extension.
@@ -3119,21 +4238,43 @@ const toolCallSummaryExtension = (pi) => {
3119
4238
  * from the origin messageId in its channel context (`resolveConversationFromApi`)
3120
4239
  * — and every run of the same conversation resolves to the same id, keeping the
3121
4240
  * shared map correctly scoped across turns. `bg_*`, the completion wake, and the
3122
- * next-session injection all filter to the resolved conversation — an agent
3123
- * never sees or is woken by a task from a different chat. Only the output log
4241
+ * next-session injection all filter by a **scope key** — the resolved
4242
+ * conversation id, or, when a session's conversation is unresolvable (a bare
4243
+ * CLI session, or a run whose channel-context ref carries no messageId), a
4244
+ * sentinel unique to that one session instance. Comparing on the raw
4245
+ * `conversationId` would bucket every unresolvable session together under
4246
+ * `null` and leak one's completion wake / status / next-session injection into
4247
+ * another; the sentinel keeps each isolated so an agent never sees or is woken
4248
+ * by a task from a different chat. Only the output log
3124
4249
  * spills to disk (/home/user/.anyone/bg-tasks/<id>.log) to avoid buffering a chatty job
3125
4250
  * in memory; exit code and run state live on the in-memory task.
3126
4251
  *
3127
- * **No cross-restart survival (v1, deliberate).** Task state lives only in
3128
- * the running harness process. A harness restart (crash → supervisord
3129
- * respawn, or `platform harness reload` after the agent edits its own
3130
- * harness) drops the map and pi's exec children are reaped with it. We don't
3131
- * resurrect from disk because the common next-run case cold-provisions a
3132
- * *different* sandbox anyway (warm reuse is the minority in prod), so on-disk
3133
- * state would rarely be the box the next run lands on. The idle-completion wake
3134
- * does cross the sandbox → platform boundary (a fresh run via `bg-task-done`),
3135
- * but a task whose harness dies before it finishes is gone — it is not
3136
- * resurrected, and this stays distinct from the scheduled-run (cron) system.
4252
+ * **Cross-restart survival is OPT-IN, via `bg_run({ resumable: true })`.**
4253
+ * A non-resumable task's state lives only in the running harness process: a
4254
+ * harness restart (crash → supervisord respawn, `platform harness reload`, or
4255
+ * a sandbox recycle on idle timeout / redeploy / template rebuild) drops the
4256
+ * map and pi's exec children are reaped with it, and the task is gone — the
4257
+ * right behavior for a one-shot side-effecting command, which must never
4258
+ * silently re-run.
4259
+ *
4260
+ * A RESUMABLE task additionally checkpoints its *spec* (command, cwd, labels,
4261
+ * origin messageId — not its live output/exit state) to a durable, api-side
4262
+ * journal keyed by conversation (`/sandbox/bg-task-journal`, redis). On the
4263
+ * NEXT run's `session_start` — which usually lands on a *different*,
4264
+ * cold-provisioned sandbox, which is exactly why the journal is api-side and
4265
+ * not on the sandbox disk — the harness lists the journal and relaunches any
4266
+ * spec it isn't already running, keeping the original id and prepending a
4267
+ * `<resumed-after-restart>` banner so the agent knows it re-ran from scratch,
4268
+ * not continued. The spec is dropped from the journal when the task
4269
+ * finishes/kills. Because relaunch RE-EXECUTES the command, resumable is only
4270
+ * for idempotent, long-lived work (pollers, watchers, retry loops); the tool
4271
+ * description enforces this and the default is false.
4272
+ *
4273
+ * The idle-completion wake (a task finishing while the agent is idle) crosses
4274
+ * the sandbox → platform boundary via a fresh run (`bg-task-done`) for both
4275
+ * resumable and non-resumable tasks; the journal is a separate, additive layer
4276
+ * that only handles a task whose harness dies BEFORE it finishes. This stays
4277
+ * distinct from the scheduled-run (cron) system.
3137
4278
  */
3138
4279
  const log = logger.child({ module: "background-tasks-ext" });
3139
4280
  const ops = createLocalBashOperations();
@@ -3144,6 +4285,7 @@ const WATCHDOG_INTERVAL_MS = 3e4;
3144
4285
  const KEEPALIVE_EVERY_MS = 6e4;
3145
4286
  const KEEPALIVE_MAX_MS = 3600 * 1e3;
3146
4287
  const STALL_HINT_AFTER_MS = 120 * 1e3;
4288
+ const PUBLISH_DEBOUNCE_MS = 300;
3147
4289
  const MAX_LOG_BYTES = 100 * 1024 * 1024;
3148
4290
  const DEFAULT_TAIL_LINES = 30;
3149
4291
  const TAIL_READ_BYTES = 64 * 1024;
@@ -3151,11 +4293,12 @@ function taskLabel(meta) {
3151
4293
  return `${meta.id} "${meta.description ?? meta.command.slice(0, 60)}"`;
3152
4294
  }
3153
4295
  let taskCounter = 0;
4296
+ let sessionScopeCounter = 0;
3154
4297
  const tasks = /* @__PURE__ */ new Map();
3155
4298
  let watchdogInterval = null;
3156
4299
  let lastKeepaliveAt = 0;
3157
- function sameConversation(meta, conversationId) {
3158
- return meta.conversationId === conversationId;
4300
+ function sameScope(meta, scopeKey) {
4301
+ return meta.scopeKey === scopeKey;
3159
4302
  }
3160
4303
  function logPath(id) {
3161
4304
  return join(tasksDir(), `${id}.log`);
@@ -3268,6 +4411,10 @@ function createBackgroundTasksExtension({ channelContext }) {
3268
4411
  return id;
3269
4412
  });
3270
4413
  }
4414
+ const unresolvedScopeSentinel = `unresolved:${process.pid.toString(36)}:${(sessionScopeCounter += 1).toString(36)}`;
4415
+ function scopeKey() {
4416
+ return conversationId ?? unresolvedScopeSentinel;
4417
+ }
3271
4418
  let agentActive = false;
3272
4419
  pi.on("agent_start", async () => {
3273
4420
  agentActive = true;
@@ -3295,13 +4442,16 @@ ${recentOutput}
3295
4442
  </background-task-finished>
3296
4443
  Run bg_logs for the full output.
3297
4444
 
3298
- This is a background-task completion, not a message from the user. If it needs no user-facing response — a routine or expected finish, a leftover or self-killed process, nothing the user must act on or would want to know right now — call \`platform channel suppress-reply\` and output nothing. Only send a message if the outcome changes what the user should do or know, or if you were explicitly waiting to report this result.`,
4445
+ This is a background-task completion, not a message from the user.
4446
+ For a routine or expected completion, output nothing.
4447
+ Only reply if the outcome changes what the user should know or do,
4448
+ or if you were explicitly waiting to report it.`,
3299
4449
  display: false
3300
4450
  };
3301
4451
  }
3302
4452
  async function notifyCompletion(meta) {
3303
4453
  if (meta.notified) return;
3304
- if (agentActive && sameConversation(meta, conversationId)) {
4454
+ if (agentActive && sameScope(meta, scopeKey())) {
3305
4455
  meta.notified = true;
3306
4456
  pi.sendMessage(await taskDoneMessage(meta), {
3307
4457
  triggerTurn: true,
@@ -3332,9 +4482,83 @@ This is a background-task completion, not a message from the user. If it needs n
3332
4482
  });
3333
4483
  } else log.info({ taskId: meta.id }, "bg task completed idle with no origin message; deferring to next session_start");
3334
4484
  }
3335
- async function launchTask({ command, description, cwd }) {
4485
+ const SNAPSHOT_TAIL_LINES = 40;
4486
+ let lastPublishedSignature = null;
4487
+ let publishInFlight = null;
4488
+ let publishQueued = false;
4489
+ function snapshotSignature(snapshot) {
4490
+ return JSON.stringify(snapshot.map((t) => ({
4491
+ id: t.id,
4492
+ state: t.state,
4493
+ exitCode: t.exitCode,
4494
+ killedReason: t.killedReason,
4495
+ outputTail: t.outputTail
4496
+ })));
4497
+ }
4498
+ async function doPublishSnapshot() {
4499
+ if (!messageId) return;
4500
+ const mine = [...tasks.values()].filter((t) => sameScope(t, scopeKey()));
4501
+ try {
4502
+ const snapshot = await Promise.all(mine.map(async (t) => ({
4503
+ id: t.id,
4504
+ command: t.command,
4505
+ description: t.description,
4506
+ startedAt: t.startedAt,
4507
+ state: t.running ? "running" : "finished",
4508
+ exitCode: t.exitCode,
4509
+ killedReason: t.killedReason,
4510
+ outputTail: await tailLog(t.id, SNAPSHOT_TAIL_LINES)
4511
+ })));
4512
+ const signature = snapshotSignature(snapshot);
4513
+ if (signature === lastPublishedSignature) return;
4514
+ lastPublishedSignature = signature;
4515
+ postBackgroundTasksSnapshot({
4516
+ messageId,
4517
+ tasks: snapshot
4518
+ });
4519
+ } catch (err) {
4520
+ log.debug({
4521
+ err,
4522
+ event: "bg_tasks_snapshot_build_failed"
4523
+ }, "building bg-tasks snapshot failed");
4524
+ }
4525
+ }
4526
+ async function publishSnapshotNow() {
4527
+ if (publishInFlight) {
4528
+ publishQueued = true;
4529
+ return;
4530
+ }
4531
+ publishInFlight = (async () => {
4532
+ try {
4533
+ do {
4534
+ publishQueued = false;
4535
+ await doPublishSnapshot();
4536
+ } while (publishQueued);
4537
+ } finally {
4538
+ publishInFlight = null;
4539
+ }
4540
+ })();
4541
+ await publishInFlight;
4542
+ }
4543
+ let publishTimer = null;
4544
+ function schedulePublishSnapshot() {
4545
+ if (publishTimer) return;
4546
+ publishTimer = setTimeout(() => {
4547
+ publishTimer = null;
4548
+ publishSnapshotNow();
4549
+ }, PUBLISH_DEBOUNCE_MS);
4550
+ publishTimer.unref?.();
4551
+ }
4552
+ async function flushPublishSnapshot() {
4553
+ if (publishTimer) {
4554
+ clearTimeout(publishTimer);
4555
+ publishTimer = null;
4556
+ }
4557
+ await publishSnapshotNow();
4558
+ }
4559
+ async function launchTask({ command, description, cwd, resumable = false, resumedFromJournal = false, id: providedId, startedAt: providedStartedAt }) {
3336
4560
  taskCounter += 1;
3337
- const id = `bg-${process.pid.toString(36)}-${taskCounter}`;
4561
+ const id = providedId ?? `bg-${process.pid.toString(36)}-${taskCounter}`;
3338
4562
  try {
3339
4563
  await mkdir(tasksDir(), { recursive: true });
3340
4564
  } catch (err) {
@@ -3350,7 +4574,7 @@ This is a background-task completion, not a message from the user. If it needs n
3350
4574
  taskId: id
3351
4575
  }, "bg task log write failed");
3352
4576
  });
3353
- const startedAt = Date.now();
4577
+ const startedAt = providedStartedAt ?? Date.now();
3354
4578
  const meta = {
3355
4579
  id,
3356
4580
  command: stripPlatformExportsForDisplay(command),
@@ -3358,6 +4582,7 @@ This is a background-task completion, not a message from the user. If it needs n
3358
4582
  logBytes: 0,
3359
4583
  lastOutputAt: startedAt,
3360
4584
  conversationId,
4585
+ scopeKey: scopeKey(),
3361
4586
  messageId,
3362
4587
  description,
3363
4588
  notified: false,
@@ -3365,9 +4590,23 @@ This is a background-task completion, not a message from the user. If it needs n
3365
4590
  controller: new AbortController(),
3366
4591
  running: true,
3367
4592
  exitCode: null,
3368
- error: null
4593
+ error: null,
4594
+ resumable,
4595
+ cwd,
4596
+ resumedFromJournal
3369
4597
  };
3370
4598
  tasks.set(id, meta);
4599
+ if (resumable && messageId) putBackgroundTaskJournalSpec({
4600
+ messageId,
4601
+ spec: {
4602
+ id,
4603
+ command,
4604
+ cwd,
4605
+ description,
4606
+ startedAt,
4607
+ messageId
4608
+ }
4609
+ });
3371
4610
  ops.exec(command, cwd, {
3372
4611
  onData: (chunk) => {
3373
4612
  meta.logBytes += chunk.length;
@@ -3393,6 +4632,11 @@ This is a background-task completion, not a message from the user. If it needs n
3393
4632
  taskId: id,
3394
4633
  exitCode: meta.exitCode
3395
4634
  }, "bg task finished");
4635
+ if (meta.resumable && meta.messageId) deleteBackgroundTaskJournalSpec({
4636
+ messageId: meta.messageId,
4637
+ taskId: id
4638
+ });
4639
+ schedulePublishSnapshot();
3396
4640
  await notifyCompletion(meta);
3397
4641
  });
3398
4642
  ensureWatchdog();
@@ -3400,19 +4644,55 @@ This is a background-task completion, not a message from the user. If it needs n
3400
4644
  taskId: id,
3401
4645
  conversationId
3402
4646
  }, "bg task started");
4647
+ schedulePublishSnapshot();
3403
4648
  return meta;
3404
4649
  }
3405
4650
  function knownTaskIds() {
3406
- return [...tasks.values()].filter((t) => sameConversation(t, conversationId)).map((t) => t.id).join(", ") || "(none)";
4651
+ return [...tasks.values()].filter((t) => sameScope(t, scopeKey())).map((t) => t.id).join(", ") || "(none)";
4652
+ }
4653
+ let journalChecked = false;
4654
+ async function rehydrateJournaledTasks() {
4655
+ if (!messageId || journalChecked) return;
4656
+ journalChecked = true;
4657
+ const specs = await listBackgroundTaskJournalSpecs({ messageId });
4658
+ if (specs.length === 0) return;
4659
+ for (const spec of specs) {
4660
+ const live = tasks.get(spec.id);
4661
+ if (live && sameScope(live, scopeKey())) continue;
4662
+ log.info({
4663
+ taskId: spec.id,
4664
+ conversationId
4665
+ }, "relaunching journaled resumable bg task after restart");
4666
+ const meta = await launchTask({
4667
+ command: spec.command,
4668
+ description: spec.description,
4669
+ cwd: spec.cwd,
4670
+ resumable: true,
4671
+ resumedFromJournal: true,
4672
+ id: spec.id,
4673
+ startedAt: spec.startedAt
4674
+ });
4675
+ try {
4676
+ const stream = createWriteStream(logPath(meta.id), { flags: "a" });
4677
+ stream.write(`<resumed-after-restart>requeued and relaunched on a new sandbox after the previous one was drained; this is a fresh execution of the command from the start, not a continuation</resumed-after-restart>\n`);
4678
+ stream.end();
4679
+ } catch (err) {
4680
+ log.debug({
4681
+ err,
4682
+ taskId: meta.id
4683
+ }, "resume banner write failed");
4684
+ }
4685
+ }
3407
4686
  }
3408
4687
  pi.on("session_start", async () => {
3409
- if (tasks.size === 0) return;
4688
+ if (tasks.size === 0 && (!messageId || journalChecked)) return;
3410
4689
  await ensureConversationId();
3411
- for (const [id, meta] of tasks) if (sameConversation(meta, conversationId) && !meta.running && meta.notified) {
4690
+ await rehydrateJournaledTasks();
4691
+ for (const [id, meta] of tasks) if (sameScope(meta, scopeKey()) && !meta.running && meta.notified) {
3412
4692
  tasks.delete(id);
3413
4693
  await unlink(logPath(id)).catch(() => {});
3414
4694
  }
3415
- const unnotified = [...tasks.values()].filter((t) => sameConversation(t, conversationId) && !t.notified && !t.running);
4695
+ const unnotified = [...tasks.values()].filter((t) => sameScope(t, scopeKey()) && !t.notified && !t.running);
3416
4696
  for (const meta of unnotified) {
3417
4697
  meta.notified = true;
3418
4698
  pi.sendMessage(await taskDoneMessage(meta));
@@ -3421,7 +4701,9 @@ This is a background-task completion, not a message from the user. If it needs n
3421
4701
  conversationId,
3422
4702
  count: unnotified.length
3423
4703
  }, "injected completed bg tasks at session_start");
3424
- if ([...tasks.values()].some((t) => sameConversation(t, conversationId) && t.running)) ensureWatchdog();
4704
+ if ([...tasks.values()].some((t) => sameScope(t, scopeKey()) && t.running)) ensureWatchdog();
4705
+ lastPublishedSignature = null;
4706
+ await flushPublishSnapshot();
3425
4707
  });
3426
4708
  function err(text) {
3427
4709
  return {
@@ -3438,11 +4720,11 @@ This is a background-task completion, not a message from the user. If it needs n
3438
4720
  }
3439
4721
  function resolveTask(taskId) {
3440
4722
  const exact = tasks.get(taskId);
3441
- if (exact && sameConversation(exact, conversationId)) return {
4723
+ if (exact && sameScope(exact, scopeKey())) return {
3442
4724
  error: null,
3443
4725
  meta: exact
3444
4726
  };
3445
- const matches = [...tasks.values()].filter((t) => sameConversation(t, conversationId) && t.id.startsWith(taskId));
4727
+ const matches = [...tasks.values()].filter((t) => sameScope(t, scopeKey()) && t.id.startsWith(taskId));
3446
4728
  if (matches.length === 1) return {
3447
4729
  error: null,
3448
4730
  meta: matches[0]
@@ -3451,7 +4733,7 @@ This is a background-task completion, not a message from the user. If it needs n
3451
4733
  return err(`Unknown task ${taskId}. Known tasks: ${knownTaskIds()}`);
3452
4734
  }
3453
4735
  function listTasks() {
3454
- const mine = [...tasks.values()].filter((t) => sameConversation(t, conversationId));
4736
+ const mine = [...tasks.values()].filter((t) => sameScope(t, scopeKey()));
3455
4737
  if (mine.length === 0) return "No background tasks.";
3456
4738
  return mine.map((t) => {
3457
4739
  const state = t.running ? "running" : t.exitCode !== null ? `exited ${t.exitCode}${t.killedReason ? ` (killed: ${t.killedReason})` : ""}` : t.killedReason ? `killed: ${t.killedReason}` : "ended";
@@ -3464,24 +4746,27 @@ This is a background-task completion, not a message from the user. If it needs n
3464
4746
  const bgRun = {
3465
4747
  name: "bg_run",
3466
4748
  label: "Run in background",
3467
- description: "Run a bash command in the background. Returns immediately with a task id; output streams to a log file. You are sent a message when it finishes — keep working or end your turn meanwhile. Use for anything over a couple of minutes (builds, batch jobs, retry loops, downloads). Inspect with bg_status / bg_logs, stop with bg_kill.",
3468
- promptSnippet: "bg_run — run a long command without blocking; you are notified on completion",
4749
+ description: "Run a bash command in the background. Returns immediately with a task id; output streams to a log file. You are sent a message when it finishes — keep working or end your turn meanwhile. Use for anything over a couple of minutes (builds, batch jobs, retry loops, downloads). Inspect with bg_status / bg_logs, stop with bg_kill. Pass resumable:true ONLY for an idempotent, long-lived command (a poller/watcher/retry loop) that should be requeued and relaunched on your next run if the sandbox goes away before it finishes (a sandbox is drained and replaced with a fresh one, not restarted in place) — the requeue re-runs the command from scratch, so never mark a one-shot side-effecting job (a migration, an apply, a send) resumable.",
4750
+ promptSnippet: "bg_run — run a long command without blocking; you are notified on completion (resumable:true requeues the task onto a fresh sandbox if the current one is drained, idempotent commands only)",
3469
4751
  parameters: Type.Object({
3470
4752
  command: Type.String({ description: "Bash command to execute" }),
3471
- description: Type.Optional(Type.String({ description: "Clear, concise description of what this command does in active voice (2-6 words)." }))
4753
+ description: Type.Optional(Type.String({ description: "Clear, concise description of what this command does in active voice (2-6 words)." })),
4754
+ resumable: Type.Optional(Type.Boolean({ description: "When true, this task is requeued and relaunched on your next run if the sandbox it is running on goes away before it finishes (a sandbox is drained and replaced with a fresh one, rather than restarted in place, so an in-flight task would otherwise be lost). Use ONLY for idempotent, long-lived commands (pollers, watchers, retry loops) — the requeue re-runs the command from scratch on the new sandbox, so never set this on a one-shot side-effecting command." }))
3472
4755
  }),
3473
4756
  async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
3474
4757
  await ensureConversationId();
3475
- const { command, description = null } = params;
4758
+ const { command, description = null, resumable = false } = params;
3476
4759
  const meta = await launchTask({
3477
4760
  command,
3478
4761
  description,
3479
- cwd: ctx.cwd
4762
+ cwd: ctx.cwd,
4763
+ resumable
3480
4764
  });
4765
+ const resumeNote = resumable && meta.messageId ? "\nResumable: if this sandbox is drained before the task finishes, it will be requeued and relaunched from the start on your next run." : resumable ? "\nNote: resumable was requested but this session has no durable conversation, so it will NOT be requeued if the sandbox is drained." : "";
3481
4766
  return {
3482
4767
  content: [{
3483
4768
  type: "text",
3484
- text: `Started background task ${taskLabel(meta)}.\nlog: ${logPath(meta.id)}\nYou will get a message when it finishes. Check on it with bg_status {"taskId":"${meta.id}"}.`
4769
+ text: `Started background task ${taskLabel(meta)}.\nlog: ${logPath(meta.id)}\nYou will get a message when it finishes. Check on it with bg_status {"taskId":"${meta.id}"}.${resumeNote}`
3485
4770
  }],
3486
4771
  details: {}
3487
4772
  };
@@ -3601,6 +4886,7 @@ const all = [
3601
4886
  localToolsExtension,
3602
4887
  toolCallEnvExtension,
3603
4888
  bashDefaultTimeoutExtension,
4889
+ diskGuardExtension,
3604
4890
  toolCallSummaryExtension
3605
4891
  ];
3606
4892
  /**
@@ -3619,7 +4905,8 @@ function platformExtensions({ sessionId, channelContext }) {
3619
4905
  selfTraceExtension,
3620
4906
  createBackgroundTasksExtension({ channelContext }),
3621
4907
  createSubagentExtension({ channelContext }),
3622
- createContextManagementExtension()
4908
+ createContextManagementExtension(),
4909
+ resourcePressureWarningExtension
3623
4910
  ];
3624
4911
  }
3625
4912
  //#endregion