@themoltnet/agent-daemon 0.4.0 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/main.js +947 -188
- package/package.json +6 -6
package/dist/main.js
CHANGED
|
@@ -9,9 +9,9 @@ import { createHash as createHash$1 } from "node:crypto";
|
|
|
9
9
|
import path, { dirname, isAbsolute, join, resolve } from "node:path";
|
|
10
10
|
import { homedir } from "node:os";
|
|
11
11
|
import { execFile, execFileSync } from "node:child_process";
|
|
12
|
-
import { DefaultResourceLoader, SessionManager, createAgentSession, createBashToolDefinition, createEditToolDefinition, createReadToolDefinition, createWriteToolDefinition, defineTool } from "@earendil-works/pi-coding-agent";
|
|
12
|
+
import { DefaultResourceLoader, SessionManager, createAgentSession, createBashToolDefinition, createEditToolDefinition, createReadToolDefinition, createSyntheticSourceInfo, createWriteToolDefinition, defineTool, parseFrontmatter } from "@earendil-works/pi-coding-agent";
|
|
13
13
|
import { Type, getModel } from "@earendil-works/pi-ai";
|
|
14
|
-
import { RealFSProvider, ShadowProvider, VM, VmCheckpoint, createHttpHooks, createShadowPathPredicate, ensureImageSelector, loadGuestAssets } from "@earendil-works/gondolin";
|
|
14
|
+
import { MemoryProvider, RealFSProvider, ShadowProvider, VM, VmCheckpoint, createHttpHooks, createShadowPathPredicate, ensureImageSelector, loadGuestAssets } from "@earendil-works/gondolin";
|
|
15
15
|
import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto";
|
|
16
16
|
import { resourceFromAttributes } from "@opentelemetry/resources";
|
|
17
17
|
import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base";
|
|
@@ -2886,6 +2886,55 @@ var UUID_RE = /^[0-9a-f]{8}-[0-9a-f]{4}-[1-5][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a
|
|
|
2886
2886
|
if (!Has$1("uuid")) Set$1("uuid", (v) => UUID_RE.test(v));
|
|
2887
2887
|
if (!Has$1("date-time")) Set$1("date-time", (v) => !Number.isNaN(Date.parse(v)));
|
|
2888
2888
|
//#endregion
|
|
2889
|
+
//#region ../../libs/tasks/src/context.ts
|
|
2890
|
+
/**
|
|
2891
|
+
* How an executor delivers a context entry to its underlying LLM.
|
|
2892
|
+
* V1 bindings only; Tier-2 (reference_file, mcp_resource, imported_file,
|
|
2893
|
+
* tool_response_seed, additional_context_hook) ship in a later slice.
|
|
2894
|
+
*/
|
|
2895
|
+
var ContextBinding = Type$2.Union([
|
|
2896
|
+
Type$2.Literal("skill"),
|
|
2897
|
+
Type$2.Literal("prompt_prefix"),
|
|
2898
|
+
Type$2.Literal("user_inline")
|
|
2899
|
+
], { $id: "ContextBinding" });
|
|
2900
|
+
/**
|
|
2901
|
+
* One context entry. Bytes are inlined: the imposer chose them, and the
|
|
2902
|
+
* task's `inputCid` already pins the entire input — including
|
|
2903
|
+
* `context[]` — so we don't need a separate per-entry hash, fetcher, or
|
|
2904
|
+
* flagged-content gate. Tasks reference rendered packs (or any other
|
|
2905
|
+
* external content) by copying their bytes into `content` at task
|
|
2906
|
+
* creation time.
|
|
2907
|
+
*
|
|
2908
|
+
* - `slug` — short identifier the daemon uses to disambiguate
|
|
2909
|
+
* entries. For `skill` binding it becomes the directory
|
|
2910
|
+
* name under the runtime's skill discovery path. Must be
|
|
2911
|
+
* kebab-case-safe (alphanumeric + dashes/underscores).
|
|
2912
|
+
* - `binding` — how the bytes are delivered to the LLM (see above).
|
|
2913
|
+
* - `content` — the actual bytes (UTF-8 text). Capped at 32 KiB per
|
|
2914
|
+
* entry; total per-task context bytes are bounded by the
|
|
2915
|
+
* soft `maxItems` cap and per-binding daemon limits.
|
|
2916
|
+
*/
|
|
2917
|
+
var ContextRef = Type$2.Object({
|
|
2918
|
+
slug: Type$2.String({
|
|
2919
|
+
minLength: 1,
|
|
2920
|
+
maxLength: 64,
|
|
2921
|
+
pattern: "^[a-zA-Z0-9_-]+$"
|
|
2922
|
+
}),
|
|
2923
|
+
binding: ContextBinding,
|
|
2924
|
+
content: Type$2.String({
|
|
2925
|
+
minLength: 1,
|
|
2926
|
+
maxLength: 32768
|
|
2927
|
+
})
|
|
2928
|
+
}, {
|
|
2929
|
+
$id: "ContextRef",
|
|
2930
|
+
additionalProperties: false
|
|
2931
|
+
});
|
|
2932
|
+
/** Reusable input fragment for any task type. Soft cap at 5 items. */
|
|
2933
|
+
var TaskContext = Type$2.Array(ContextRef, {
|
|
2934
|
+
$id: "TaskContext",
|
|
2935
|
+
maxItems: 5
|
|
2936
|
+
});
|
|
2937
|
+
//#endregion
|
|
2889
2938
|
//#region ../../libs/tasks/src/rubric.ts
|
|
2890
2939
|
/**
|
|
2891
2940
|
* Rubric — structured acceptance criteria used by judgment tasks.
|
|
@@ -4275,6 +4324,60 @@ var RenderPackOutput = Type$2.Object({
|
|
|
4275
4324
|
additionalProperties: false
|
|
4276
4325
|
});
|
|
4277
4326
|
//#endregion
|
|
4327
|
+
//#region ../../libs/tasks/src/task-types/run-eval.ts
|
|
4328
|
+
/**
|
|
4329
|
+
* `run_eval` — execute a scenario prompt under a named variant for
|
|
4330
|
+
* later cross-variant grading by `judge_eval_variant` (Slice 2).
|
|
4331
|
+
*
|
|
4332
|
+
* output_kind: artifact
|
|
4333
|
+
* criteria: optional (when set, output.verification is required —
|
|
4334
|
+
* producer self-assessment; the judge is the binding evaluator)
|
|
4335
|
+
* references: not required (scenario lives entirely in input)
|
|
4336
|
+
*/
|
|
4337
|
+
var RUN_EVAL_TYPE = "run_eval";
|
|
4338
|
+
var RunEvalInput = Type$2.Object({
|
|
4339
|
+
scenario: Type$2.Object({
|
|
4340
|
+
prompt: Type$2.String({ minLength: 1 }),
|
|
4341
|
+
inputFiles: Type$2.Optional(Type$2.Array(Type$2.String({ minLength: 1 })))
|
|
4342
|
+
}, { additionalProperties: false }),
|
|
4343
|
+
variantLabel: Type$2.String({
|
|
4344
|
+
minLength: 1,
|
|
4345
|
+
maxLength: 64
|
|
4346
|
+
}),
|
|
4347
|
+
context: TaskContext,
|
|
4348
|
+
successCriteria: Type$2.Optional(SuccessCriteria)
|
|
4349
|
+
}, {
|
|
4350
|
+
$id: "RunEvalInput",
|
|
4351
|
+
additionalProperties: false
|
|
4352
|
+
});
|
|
4353
|
+
var RunEvalOutput = Type$2.Object({
|
|
4354
|
+
response: Type$2.String({ minLength: 1 }),
|
|
4355
|
+
artifacts: Type$2.Optional(Type$2.Array(Type$2.Object({
|
|
4356
|
+
path: Type$2.String({ minLength: 1 }),
|
|
4357
|
+
cid: Type$2.String({ minLength: 1 })
|
|
4358
|
+
}, { additionalProperties: false }))),
|
|
4359
|
+
totalTokens: Type$2.Integer({ minimum: 0 }),
|
|
4360
|
+
durationMs: Type$2.Integer({ minimum: 0 }),
|
|
4361
|
+
traceparent: Type$2.String({ minLength: 1 }),
|
|
4362
|
+
verification: Type$2.Optional(VerificationRecord)
|
|
4363
|
+
}, {
|
|
4364
|
+
$id: "RunEvalOutput",
|
|
4365
|
+
additionalProperties: false
|
|
4366
|
+
});
|
|
4367
|
+
/**
|
|
4368
|
+
* Cross-field rule mirroring the `requireVerificationWhenCriteriaPresent`
|
|
4369
|
+
* rule used by the brief task types: when input declares
|
|
4370
|
+
* `successCriteria`, output MUST carry `verification`; when it doesn't,
|
|
4371
|
+
* output MUST NOT carry one.
|
|
4372
|
+
*/
|
|
4373
|
+
function validateRunEvalOutput(output, input) {
|
|
4374
|
+
const hasCriteria = input !== null && input !== void 0 && input.successCriteria !== void 0;
|
|
4375
|
+
const hasVerification = output !== null && output !== void 0 && output.verification !== void 0;
|
|
4376
|
+
if (hasCriteria && !hasVerification) return "output.verification is required because input.successCriteria is set; the producer LLM must self-assess against the criteria";
|
|
4377
|
+
if (!hasCriteria && hasVerification) return "output.verification was supplied but input.successCriteria is unset; omit verification when there are no criteria to assess against";
|
|
4378
|
+
return null;
|
|
4379
|
+
}
|
|
4380
|
+
//#endregion
|
|
4278
4381
|
//#region ../../libs/tasks/src/task-types/index.ts
|
|
4279
4382
|
/**
|
|
4280
4383
|
* Validate that a judgment-task input carries a rubric inside its
|
|
@@ -4353,6 +4456,14 @@ var BUILT_IN_TASK_TYPES = {
|
|
|
4353
4456
|
requiresReferences: true,
|
|
4354
4457
|
validateInput: validateJudgmentInput,
|
|
4355
4458
|
validateOutput: validateJudgePackOutput
|
|
4459
|
+
},
|
|
4460
|
+
[RUN_EVAL_TYPE]: {
|
|
4461
|
+
name: RUN_EVAL_TYPE,
|
|
4462
|
+
inputSchema: RunEvalInput,
|
|
4463
|
+
outputSchema: RunEvalOutput,
|
|
4464
|
+
outputKind: "artifact",
|
|
4465
|
+
requiresReferences: false,
|
|
4466
|
+
validateOutput: validateRunEvalOutput
|
|
4356
4467
|
}
|
|
4357
4468
|
};
|
|
4358
4469
|
//#endregion
|
|
@@ -5251,6 +5362,15 @@ function validateTaskOutput(taskType, output, input) {
|
|
|
5251
5362
|
function getTaskOutputSchema(taskType) {
|
|
5252
5363
|
return getTaskTypeEntry(taskType)?.outputSchema ?? null;
|
|
5253
5364
|
}
|
|
5365
|
+
/**
|
|
5366
|
+
* Whether sessions running this task type should have the generic
|
|
5367
|
+
* `subagent` custom tool registered. Returns `false` for unknown task
|
|
5368
|
+
* types and for task types that didn't opt in. See `TaskTypeEntry`
|
|
5369
|
+
* for the design rationale.
|
|
5370
|
+
*/
|
|
5371
|
+
function taskTypeUsesSubagents(taskType) {
|
|
5372
|
+
return getTaskTypeEntry(taskType)?.usesSubagents === true;
|
|
5373
|
+
}
|
|
5254
5374
|
//#endregion
|
|
5255
5375
|
//#region ../../libs/tasks/src/wire.ts
|
|
5256
5376
|
/**
|
|
@@ -5295,6 +5415,14 @@ var ExecutorTrustLevel = Type$2.Union([
|
|
|
5295
5415
|
Type$2.Literal("releaseVerifiedTool"),
|
|
5296
5416
|
Type$2.Literal("sandboxAttested")
|
|
5297
5417
|
], { $id: "ExecutorTrustLevel" });
|
|
5418
|
+
/** Identifies a (provider, model) daemon pair allowed to claim a task. */
|
|
5419
|
+
var ExecutorRef = Type$2.Object({
|
|
5420
|
+
provider: Type$2.String({ minLength: 1 }),
|
|
5421
|
+
model: Type$2.String({ minLength: 1 })
|
|
5422
|
+
}, {
|
|
5423
|
+
$id: "ExecutorRef",
|
|
5424
|
+
additionalProperties: false
|
|
5425
|
+
});
|
|
5298
5426
|
var OutputKind = Type$2.Union([Type$2.Literal("artifact"), Type$2.Literal("judgment")], { $id: "OutputKind" });
|
|
5299
5427
|
var TaskMessageKind = Type$2.Union([
|
|
5300
5428
|
Type$2.Literal("text_delta"),
|
|
@@ -5387,6 +5515,7 @@ Type$2.Object({
|
|
|
5387
5515
|
imposedByHumanId: Type$2.Union([Uuid, Type$2.Null()]),
|
|
5388
5516
|
acceptedAttemptN: Type$2.Union([Type$2.Number(), Type$2.Null()]),
|
|
5389
5517
|
requiredExecutorTrustLevel: ExecutorTrustLevel,
|
|
5518
|
+
allowedExecutors: Type$2.Array(ExecutorRef, { maxItems: 16 }),
|
|
5390
5519
|
status: TaskStatus,
|
|
5391
5520
|
queuedAt: IsoTimestamp,
|
|
5392
5521
|
completedAt: Type$2.Union([IsoTimestamp, Type$2.Null()]),
|
|
@@ -5608,6 +5737,61 @@ function isHelpFlag(args) {
|
|
|
5608
5737
|
return args.includes("--help") || args.includes("-h");
|
|
5609
5738
|
}
|
|
5610
5739
|
//#endregion
|
|
5740
|
+
//#region ../../libs/agent-runtime/src/context-bindings.ts
|
|
5741
|
+
var PROMPT_SEPARATOR = "\n\n---\n\n";
|
|
5742
|
+
/**
|
|
5743
|
+
* Resolve `task.input.context[]` into delivered side-effects (skills
|
|
5744
|
+
* persisted via `deliver.skill`) and prompt fragments
|
|
5745
|
+
* (`systemPromptPrefix`, `userInlineSuffix`) the caller weaves into the
|
|
5746
|
+
* built prompt.
|
|
5747
|
+
*
|
|
5748
|
+
* Per-binding semantics (V1):
|
|
5749
|
+
* - `skill` → `deliver.skill({ slug, content })` once per ref.
|
|
5750
|
+
* Slug collisions on distinct contents are
|
|
5751
|
+
* refused loudly.
|
|
5752
|
+
* - `prompt_prefix` → content appended to `systemPromptPrefix` with
|
|
5753
|
+
* the canonical `\n\n---\n\n` separator (in
|
|
5754
|
+
* declared order).
|
|
5755
|
+
* - `user_inline` → content appended to `userInlineSuffix` in
|
|
5756
|
+
* declared order, same separator.
|
|
5757
|
+
*
|
|
5758
|
+
* No fetching, no hashing — bytes are inlined in `ContextRef.content`,
|
|
5759
|
+
* and the task's `inputCid` already pins the entire input. The imposer
|
|
5760
|
+
* chose these bytes; the resolver just dispatches them.
|
|
5761
|
+
*
|
|
5762
|
+
* The function is pure with respect to its arguments: file writes are
|
|
5763
|
+
* confined to the injected `deliver` callback, which makes the
|
|
5764
|
+
* resolver trivial to test.
|
|
5765
|
+
*/
|
|
5766
|
+
async function resolveTaskContext(args) {
|
|
5767
|
+
const promptParts = [];
|
|
5768
|
+
const userParts = [];
|
|
5769
|
+
const injected = [];
|
|
5770
|
+
const usedSlugs = /* @__PURE__ */ new Map();
|
|
5771
|
+
for (const ref of args.context) {
|
|
5772
|
+
if (ref.binding === "skill") {
|
|
5773
|
+
const prior = usedSlugs.get(ref.slug);
|
|
5774
|
+
if (prior !== void 0) {
|
|
5775
|
+
if (prior !== ref.content) throw new Error(`slug collision on '${ref.slug}': two skill entries share the same slug but have different content`);
|
|
5776
|
+
injected.push(ref);
|
|
5777
|
+
continue;
|
|
5778
|
+
}
|
|
5779
|
+
usedSlugs.set(ref.slug, ref.content);
|
|
5780
|
+
await args.deliver.skill({
|
|
5781
|
+
slug: ref.slug,
|
|
5782
|
+
content: ref.content
|
|
5783
|
+
});
|
|
5784
|
+
} else if (ref.binding === "prompt_prefix") promptParts.push(ref.content);
|
|
5785
|
+
else userParts.push(ref.content);
|
|
5786
|
+
injected.push(ref);
|
|
5787
|
+
}
|
|
5788
|
+
return {
|
|
5789
|
+
injected,
|
|
5790
|
+
systemPromptPrefix: promptParts.join(PROMPT_SEPARATOR),
|
|
5791
|
+
userInlineSuffix: userParts.join(PROMPT_SEPARATOR)
|
|
5792
|
+
};
|
|
5793
|
+
}
|
|
5794
|
+
//#endregion
|
|
5611
5795
|
//#region ../../libs/agent-runtime/src/output-tools.ts
|
|
5612
5796
|
/**
|
|
5613
5797
|
* Submit-output tool contract.
|
|
@@ -5702,7 +5886,7 @@ function buildFinalOutputBlock(opts) {
|
|
|
5702
5886
|
//#endregion
|
|
5703
5887
|
//#region ../../libs/agent-runtime/src/prompts/assess-brief.ts
|
|
5704
5888
|
/**
|
|
5705
|
-
* Build the
|
|
5889
|
+
* Build the first user-message prompt for an `assess_brief` judge attempt.
|
|
5706
5890
|
*
|
|
5707
5891
|
* Design note — no pre-resolved `target` projection
|
|
5708
5892
|
* --------------------------------------------------
|
|
@@ -5723,7 +5907,7 @@ function buildFinalOutputBlock(opts) {
|
|
|
5723
5907
|
* future task types whose products are docs / configs / changes /
|
|
5724
5908
|
* anything) work without any code path here.
|
|
5725
5909
|
*/
|
|
5726
|
-
function
|
|
5910
|
+
function buildAssessBriefUserPrompt(input, ctx) {
|
|
5727
5911
|
const rubric = input.successCriteria.rubric;
|
|
5728
5912
|
const criteriaList = rubric.criteria.map((c, i) => `${i + 1}. **${c.id}** (weight ${c.weight}, scoring: \`${c.scoring}\`) — ${c.description}`).join("\n");
|
|
5729
5913
|
const preambleSection = rubric.preamble ? [
|
|
@@ -5838,7 +6022,7 @@ function buildSelfVerificationBlock(taskId) {
|
|
|
5838
6022
|
//#endregion
|
|
5839
6023
|
//#region ../../libs/agent-runtime/src/prompts/curate-pack.ts
|
|
5840
6024
|
/**
|
|
5841
|
-
* Build the
|
|
6025
|
+
* Build the first user-message prompt for a `curate_pack` task.
|
|
5842
6026
|
*
|
|
5843
6027
|
* Design note: this prompt is deliberately NOT a numbered command
|
|
5844
6028
|
* sequence. The curator's value comes from judgment — inferring scope
|
|
@@ -5859,7 +6043,7 @@ function buildSelfVerificationBlock(taskId) {
|
|
|
5859
6043
|
* emits pruned state at phase boundaries so a follow-up session can
|
|
5860
6044
|
* resume without replaying the tool history.
|
|
5861
6045
|
*/
|
|
5862
|
-
function
|
|
6046
|
+
function buildCuratePackUserPrompt(input, ctx) {
|
|
5863
6047
|
const { diaryId, taskPrompt, entryTypes, tagFilters, tokenBudget, recipe } = input;
|
|
5864
6048
|
const entryTypesPinned = Boolean(entryTypes);
|
|
5865
6049
|
const resolvedRecipe = recipe ?? "topic-focused-v1";
|
|
@@ -5995,13 +6179,13 @@ function buildCuratePackPrompt(input, ctx) {
|
|
|
5995
6179
|
//#endregion
|
|
5996
6180
|
//#region ../../libs/agent-runtime/src/prompts/fulfill-brief.ts
|
|
5997
6181
|
/**
|
|
5998
|
-
* Build the
|
|
6182
|
+
* Build the first user-message prompt for a `fulfill_brief` task.
|
|
5999
6183
|
*
|
|
6000
6184
|
* Generalized from the original `resolve-issue` prompt. No longer
|
|
6001
6185
|
* GitHub-specific; references live on `Task.references[]` and the agent
|
|
6002
6186
|
* is told to inspect them itself.
|
|
6003
6187
|
*/
|
|
6004
|
-
function
|
|
6188
|
+
function buildFulfillBriefUserPrompt(input, ctx) {
|
|
6005
6189
|
const { brief, title, acceptanceCriteria, seedFiles, scopeHint } = input;
|
|
6006
6190
|
const criteriaSection = acceptanceCriteria?.length ? [
|
|
6007
6191
|
"### Acceptance criteria",
|
|
@@ -6081,7 +6265,7 @@ function buildFulfillBriefPrompt(input, ctx) {
|
|
|
6081
6265
|
}
|
|
6082
6266
|
//#endregion
|
|
6083
6267
|
//#region ../../libs/agent-runtime/src/prompts/judge-pack.ts
|
|
6084
|
-
function
|
|
6268
|
+
function buildJudgePackUserPrompt(input, ctx) {
|
|
6085
6269
|
const { renderedPackId, sourcePackId, successCriteria } = input;
|
|
6086
6270
|
const rubric = successCriteria.rubric;
|
|
6087
6271
|
const criteriaList = rubric.criteria.map((c, i) => `${i + 1}. **${c.id}** (weight ${c.weight}, scoring: \`${c.scoring}\`) — ${c.description}`).join("\n");
|
|
@@ -6208,10 +6392,10 @@ function buildJudgePackPrompt(input, ctx) {
|
|
|
6208
6392
|
//#endregion
|
|
6209
6393
|
//#region ../../libs/agent-runtime/src/prompts/render-pack.ts
|
|
6210
6394
|
/**
|
|
6211
|
-
* Build the
|
|
6395
|
+
* Build the first user-message prompt for a `render_pack` task. Almost mechanical:
|
|
6212
6396
|
* wraps `moltnet_pack_render` and emits the receipt.
|
|
6213
6397
|
*/
|
|
6214
|
-
function
|
|
6398
|
+
function buildRenderPackUserPrompt(input, ctx) {
|
|
6215
6399
|
const { packId, persist = true, pinned = false } = input;
|
|
6216
6400
|
return [
|
|
6217
6401
|
"# Render Pack Agent",
|
|
@@ -6265,19 +6449,87 @@ function buildRenderPackPrompt(input, ctx) {
|
|
|
6265
6449
|
].join("\n");
|
|
6266
6450
|
}
|
|
6267
6451
|
//#endregion
|
|
6452
|
+
//#region ../../libs/agent-runtime/src/prompts/run-eval.ts
|
|
6453
|
+
/**
|
|
6454
|
+
* Build the first user-message prompt for a `run_eval` task.
|
|
6455
|
+
*
|
|
6456
|
+
* Free-form: no git workflow, no commit ceremony. The executor produces
|
|
6457
|
+
* a textual response (and optional file artifacts) that a later
|
|
6458
|
+
* `judge_eval_variant` task (Slice 2) grades against the rubric.
|
|
6459
|
+
*
|
|
6460
|
+
* Context delivery is handled by `resolveTaskContext` (see
|
|
6461
|
+
* libs/agent-runtime/src/context-bindings.ts) and runs BEFORE this
|
|
6462
|
+
* prompt is rendered: `prompt_prefix` items are concatenated ahead of
|
|
6463
|
+
* the body, `skill` items are persisted at the runtime's skill path,
|
|
6464
|
+
* and `user_inline` items are appended to the first user message. This
|
|
6465
|
+
* builder does NOT inline `input.context[]` itself.
|
|
6466
|
+
*/
|
|
6467
|
+
function buildRunEvalUserPrompt(input, ctx) {
|
|
6468
|
+
const { scenario, variantLabel, successCriteria } = input;
|
|
6469
|
+
const inputFilesSection = scenario.inputFiles?.length ? [
|
|
6470
|
+
"### Input files",
|
|
6471
|
+
"",
|
|
6472
|
+
...scenario.inputFiles.map((f) => `- \`${f}\``),
|
|
6473
|
+
""
|
|
6474
|
+
].join("\n") : "";
|
|
6475
|
+
const verificationSection = successCriteria ? buildSelfVerificationBlock(ctx.taskId) : "";
|
|
6476
|
+
const correlationSection = ctx.correlationId ? [
|
|
6477
|
+
"### Correlation",
|
|
6478
|
+
"",
|
|
6479
|
+
`This task carries correlationId \`${ctx.correlationId}\`. It joins`,
|
|
6480
|
+
"this variant to its sibling `run_eval` tasks (other variants of the",
|
|
6481
|
+
"same scenario) and to the eventual `judge_eval_variant` task that",
|
|
6482
|
+
"will grade them together. You do not need to act on it directly —",
|
|
6483
|
+
"it is recorded for cross-variant aggregation at query time.",
|
|
6484
|
+
""
|
|
6485
|
+
].join("\n") : "";
|
|
6486
|
+
const finalOutputBlock = buildFinalOutputBlock({
|
|
6487
|
+
taskType: "run_eval",
|
|
6488
|
+
outputSchemaName: "RunEvalOutput",
|
|
6489
|
+
shapeSketch: [
|
|
6490
|
+
"{",
|
|
6491
|
+
" \"response\": \"<your free-form answer>\",",
|
|
6492
|
+
" \"artifacts\": [{ \"path\": \"...\", \"cid\": \"...\" }], // optional",
|
|
6493
|
+
" \"totalTokens\": <int>,",
|
|
6494
|
+
" \"durationMs\": <int>,",
|
|
6495
|
+
" \"traceparent\": \"<from claim>\",",
|
|
6496
|
+
" \"verification\": <required iff input.successCriteria; see Self-verification>",
|
|
6497
|
+
"}"
|
|
6498
|
+
].join("\n")
|
|
6499
|
+
});
|
|
6500
|
+
return [
|
|
6501
|
+
"# Run Eval Agent\n",
|
|
6502
|
+
`You are running an evaluation scenario as variant \`${variantLabel}\`.\nTask id: \`${ctx.taskId}\`\n`,
|
|
6503
|
+
correlationSection,
|
|
6504
|
+
`### Scenario\n\n${scenario.prompt}\n`,
|
|
6505
|
+
inputFilesSection,
|
|
6506
|
+
verificationSection,
|
|
6507
|
+
finalOutputBlock
|
|
6508
|
+
].filter((s) => s !== "").join("\n");
|
|
6509
|
+
}
|
|
6510
|
+
//#endregion
|
|
6268
6511
|
//#region ../../libs/agent-runtime/src/prompts/index.ts
|
|
6269
6512
|
/**
|
|
6270
|
-
* Resolve the correct prompt builder for `task.taskType` and
|
|
6271
|
-
* Throws if the type is unknown or the input fails TypeBox
|
|
6272
|
-
|
|
6273
|
-
|
|
6513
|
+
* Resolve the correct user-prompt builder for `task.taskType` and
|
|
6514
|
+
* invoke it. Throws if the type is unknown or the input fails TypeBox
|
|
6515
|
+
* validation.
|
|
6516
|
+
*
|
|
6517
|
+
* Role note: the returned string is delivered as the **first user
|
|
6518
|
+
* message** of the agent's session (pi-coding-agent's
|
|
6519
|
+
* `session.prompt(text)` puts text in the user role). The system
|
|
6520
|
+
* prompt is built separately by pi from `appendSystemPrompt` (the
|
|
6521
|
+
* runtime instructor lives there). Builders here are free-form Markdown
|
|
6522
|
+
* for the user turn; they don't replace or prepend to the system
|
|
6523
|
+
* prompt.
|
|
6524
|
+
*/
|
|
6525
|
+
function buildTaskUserPrompt(task, ctx) {
|
|
6274
6526
|
switch (task.taskType) {
|
|
6275
6527
|
case FULFILL_BRIEF_TYPE:
|
|
6276
6528
|
if (!Check(FulfillBriefInput, task.input)) {
|
|
6277
6529
|
const errors = [...Errors(FulfillBriefInput, task.input)];
|
|
6278
6530
|
throw new Error(`fulfill_brief input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
|
|
6279
6531
|
}
|
|
6280
|
-
return
|
|
6532
|
+
return buildFulfillBriefUserPrompt(task.input, {
|
|
6281
6533
|
diaryId: ctx.diaryId,
|
|
6282
6534
|
taskId: ctx.taskId,
|
|
6283
6535
|
correlationId: task.correlationId
|
|
@@ -6287,7 +6539,7 @@ function buildPromptForTask(task, ctx) {
|
|
|
6287
6539
|
const errors = [...Errors(AssessBriefInput, task.input)];
|
|
6288
6540
|
throw new Error(`assess_brief input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
|
|
6289
6541
|
}
|
|
6290
|
-
return
|
|
6542
|
+
return buildAssessBriefUserPrompt(task.input, {
|
|
6291
6543
|
diaryId: ctx.diaryId,
|
|
6292
6544
|
taskId: ctx.taskId
|
|
6293
6545
|
});
|
|
@@ -6296,7 +6548,7 @@ function buildPromptForTask(task, ctx) {
|
|
|
6296
6548
|
const errors = [...Errors(CuratePackInput, task.input)];
|
|
6297
6549
|
throw new Error(`curate_pack input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
|
|
6298
6550
|
}
|
|
6299
|
-
return
|
|
6551
|
+
return buildCuratePackUserPrompt(task.input, {
|
|
6300
6552
|
diaryId: ctx.diaryId,
|
|
6301
6553
|
taskId: ctx.taskId
|
|
6302
6554
|
});
|
|
@@ -6305,7 +6557,7 @@ function buildPromptForTask(task, ctx) {
|
|
|
6305
6557
|
const errors = [...Errors(RenderPackInput, task.input)];
|
|
6306
6558
|
throw new Error(`render_pack input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
|
|
6307
6559
|
}
|
|
6308
|
-
return
|
|
6560
|
+
return buildRenderPackUserPrompt(task.input, {
|
|
6309
6561
|
diaryId: ctx.diaryId,
|
|
6310
6562
|
taskId: ctx.taskId
|
|
6311
6563
|
});
|
|
@@ -6314,10 +6566,20 @@ function buildPromptForTask(task, ctx) {
|
|
|
6314
6566
|
const errors = [...Errors(JudgePackInput, task.input)];
|
|
6315
6567
|
throw new Error(`judge_pack input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
|
|
6316
6568
|
}
|
|
6317
|
-
return
|
|
6569
|
+
return buildJudgePackUserPrompt(task.input, {
|
|
6318
6570
|
diaryId: ctx.diaryId,
|
|
6319
6571
|
taskId: ctx.taskId
|
|
6320
6572
|
});
|
|
6573
|
+
case RUN_EVAL_TYPE:
|
|
6574
|
+
if (!Check(RunEvalInput, task.input)) {
|
|
6575
|
+
const errors = [...Errors(RunEvalInput, task.input)];
|
|
6576
|
+
throw new Error(`run_eval input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
|
|
6577
|
+
}
|
|
6578
|
+
return buildRunEvalUserPrompt(task.input, {
|
|
6579
|
+
diaryId: ctx.diaryId,
|
|
6580
|
+
taskId: ctx.taskId,
|
|
6581
|
+
correlationId: task.correlationId
|
|
6582
|
+
});
|
|
6321
6583
|
default: throw new Error(`No prompt builder registered for taskType="${task.taskType}"`);
|
|
6322
6584
|
}
|
|
6323
6585
|
}
|
|
@@ -9098,13 +9360,31 @@ function problemToError(problem, statusCode) {
|
|
|
9098
9360
|
//#endregion
|
|
9099
9361
|
//#region ../../libs/sdk/src/agent-context.ts
|
|
9100
9362
|
function unwrapResult(result) {
|
|
9101
|
-
if (result.error) {
|
|
9363
|
+
if (result.error !== void 0 && result.error !== null) {
|
|
9102
9364
|
const error = result.error;
|
|
9103
|
-
throw problemToError(error, error.status
|
|
9365
|
+
if (isProblemDetails(error)) throw problemToError(error, error.status);
|
|
9366
|
+
if (error instanceof Error && result.response === void 0) {
|
|
9367
|
+
const networkError = new NetworkError(error.message, { detail: error.cause ? stringifyUnknown(error.cause) : void 0 });
|
|
9368
|
+
networkError.stack = error.stack;
|
|
9369
|
+
throw networkError;
|
|
9370
|
+
}
|
|
9371
|
+
throw new MoltNetError(`Unexpected error from MoltNet API: ${stringifyUnknown(error)}`, { code: "UNKNOWN" });
|
|
9104
9372
|
}
|
|
9105
9373
|
if (result.data === void 0) throw new MoltNetError("Unexpected empty response from MoltNet API", { code: "EMPTY_RESPONSE" });
|
|
9106
9374
|
return result.data;
|
|
9107
9375
|
}
|
|
9376
|
+
function isProblemDetails(error) {
|
|
9377
|
+
if (!error || typeof error !== "object") return false;
|
|
9378
|
+
return typeof error.status === "number" && ("title" in error || "detail" in error);
|
|
9379
|
+
}
|
|
9380
|
+
function stringifyUnknown(value) {
|
|
9381
|
+
if (value instanceof Error) return `${value.name}: ${value.message}`;
|
|
9382
|
+
try {
|
|
9383
|
+
return JSON.stringify(value) ?? String(value);
|
|
9384
|
+
} catch {
|
|
9385
|
+
return String(value);
|
|
9386
|
+
}
|
|
9387
|
+
}
|
|
9108
9388
|
function unwrapRequired(result, message, code) {
|
|
9109
9389
|
if (result.error || !result.data) throw new MoltNetError(message, { code });
|
|
9110
9390
|
return result.data;
|
|
@@ -12870,6 +13150,7 @@ var PollingApiTaskSource = class {
|
|
|
12870
13150
|
this.minBackoffMs = opts.pollIntervalMs ?? DEFAULT_POLL_INTERVAL_MS;
|
|
12871
13151
|
this.maxBackoffMs = opts.maxPollIntervalMs ?? DEFAULT_MAX_POLL_INTERVAL_MS;
|
|
12872
13152
|
if (this.maxBackoffMs < this.minBackoffMs) throw new Error(`PollingApiTaskSource: maxPollIntervalMs (${this.maxBackoffMs}) must be >= pollIntervalMs (${this.minBackoffMs})`);
|
|
13153
|
+
if (Boolean(opts.provider) !== Boolean(opts.model)) throw new Error("PollingApiTaskSource: provider and model must be set together");
|
|
12873
13154
|
this.listLimit = opts.listLimit ?? DEFAULT_LIST_LIMIT;
|
|
12874
13155
|
this.currentBackoffMs = this.minBackoffMs;
|
|
12875
13156
|
this.logger = (opts.logger ?? pino({ name: "polling-api-source" })).child({ teamId: opts.teamId });
|
|
@@ -12903,6 +13184,10 @@ var PollingApiTaskSource = class {
|
|
|
12903
13184
|
teamId: this.opts.teamId,
|
|
12904
13185
|
status: "queued",
|
|
12905
13186
|
...taskType ? { taskType } : {},
|
|
13187
|
+
...this.opts.provider && this.opts.model ? {
|
|
13188
|
+
provider: this.opts.provider,
|
|
13189
|
+
model: this.opts.model
|
|
13190
|
+
} : {},
|
|
12906
13191
|
limit: this.listLimit
|
|
12907
13192
|
});
|
|
12908
13193
|
if (this.opts.debug) this.logger.debug({
|
|
@@ -12913,6 +13198,10 @@ var PollingApiTaskSource = class {
|
|
|
12913
13198
|
for (const item of result.items) {
|
|
12914
13199
|
if (seen.has(item.id)) continue;
|
|
12915
13200
|
if (this.opts.diaryIds && this.opts.diaryIds.length > 0 && (item.diaryId === null || !this.opts.diaryIds.includes(item.diaryId))) continue;
|
|
13201
|
+
if (this.opts.provider && this.opts.model) {
|
|
13202
|
+
const allowed = item.allowedExecutors ?? [];
|
|
13203
|
+
if (allowed.length > 0 && !allowed.some((e) => e.provider === this.opts.provider && e.model === this.opts.model)) continue;
|
|
13204
|
+
}
|
|
12916
13205
|
if (item.status !== "queued") continue;
|
|
12917
13206
|
seen.add(item.id);
|
|
12918
13207
|
out.push(item);
|
|
@@ -12987,6 +13276,25 @@ function abortableSleep(ms, signal) {
|
|
|
12987
13276
|
});
|
|
12988
13277
|
}
|
|
12989
13278
|
//#endregion
|
|
13279
|
+
//#region ../../libs/agent-runtime/src/subagent-output-contracts.ts
|
|
13280
|
+
var REGISTRY = /* @__PURE__ */ new Map();
|
|
13281
|
+
/**
|
|
13282
|
+
* Resolve a subagent output contract by name. Returns `null` for
|
|
13283
|
+
* unknown names — callers (the subagent custom tool) decide whether
|
|
13284
|
+
* that's a tool error the parent LLM can recover from or a hard fail.
|
|
13285
|
+
*/
|
|
13286
|
+
function getSubagentOutputContract(name) {
|
|
13287
|
+
return REGISTRY.get(name) ?? null;
|
|
13288
|
+
}
|
|
13289
|
+
/**
|
|
13290
|
+
* List all registered contracts. Useful for diagnostics and for the
|
|
13291
|
+
* subagent tool's parameter description so a parent LLM can see what
|
|
13292
|
+
* contracts are available without enumerating them in its prompt.
|
|
13293
|
+
*/
|
|
13294
|
+
function listSubagentOutputContracts() {
|
|
13295
|
+
return [...REGISTRY.values()];
|
|
13296
|
+
}
|
|
13297
|
+
//#endregion
|
|
12990
13298
|
//#region ../../libs/pi-extension/src/moltnet/render-phase6.ts
|
|
12991
13299
|
function slugToTitle(value) {
|
|
12992
13300
|
return value.split(/[:/_-]+/).filter(Boolean).map((part) => part[0]?.toUpperCase() + part.slice(1)).join(" ");
|
|
@@ -13906,138 +14214,29 @@ function pruneOldSnapshots(maxCached, currentDir) {
|
|
|
13906
14214
|
});
|
|
13907
14215
|
}
|
|
13908
14216
|
//#endregion
|
|
13909
|
-
//#region ../../libs/pi-extension/src/
|
|
13910
|
-
/**
|
|
13911
|
-
* Gondolin tool operations: redirect pi's built-in tool operations
|
|
13912
|
-
* (read, write, edit, bash) to execute inside the VM.
|
|
13913
|
-
*
|
|
13914
|
-
* Follows the same pattern as upstream pi-gondolin.ts — pi's tool factories
|
|
13915
|
-
* accept an `operations` object that provides the underlying I/O.
|
|
13916
|
-
*/
|
|
14217
|
+
//#region ../../libs/pi-extension/src/vm-manager.ts
|
|
13917
14218
|
var GUEST_WORKSPACE$1 = "/workspace";
|
|
13918
|
-
function shQuote(s) {
|
|
13919
|
-
return "'" + s.replace(/'/g, "'\\''") + "'";
|
|
13920
|
-
}
|
|
13921
14219
|
/**
|
|
13922
|
-
*
|
|
13923
|
-
*
|
|
13924
|
-
|
|
13925
|
-
|
|
13926
|
-
|
|
13927
|
-
|
|
13928
|
-
|
|
13929
|
-
|
|
13930
|
-
|
|
13931
|
-
|
|
13932
|
-
|
|
13933
|
-
|
|
13934
|
-
|
|
13935
|
-
|
|
13936
|
-
|
|
13937
|
-
|
|
13938
|
-
|
|
13939
|
-
|
|
13940
|
-
|
|
13941
|
-
|
|
13942
|
-
"/bin/sh",
|
|
13943
|
-
"-lc",
|
|
13944
|
-
`test -r ${shQuote(toGuestPath(localCwd, p))}`
|
|
13945
|
-
])).ok) throw new Error(`not readable: ${p}`);
|
|
13946
|
-
},
|
|
13947
|
-
detectImageMimeType: async (p) => {
|
|
13948
|
-
try {
|
|
13949
|
-
const r = await vm.exec([
|
|
13950
|
-
"/bin/sh",
|
|
13951
|
-
"-lc",
|
|
13952
|
-
`file --mime-type -b ${shQuote(toGuestPath(localCwd, p))}`
|
|
13953
|
-
]);
|
|
13954
|
-
if (!r.ok) return null;
|
|
13955
|
-
const m = r.stdout.trim();
|
|
13956
|
-
return [
|
|
13957
|
-
"image/jpeg",
|
|
13958
|
-
"image/png",
|
|
13959
|
-
"image/gif",
|
|
13960
|
-
"image/webp"
|
|
13961
|
-
].includes(m) ? m : null;
|
|
13962
|
-
} catch {
|
|
13963
|
-
return null;
|
|
13964
|
-
}
|
|
13965
|
-
}
|
|
13966
|
-
};
|
|
13967
|
-
}
|
|
13968
|
-
function createGondolinWriteOps(vm, localCwd) {
|
|
13969
|
-
return {
|
|
13970
|
-
writeFile: async (p, content) => {
|
|
13971
|
-
const guestPath = toGuestPath(localCwd, p);
|
|
13972
|
-
const dir = path.posix.dirname(guestPath);
|
|
13973
|
-
const b64 = Buffer.from(content, "utf8").toString("base64");
|
|
13974
|
-
const r = await vm.exec([
|
|
13975
|
-
"/bin/sh",
|
|
13976
|
-
"-lc",
|
|
13977
|
-
[
|
|
13978
|
-
"set -eu",
|
|
13979
|
-
`mkdir -p ${shQuote(dir)}`,
|
|
13980
|
-
`echo ${shQuote(b64)} | base64 -d > ${shQuote(guestPath)}`
|
|
13981
|
-
].join("\n")
|
|
13982
|
-
]);
|
|
13983
|
-
if (!r.ok) throw new Error(`write failed (${r.exitCode}): ${r.stderr}`);
|
|
13984
|
-
},
|
|
13985
|
-
mkdir: async (dir) => {
|
|
13986
|
-
const r = await vm.exec([
|
|
13987
|
-
"/bin/mkdir",
|
|
13988
|
-
"-p",
|
|
13989
|
-
toGuestPath(localCwd, dir)
|
|
13990
|
-
]);
|
|
13991
|
-
if (!r.ok) throw new Error(`mkdir failed (${r.exitCode}): ${r.stderr}`);
|
|
13992
|
-
}
|
|
13993
|
-
};
|
|
13994
|
-
}
|
|
13995
|
-
function createGondolinEditOps(vm, localCwd) {
|
|
13996
|
-
const r = createGondolinReadOps(vm, localCwd);
|
|
13997
|
-
const w = createGondolinWriteOps(vm, localCwd);
|
|
13998
|
-
return {
|
|
13999
|
-
readFile: r.readFile,
|
|
14000
|
-
access: r.access,
|
|
14001
|
-
writeFile: w.writeFile
|
|
14002
|
-
};
|
|
14003
|
-
}
|
|
14004
|
-
function createGondolinBashOps(vm, localCwd) {
|
|
14005
|
-
return { exec: async (command, cwd, { onData, signal, timeout, env }) => {
|
|
14006
|
-
const guestCwd = toGuestPath(localCwd, cwd);
|
|
14007
|
-
const ac = new AbortController();
|
|
14008
|
-
const onAbort = () => ac.abort();
|
|
14009
|
-
signal?.addEventListener("abort", onAbort, { once: true });
|
|
14010
|
-
let timedOut = false;
|
|
14011
|
-
const timer = timeout && timeout > 0 ? setTimeout(() => {
|
|
14012
|
-
timedOut = true;
|
|
14013
|
-
ac.abort();
|
|
14014
|
-
}, timeout * 1e3) : void 0;
|
|
14015
|
-
try {
|
|
14016
|
-
const proc = vm.exec([
|
|
14017
|
-
"/bin/sh",
|
|
14018
|
-
"-lc",
|
|
14019
|
-
command
|
|
14020
|
-
], {
|
|
14021
|
-
cwd: guestCwd,
|
|
14022
|
-
signal: ac.signal,
|
|
14023
|
-
stdout: "pipe",
|
|
14024
|
-
stderr: "pipe"
|
|
14025
|
-
});
|
|
14026
|
-
for await (const chunk of proc.output()) onData(typeof chunk.data === "string" ? Buffer.from(chunk.data, "utf8") : chunk.data);
|
|
14027
|
-
return { exitCode: (await proc).exitCode };
|
|
14028
|
-
} catch (err) {
|
|
14029
|
-
if (signal?.aborted) throw new Error("aborted");
|
|
14030
|
-
if (timedOut) throw new Error(`timeout:${timeout}`);
|
|
14031
|
-
throw err;
|
|
14032
|
-
} finally {
|
|
14033
|
-
if (timer) clearTimeout(timer);
|
|
14034
|
-
signal?.removeEventListener("abort", onAbort);
|
|
14035
|
-
}
|
|
14036
|
-
} };
|
|
14037
|
-
}
|
|
14038
|
-
//#endregion
|
|
14039
|
-
//#region ../../libs/pi-extension/src/vm-manager.ts
|
|
14040
|
-
var GUEST_WORKSPACE = "/workspace";
|
|
14220
|
+
* Memory-backed VFS mount used by the daemon to inject task-context
|
|
14221
|
+
* skills (#943 slice 1.5). Sibling of /workspace, NOT a sub-path —
|
|
14222
|
+
* Gondolin mounts can't nest. The agent's Gondolin-bound Read tool
|
|
14223
|
+
* accepts paths under this prefix (see toGuestPath in tool-operations.ts).
|
|
14224
|
+
*
|
|
14225
|
+
* Why MemoryProvider rather than a path under /workspace:
|
|
14226
|
+
* - Injected skills are ephemeral by intent: per-task-attempt input
|
|
14227
|
+
* scoped to the VM lifetime. MemoryProvider models that exactly —
|
|
14228
|
+
* in-memory, per-VM-instance, zero host artefacts, automatic
|
|
14229
|
+
* cleanup on VM close.
|
|
14230
|
+
* - Writing under /workspace fails in worktrees because we symlink
|
|
14231
|
+
* `.moltnet/` to the main repo (so credentials are reachable from
|
|
14232
|
+
* worktrees), and Gondolin's RealFSProvider correctly refuses to
|
|
14233
|
+
* create paths whose ancestors' realpath escapes the mount root.
|
|
14234
|
+
* That refusal is a deliberate sandbox-escape protection, not a
|
|
14235
|
+
* bug. See diary semantic entry cd27d9d3-efdc-4aec-ac0d-5fd8ce258d1f
|
|
14236
|
+
* and episodic 7affbfeb-18a2-4963-aeac-c177eb2afa2d for the full
|
|
14237
|
+
* investigation and the alternatives we rejected.
|
|
14238
|
+
*/
|
|
14239
|
+
var GUEST_TASK_SKILLS_MOUNT = "/moltnet-task-skills";
|
|
14041
14240
|
/**
|
|
14042
14241
|
* Resolve the main worktree root (where .moltnet/ lives — it's untracked,
|
|
14043
14242
|
* only exists in the main worktree, not in git worktrees).
|
|
@@ -14166,7 +14365,10 @@ async function resumeVm(config) {
|
|
|
14166
14365
|
env: vmEnv,
|
|
14167
14366
|
...resources?.memory && { memory: resources.memory },
|
|
14168
14367
|
...resources?.cpus && { cpus: resources.cpus },
|
|
14169
|
-
vfs: { mounts: {
|
|
14368
|
+
vfs: { mounts: {
|
|
14369
|
+
[GUEST_WORKSPACE$1]: workspaceProvider,
|
|
14370
|
+
[GUEST_TASK_SKILLS_MOUNT]: new MemoryProvider()
|
|
14371
|
+
} }
|
|
14170
14372
|
});
|
|
14171
14373
|
await vm.exec(`sh -c '
|
|
14172
14374
|
cp /etc/gondolin/mitm/ca.crt /usr/local/share/ca-certificates/gondolin-mitm.crt
|
|
@@ -14196,7 +14398,7 @@ nameserver 1.1.1.1" > /etc/resolv.conf'`);
|
|
|
14196
14398
|
vm,
|
|
14197
14399
|
credentials: creds,
|
|
14198
14400
|
mountPath: config.mountPath,
|
|
14199
|
-
guestWorkspace: GUEST_WORKSPACE,
|
|
14401
|
+
guestWorkspace: GUEST_WORKSPACE$1,
|
|
14200
14402
|
agentDir
|
|
14201
14403
|
};
|
|
14202
14404
|
}
|
|
@@ -14249,6 +14451,137 @@ function ensureRelativeWorktreePaths(gitconfig) {
|
|
|
14249
14451
|
return `${gitconfig}${gitconfig.endsWith("\n") ? "" : "\n"}[worktree]\n\tuseRelativePaths = true\n`;
|
|
14250
14452
|
}
|
|
14251
14453
|
//#endregion
|
|
14454
|
+
//#region ../../libs/pi-extension/src/tool-operations.ts
|
|
14455
|
+
/**
|
|
14456
|
+
* Gondolin tool operations: redirect pi's built-in tool operations
|
|
14457
|
+
* (read, write, edit, bash) to execute inside the VM.
|
|
14458
|
+
*
|
|
14459
|
+
* Follows the same pattern as upstream pi-gondolin.ts — pi's tool factories
|
|
14460
|
+
* accept an `operations` object that provides the underlying I/O.
|
|
14461
|
+
*/
|
|
14462
|
+
var GUEST_WORKSPACE = "/workspace";
|
|
14463
|
+
function shQuote(s) {
|
|
14464
|
+
return "'" + s.replace(/'/g, "'\\''") + "'";
|
|
14465
|
+
}
|
|
14466
|
+
/**
|
|
14467
|
+
* Map a host-side absolute path to a guest-side /workspace path.
|
|
14468
|
+
* Throws if the path escapes the workspace.
|
|
14469
|
+
*/
|
|
14470
|
+
function toGuestPath(localCwd, localPath) {
|
|
14471
|
+
if (localPath === GUEST_WORKSPACE || localPath.startsWith(`${GUEST_WORKSPACE}/`)) return localPath;
|
|
14472
|
+
if (localPath === "/moltnet-task-skills" || localPath.startsWith(`/moltnet-task-skills/`)) return localPath;
|
|
14473
|
+
const rel = path.relative(localCwd, localPath);
|
|
14474
|
+
if (rel === "") return GUEST_WORKSPACE;
|
|
14475
|
+
if (rel.startsWith("..") || path.isAbsolute(rel)) throw new Error(`path escapes workspace: ${localPath}`);
|
|
14476
|
+
const posixRel = rel.split(path.sep).join(path.posix.sep);
|
|
14477
|
+
return path.posix.join(GUEST_WORKSPACE, posixRel);
|
|
14478
|
+
}
|
|
14479
|
+
function createGondolinReadOps(vm, localCwd) {
|
|
14480
|
+
return {
|
|
14481
|
+
readFile: async (p) => {
|
|
14482
|
+
const r = await vm.exec(["/bin/cat", toGuestPath(localCwd, p)]);
|
|
14483
|
+
if (!r.ok) throw new Error(`cat failed (${r.exitCode}): ${r.stderr}`);
|
|
14484
|
+
return r.stdoutBuffer;
|
|
14485
|
+
},
|
|
14486
|
+
access: async (p) => {
|
|
14487
|
+
if (!(await vm.exec([
|
|
14488
|
+
"/bin/sh",
|
|
14489
|
+
"-lc",
|
|
14490
|
+
`test -r ${shQuote(toGuestPath(localCwd, p))}`
|
|
14491
|
+
])).ok) throw new Error(`not readable: ${p}`);
|
|
14492
|
+
},
|
|
14493
|
+
detectImageMimeType: async (p) => {
|
|
14494
|
+
try {
|
|
14495
|
+
const r = await vm.exec([
|
|
14496
|
+
"/bin/sh",
|
|
14497
|
+
"-lc",
|
|
14498
|
+
`file --mime-type -b ${shQuote(toGuestPath(localCwd, p))}`
|
|
14499
|
+
]);
|
|
14500
|
+
if (!r.ok) return null;
|
|
14501
|
+
const m = r.stdout.trim();
|
|
14502
|
+
return [
|
|
14503
|
+
"image/jpeg",
|
|
14504
|
+
"image/png",
|
|
14505
|
+
"image/gif",
|
|
14506
|
+
"image/webp"
|
|
14507
|
+
].includes(m) ? m : null;
|
|
14508
|
+
} catch {
|
|
14509
|
+
return null;
|
|
14510
|
+
}
|
|
14511
|
+
}
|
|
14512
|
+
};
|
|
14513
|
+
}
|
|
14514
|
+
function createGondolinWriteOps(vm, localCwd) {
|
|
14515
|
+
return {
|
|
14516
|
+
writeFile: async (p, content) => {
|
|
14517
|
+
const guestPath = toGuestPath(localCwd, p);
|
|
14518
|
+
const dir = path.posix.dirname(guestPath);
|
|
14519
|
+
const b64 = Buffer.from(content, "utf8").toString("base64");
|
|
14520
|
+
const r = await vm.exec([
|
|
14521
|
+
"/bin/sh",
|
|
14522
|
+
"-lc",
|
|
14523
|
+
[
|
|
14524
|
+
"set -eu",
|
|
14525
|
+
`mkdir -p ${shQuote(dir)}`,
|
|
14526
|
+
`echo ${shQuote(b64)} | base64 -d > ${shQuote(guestPath)}`
|
|
14527
|
+
].join("\n")
|
|
14528
|
+
]);
|
|
14529
|
+
if (!r.ok) throw new Error(`write failed (${r.exitCode}): ${r.stderr}`);
|
|
14530
|
+
},
|
|
14531
|
+
mkdir: async (dir) => {
|
|
14532
|
+
const r = await vm.exec([
|
|
14533
|
+
"/bin/mkdir",
|
|
14534
|
+
"-p",
|
|
14535
|
+
toGuestPath(localCwd, dir)
|
|
14536
|
+
]);
|
|
14537
|
+
if (!r.ok) throw new Error(`mkdir failed (${r.exitCode}): ${r.stderr}`);
|
|
14538
|
+
}
|
|
14539
|
+
};
|
|
14540
|
+
}
|
|
14541
|
+
function createGondolinEditOps(vm, localCwd) {
|
|
14542
|
+
const r = createGondolinReadOps(vm, localCwd);
|
|
14543
|
+
const w = createGondolinWriteOps(vm, localCwd);
|
|
14544
|
+
return {
|
|
14545
|
+
readFile: r.readFile,
|
|
14546
|
+
access: r.access,
|
|
14547
|
+
writeFile: w.writeFile
|
|
14548
|
+
};
|
|
14549
|
+
}
|
|
14550
|
+
function createGondolinBashOps(vm, localCwd) {
|
|
14551
|
+
return { exec: async (command, cwd, { onData, signal, timeout, env }) => {
|
|
14552
|
+
const guestCwd = toGuestPath(localCwd, cwd);
|
|
14553
|
+
const ac = new AbortController();
|
|
14554
|
+
const onAbort = () => ac.abort();
|
|
14555
|
+
signal?.addEventListener("abort", onAbort, { once: true });
|
|
14556
|
+
let timedOut = false;
|
|
14557
|
+
const timer = timeout && timeout > 0 ? setTimeout(() => {
|
|
14558
|
+
timedOut = true;
|
|
14559
|
+
ac.abort();
|
|
14560
|
+
}, timeout * 1e3) : void 0;
|
|
14561
|
+
try {
|
|
14562
|
+
const proc = vm.exec([
|
|
14563
|
+
"/bin/sh",
|
|
14564
|
+
"-lc",
|
|
14565
|
+
command
|
|
14566
|
+
], {
|
|
14567
|
+
cwd: guestCwd,
|
|
14568
|
+
signal: ac.signal,
|
|
14569
|
+
stdout: "pipe",
|
|
14570
|
+
stderr: "pipe"
|
|
14571
|
+
});
|
|
14572
|
+
for await (const chunk of proc.output()) onData(typeof chunk.data === "string" ? Buffer.from(chunk.data, "utf8") : chunk.data);
|
|
14573
|
+
return { exitCode: (await proc).exitCode };
|
|
14574
|
+
} catch (err) {
|
|
14575
|
+
if (signal?.aborted) throw new Error("aborted");
|
|
14576
|
+
if (timedOut) throw new Error(`timeout:${timeout}`);
|
|
14577
|
+
throw err;
|
|
14578
|
+
} finally {
|
|
14579
|
+
if (timer) clearTimeout(timer);
|
|
14580
|
+
signal?.removeEventListener("abort", onAbort);
|
|
14581
|
+
}
|
|
14582
|
+
} };
|
|
14583
|
+
}
|
|
14584
|
+
//#endregion
|
|
14252
14585
|
//#region ../../libs/pi-extension/src/otel/index.ts
|
|
14253
14586
|
var TRACER_NAME = "@themoltnet/pi-extension/otel";
|
|
14254
14587
|
function stripReservedAttrs(attrs) {
|
|
@@ -14386,6 +14719,147 @@ function extractUsage(message) {
|
|
|
14386
14719
|
};
|
|
14387
14720
|
}
|
|
14388
14721
|
//#endregion
|
|
14722
|
+
//#region ../../libs/pi-extension/src/runtime/agent-session-factory.ts
|
|
14723
|
+
var NO_SKILLS = () => ({
|
|
14724
|
+
skills: [],
|
|
14725
|
+
diagnostics: []
|
|
14726
|
+
});
|
|
14727
|
+
/**
|
|
14728
|
+
* Construct an in-memory `AgentSession`. The caller is responsible for
|
|
14729
|
+
* eventually invoking `session.prompt(...)` and for tearing down — the
|
|
14730
|
+
* helper does no lifecycle management beyond construction.
|
|
14731
|
+
*/
|
|
14732
|
+
async function buildAgentSession(args) {
|
|
14733
|
+
const piOtelExtension = createPiOtelExtension({
|
|
14734
|
+
agentName: args.agentName,
|
|
14735
|
+
spanAttributes: args.otelSpanAttrs
|
|
14736
|
+
});
|
|
14737
|
+
const resourceLoader = new DefaultResourceLoader({
|
|
14738
|
+
cwd: args.mountPath,
|
|
14739
|
+
agentDir: args.piAuthDir,
|
|
14740
|
+
extensionFactories: [piOtelExtension],
|
|
14741
|
+
appendSystemPrompt: args.appendSystemPrompt,
|
|
14742
|
+
skillsOverride: args.skillsOverride ?? NO_SKILLS
|
|
14743
|
+
});
|
|
14744
|
+
await resourceLoader.reload();
|
|
14745
|
+
return (await createAgentSession({
|
|
14746
|
+
agentDir: args.piAuthDir,
|
|
14747
|
+
cwd: args.mountPath,
|
|
14748
|
+
model: args.modelHandle,
|
|
14749
|
+
customTools: args.customTools,
|
|
14750
|
+
sessionManager: SessionManager.inMemory(),
|
|
14751
|
+
resourceLoader
|
|
14752
|
+
})).session;
|
|
14753
|
+
}
|
|
14754
|
+
//#endregion
|
|
14755
|
+
//#region ../../libs/pi-extension/src/runtime/inject-task-context.ts
|
|
14756
|
+
/**
|
|
14757
|
+
* Slice 1.5 of #943 — wire the agent-runtime resolver into the
|
|
14758
|
+
* pi-extension execution path.
|
|
14759
|
+
*
|
|
14760
|
+
* `resolveTaskContext` is a pure dispatcher; this module provides the
|
|
14761
|
+
* Gondolin-aware deliverer and the post-resolution shape the
|
|
14762
|
+
* `execute-pi-task` caller needs to splice into pi's setup:
|
|
14763
|
+
*
|
|
14764
|
+
* - `systemPromptPrefix` → fed into `appendSystemPrompt` alongside
|
|
14765
|
+
* the runtime instructor (it IS a system-prompt fragment).
|
|
14766
|
+
* - `userInlineSuffix` → appended to the `buildTaskUserPrompt`
|
|
14767
|
+
* output BEFORE `session.prompt(text)`.
|
|
14768
|
+
* - `skills` → spliced into the `skillsOverride` callback's
|
|
14769
|
+
* return value. pi includes them in `<available_skills>` in the
|
|
14770
|
+
* system prompt; the agent fetches the body on demand via the
|
|
14771
|
+
* Read tool.
|
|
14772
|
+
*
|
|
14773
|
+
* Skill files are written into the VM at
|
|
14774
|
+
* `/workspace/.moltnet/skills/<slug>/SKILL.md`. The agent's
|
|
14775
|
+
* Gondolin-bound Read tool is scoped to `/workspace`, so that path is
|
|
14776
|
+
* the only location the agent can actually read at runtime. pi only
|
|
14777
|
+
* reads `<available_skills>` metadata (name, description, location),
|
|
14778
|
+
* never the file body, so we construct synthetic `Skill` objects
|
|
14779
|
+
* pointing at the in-VM path without ever materialising the file on
|
|
14780
|
+
* the host.
|
|
14781
|
+
*/
|
|
14782
|
+
/**
|
|
14783
|
+
* Where in the VM we write skill bodies — the memory-backed mount
|
|
14784
|
+
* declared in `vm-manager.ts`. See the comment on
|
|
14785
|
+
* `GUEST_TASK_SKILLS_MOUNT` there for the full rationale (ephemeral
|
|
14786
|
+
* by intent + the worktree symlink interaction with Gondolin's
|
|
14787
|
+
* sandbox-escape protection). The agent's Gondolin Read tool accepts
|
|
14788
|
+
* paths under this mount via `toGuestPath` in `tool-operations.ts`.
|
|
14789
|
+
*/
|
|
14790
|
+
var SKILL_ROOT_IN_VM = GUEST_TASK_SKILLS_MOUNT;
|
|
14791
|
+
/** Bounds borrowed from pi's skill validation; conservative caps so a
|
|
14792
|
+
* malformed SKILL.md doesn't bloat the system prompt. */
|
|
14793
|
+
var MAX_SKILL_NAME = 64;
|
|
14794
|
+
var MAX_SKILL_DESCRIPTION = 1024;
|
|
14795
|
+
/**
|
|
14796
|
+
* Resolve a task's `input.context[]` and inject the side effects pi
|
|
14797
|
+
* needs. Safe to call with an empty array — returns an inert result.
|
|
14798
|
+
*/
|
|
14799
|
+
async function injectTaskContext(args) {
|
|
14800
|
+
const skills = [];
|
|
14801
|
+
const resolved = await resolveTaskContext({
|
|
14802
|
+
context: args.context,
|
|
14803
|
+
deliver: { skill: async ({ slug, content }) => {
|
|
14804
|
+
const dir = `${SKILL_ROOT_IN_VM}/${slug}`;
|
|
14805
|
+
const filePath = `${dir}/SKILL.md`;
|
|
14806
|
+
await args.fs.mkdir(dir, { recursive: true });
|
|
14807
|
+
await args.fs.writeFile(filePath, content, { mode: 420 });
|
|
14808
|
+
skills.push(buildSyntheticSkill({
|
|
14809
|
+
slug,
|
|
14810
|
+
content,
|
|
14811
|
+
filePath,
|
|
14812
|
+
dir
|
|
14813
|
+
}));
|
|
14814
|
+
} }
|
|
14815
|
+
});
|
|
14816
|
+
return {
|
|
14817
|
+
injected: resolved.injected,
|
|
14818
|
+
skills,
|
|
14819
|
+
systemPromptPrefix: resolved.systemPromptPrefix,
|
|
14820
|
+
userInlineSuffix: resolved.userInlineSuffix
|
|
14821
|
+
};
|
|
14822
|
+
}
|
|
14823
|
+
/**
|
|
14824
|
+
* Build a `Skill` object pi will faithfully render in
|
|
14825
|
+
* `<available_skills>`. We extract `name` and `description` from the
|
|
14826
|
+
* skill content's YAML frontmatter using pi's own `parseFrontmatter`
|
|
14827
|
+
* helper (proper YAML, not a regex hack) and fall back to the slug +
|
|
14828
|
+
* a generic description so a SKILL.md without frontmatter still
|
|
14829
|
+
* renders something meaningful.
|
|
14830
|
+
*
|
|
14831
|
+
* Frontmatter parsing is best-effort: a malformed YAML block is
|
|
14832
|
+
* optional metadata, not a reason to fail the task. We swallow parser
|
|
14833
|
+
* errors and fall back to the slug-derived metadata; the skill body
|
|
14834
|
+
* is unaffected.
|
|
14835
|
+
*
|
|
14836
|
+
* pi's `formatSkillsForPrompt` only reads `name`, `description`, and
|
|
14837
|
+
* `filePath` — `sourceInfo`/`baseDir` exist on the type but never
|
|
14838
|
+
* surface in the prompt, so a synthetic `SourceInfo` is enough.
|
|
14839
|
+
*/
|
|
14840
|
+
function buildSyntheticSkill(args) {
|
|
14841
|
+
let fm = {};
|
|
14842
|
+
try {
|
|
14843
|
+
fm = parseFrontmatter(args.content).frontmatter;
|
|
14844
|
+
} catch {}
|
|
14845
|
+
return {
|
|
14846
|
+
name: clip(typeof fm.name === "string" && fm.name.trim().length > 0 ? fm.name.trim() : args.slug, MAX_SKILL_NAME),
|
|
14847
|
+
description: clip(typeof fm.description === "string" && fm.description.trim().length > 0 ? fm.description.trim() : `Task-injected context skill (${args.slug})`, MAX_SKILL_DESCRIPTION),
|
|
14848
|
+
filePath: args.filePath,
|
|
14849
|
+
baseDir: args.dir,
|
|
14850
|
+
sourceInfo: createSyntheticSourceInfo(args.filePath, {
|
|
14851
|
+
source: "moltnet:task-context",
|
|
14852
|
+
scope: "temporary",
|
|
14853
|
+
origin: "top-level",
|
|
14854
|
+
baseDir: args.dir
|
|
14855
|
+
}),
|
|
14856
|
+
disableModelInvocation: fm["disable-model-invocation"] === true
|
|
14857
|
+
};
|
|
14858
|
+
}
|
|
14859
|
+
function clip(s, max) {
|
|
14860
|
+
return s.length > max ? s.slice(0, max) : s;
|
|
14861
|
+
}
|
|
14862
|
+
//#endregion
|
|
14389
14863
|
//#region ../../libs/pi-extension/src/runtime/runtime-instructor.ts
|
|
14390
14864
|
/**
|
|
14391
14865
|
* Build the daemon-controlled invariant prose injected into the system prompt
|
|
@@ -14471,6 +14945,190 @@ function buildRuntimeInstructor(ctx) {
|
|
|
14471
14945
|
].join("\n");
|
|
14472
14946
|
}
|
|
14473
14947
|
//#endregion
|
|
14948
|
+
//#region ../../libs/pi-extension/src/runtime/subagent-tool.ts
|
|
14949
|
+
var SUBAGENT_SUBMIT_TOOL_NAME = "submit_subagent_output";
|
|
14950
|
+
/**
|
|
14951
|
+
* Parameters shape the parent LLM sees when calling the subagent tool.
|
|
14952
|
+
*
|
|
14953
|
+
* - `task` — natural-language instructions for the subagent.
|
|
14954
|
+
* The parent authors this per call. Must be
|
|
14955
|
+
* non-empty.
|
|
14956
|
+
* - `output_schema` — name of a registered SubagentOutputContract.
|
|
14957
|
+
* Resolved at call time; unknown names error.
|
|
14958
|
+
*/
|
|
14959
|
+
var SubagentToolParameters = Type$2.Object({
|
|
14960
|
+
task: Type$2.String({
|
|
14961
|
+
minLength: 1,
|
|
14962
|
+
description: "Natural-language instructions for the subagent. The subagent starts with a fresh conversation and a narrowed system prompt; this is the only context it has from you."
|
|
14963
|
+
}),
|
|
14964
|
+
output_schema: Type$2.String({
|
|
14965
|
+
minLength: 1,
|
|
14966
|
+
description: "Name of a registered subagent output contract. The subagent must submit a structured payload via `submit_subagent_output` matching this contract."
|
|
14967
|
+
})
|
|
14968
|
+
}, { additionalProperties: false });
|
|
14969
|
+
var DEFAULT_SUBAGENT_TIMEOUT_MS = 300 * 1e3;
|
|
14970
|
+
/**
|
|
14971
|
+
* Build the subagent custom tool for a parent session. The handle
|
|
14972
|
+
* exposes the call counter so executors can emit summary telemetry
|
|
14973
|
+
* when the parent terminates.
|
|
14974
|
+
*/
|
|
14975
|
+
function createSubagentTool(args) {
|
|
14976
|
+
const buildSession = args.buildAgentSession ?? buildAgentSession;
|
|
14977
|
+
let callCount = 0;
|
|
14978
|
+
return {
|
|
14979
|
+
tool: defineTool({
|
|
14980
|
+
name: "subagent",
|
|
14981
|
+
label: "Delegate to subagent",
|
|
14982
|
+
description: subagentToolDescription(),
|
|
14983
|
+
parameters: SubagentToolParameters,
|
|
14984
|
+
async execute(_id, params) {
|
|
14985
|
+
if (!Check(SubagentToolParameters, params)) return toolError(`subagent: invalid parameters: ${JSON.stringify([...Errors(SubagentToolParameters, params)].slice(0, 3))}`);
|
|
14986
|
+
const { task, output_schema } = params;
|
|
14987
|
+
const contract = getSubagentOutputContract(output_schema);
|
|
14988
|
+
if (!contract) return toolError(`subagent: unknown output_schema "${output_schema}". Registered contracts: [${listSubagentOutputContracts().map((c) => c.name).join(", ")}]`);
|
|
14989
|
+
callCount += 1;
|
|
14990
|
+
const callIndex = callCount;
|
|
14991
|
+
let captured = null;
|
|
14992
|
+
const submitTool = defineTool({
|
|
14993
|
+
name: SUBAGENT_SUBMIT_TOOL_NAME,
|
|
14994
|
+
label: `Submit ${output_schema}`,
|
|
14995
|
+
description: `Submit your structured output for this subagent task. Call exactly once when done. Args MUST match the ${output_schema} contract; mismatches return a tool error you can recover from in the same session.`,
|
|
14996
|
+
parameters: contract.parametersSchema,
|
|
14997
|
+
async execute(_innerId, innerParams) {
|
|
14998
|
+
if (!Check(contract.parametersSchema, innerParams)) return toolError(`submit_subagent_output: schema validation failed: ${[...Errors(contract.parametersSchema, innerParams)].slice(0, 3).map((e) => `${e.path}: ${e.message}`).join("; ")}. Re-call with a corrected payload.`);
|
|
14999
|
+
captured = innerParams;
|
|
15000
|
+
return {
|
|
15001
|
+
content: [{
|
|
15002
|
+
type: "text",
|
|
15003
|
+
text: "Output captured. Subagent session will terminate; no further action needed."
|
|
15004
|
+
}],
|
|
15005
|
+
details: { captured: true },
|
|
15006
|
+
terminate: true
|
|
15007
|
+
};
|
|
15008
|
+
}
|
|
15009
|
+
});
|
|
15010
|
+
const subagentInstructor = buildSubagentInstructor({
|
|
15011
|
+
contractName: output_schema,
|
|
15012
|
+
contractDescription: contract.description,
|
|
15013
|
+
parentTaskId: args.parentTaskId,
|
|
15014
|
+
callIndex
|
|
15015
|
+
});
|
|
15016
|
+
const session = await buildSession({
|
|
15017
|
+
mountPath: args.mountPath,
|
|
15018
|
+
piAuthDir: args.piAuthDir,
|
|
15019
|
+
modelHandle: args.modelHandle,
|
|
15020
|
+
agentName: args.agentName,
|
|
15021
|
+
customTools: [...args.inheritedCustomTools, submitTool],
|
|
15022
|
+
appendSystemPrompt: [args.parentRuntimeInstructor, subagentInstructor],
|
|
15023
|
+
skillsOverride: () => ({
|
|
15024
|
+
skills: [],
|
|
15025
|
+
diagnostics: []
|
|
15026
|
+
}),
|
|
15027
|
+
otelSpanAttrs: {
|
|
15028
|
+
"moltnet.task.id": args.parentTaskId,
|
|
15029
|
+
"moltnet.task.type": args.parentTaskType,
|
|
15030
|
+
"moltnet.task.attempt": args.parentAttemptN,
|
|
15031
|
+
"moltnet.subagent.contract": output_schema,
|
|
15032
|
+
"moltnet.subagent.index": callIndex
|
|
15033
|
+
}
|
|
15034
|
+
});
|
|
15035
|
+
let abortReason = null;
|
|
15036
|
+
let abortInvoked = false;
|
|
15037
|
+
const fireAbort = (reason) => {
|
|
15038
|
+
if (abortInvoked) return;
|
|
15039
|
+
abortInvoked = true;
|
|
15040
|
+
abortReason = reason;
|
|
15041
|
+
session.abort().catch((err) => {
|
|
15042
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
15043
|
+
process.stderr.write(`[subagent] inner session.abort() failed: ${message}\n`);
|
|
15044
|
+
});
|
|
15045
|
+
};
|
|
15046
|
+
const cancelListener = args.parentCancelSignal ? (() => {
|
|
15047
|
+
const signal = args.parentCancelSignal;
|
|
15048
|
+
const listener = () => fireAbort("parent_cancelled");
|
|
15049
|
+
if (signal.aborted) listener();
|
|
15050
|
+
else signal.addEventListener("abort", listener, { once: true });
|
|
15051
|
+
return () => signal.removeEventListener("abort", listener);
|
|
15052
|
+
})() : null;
|
|
15053
|
+
const timeoutMs = args.timeoutMs === void 0 || args.timeoutMs < 0 ? DEFAULT_SUBAGENT_TIMEOUT_MS : args.timeoutMs;
|
|
15054
|
+
const timeoutHandle = timeoutMs > 0 ? setTimeout(() => fireAbort("subagent_timed_out"), timeoutMs) : null;
|
|
15055
|
+
try {
|
|
15056
|
+
await session.prompt(task);
|
|
15057
|
+
} catch (err) {
|
|
15058
|
+
return toolError(`subagent: inner session.prompt() threw: ${err instanceof Error ? err.message : String(err)}`);
|
|
15059
|
+
} finally {
|
|
15060
|
+
if (timeoutHandle) clearTimeout(timeoutHandle);
|
|
15061
|
+
if (cancelListener) cancelListener();
|
|
15062
|
+
}
|
|
15063
|
+
if (abortReason !== null) return toolError(`subagent: ${abortReason === "subagent_timed_out" ? `subagent timed out after ${timeoutMs}ms` : "parent task was cancelled"}. The parent should fail this task or retry with a clearer scope.`);
|
|
15064
|
+
if (captured === null) return toolError(`subagent: inner session ended without calling ${SUBAGENT_SUBMIT_TOOL_NAME}. The parent should retry with clearer instructions or fail the task.`);
|
|
15065
|
+
return {
|
|
15066
|
+
content: [{
|
|
15067
|
+
type: "text",
|
|
15068
|
+
text: JSON.stringify(captured)
|
|
15069
|
+
}],
|
|
15070
|
+
details: {
|
|
15071
|
+
captured: true,
|
|
15072
|
+
contract: output_schema,
|
|
15073
|
+
callIndex
|
|
15074
|
+
}
|
|
15075
|
+
};
|
|
15076
|
+
}
|
|
15077
|
+
}),
|
|
15078
|
+
getCallCount: () => callCount
|
|
15079
|
+
};
|
|
15080
|
+
}
|
|
15081
|
+
function subagentToolDescription() {
|
|
15082
|
+
return [
|
|
15083
|
+
"Delegate a sub-task to a fresh subagent session with isolated context.",
|
|
15084
|
+
"",
|
|
15085
|
+
"The subagent starts with no conversation history and only the `task` ",
|
|
15086
|
+
"string you provide as its instructions. It runs in the same VM with ",
|
|
15087
|
+
"the same tools you have (Gondolin-routed Read/Write/Edit/Bash, ",
|
|
15088
|
+
"moltnet_* tools), and is expected to call ",
|
|
15089
|
+
`\`${SUBAGENT_SUBMIT_TOOL_NAME}\` with a payload matching the named `,
|
|
15090
|
+
"contract before its session ends.",
|
|
15091
|
+
"",
|
|
15092
|
+
"On success, the tool result is the JSON-stringified subagent payload.",
|
|
15093
|
+
"On failure (unknown contract, validation error, subagent did not ",
|
|
15094
|
+
"submit) the tool returns isError:true with a recoverable message."
|
|
15095
|
+
].join("\n");
|
|
15096
|
+
}
|
|
15097
|
+
function buildSubagentInstructor(args) {
|
|
15098
|
+
return [
|
|
15099
|
+
"# You are a subagent",
|
|
15100
|
+
"",
|
|
15101
|
+
`Parent task: \`${args.parentTaskId}\` (subagent call #${args.callIndex}).`,
|
|
15102
|
+
"",
|
|
15103
|
+
`Your assigned output contract is \`${args.contractName}\`:`,
|
|
15104
|
+
`${args.contractDescription}`,
|
|
15105
|
+
"",
|
|
15106
|
+
"Rules for this session:",
|
|
15107
|
+
"",
|
|
15108
|
+
`- You MUST call \`${SUBAGENT_SUBMIT_TOOL_NAME}\` exactly once with a `,
|
|
15109
|
+
" payload matching the contract above. Your session terminates on ",
|
|
15110
|
+
" the valid call.",
|
|
15111
|
+
"- The parent's message above is your task. Do not invent additional ",
|
|
15112
|
+
" steps the parent did not request.",
|
|
15113
|
+
"- All MoltNet runtime invariants from the parent runtime instructor ",
|
|
15114
|
+
" apply (diary discipline, gh-auth pattern, etc.) IF you take any ",
|
|
15115
|
+
" action that would trigger them. Most subagents do not commit code ",
|
|
15116
|
+
" or open PRs — only do so if your task message explicitly requires it.",
|
|
15117
|
+
"- You do NOT have access to the `subagent` tool. Do not attempt nested ",
|
|
15118
|
+
" delegation; do the work yourself."
|
|
15119
|
+
].join("\n");
|
|
15120
|
+
}
|
|
15121
|
+
function toolError(text) {
|
|
15122
|
+
return {
|
|
15123
|
+
content: [{
|
|
15124
|
+
type: "text",
|
|
15125
|
+
text
|
|
15126
|
+
}],
|
|
15127
|
+
details: { captured: false },
|
|
15128
|
+
isError: true
|
|
15129
|
+
};
|
|
15130
|
+
}
|
|
15131
|
+
//#endregion
|
|
14474
15132
|
//#region ../../libs/pi-extension/src/runtime/task-output.ts
|
|
14475
15133
|
var METER_NAME = "@themoltnet/pi-extension/task-output";
|
|
14476
15134
|
var parseResultCounter = null;
|
|
@@ -14709,6 +15367,7 @@ function resolveSubmitTools(taskType, opts = {}) {
|
|
|
14709
15367
|
* Anthropic-SDK one) plug in via the `executeTask` function injected into
|
|
14710
15368
|
* `AgentRuntime`.
|
|
14711
15369
|
*/
|
|
15370
|
+
var noopTurnEventHandler = () => {};
|
|
14712
15371
|
/**
|
|
14713
15372
|
* Factory that builds a pi-specific `executeTask` function suitable for
|
|
14714
15373
|
* injection into `AgentRuntime`. The returned function caches the resolved
|
|
@@ -14781,6 +15440,7 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
14781
15440
|
const taskTeamId = task.teamId ?? "";
|
|
14782
15441
|
let reporterOpen = false;
|
|
14783
15442
|
let session = null;
|
|
15443
|
+
let subagentHandle = null;
|
|
14784
15444
|
const finalUsage = emptyUsage(opts.provider, opts.model);
|
|
14785
15445
|
let cancelListener = null;
|
|
14786
15446
|
const makeFailedOutput = (code, message, usage = finalUsage) => ({
|
|
@@ -14805,10 +15465,25 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
14805
15465
|
attemptN
|
|
14806
15466
|
});
|
|
14807
15467
|
reporterOpen = true;
|
|
14808
|
-
|
|
14809
|
-
|
|
14810
|
-
|
|
14811
|
-
})
|
|
15468
|
+
let onTurnEvent;
|
|
15469
|
+
if (opts.makeOnTurnEvent) try {
|
|
15470
|
+
onTurnEvent = opts.makeOnTurnEvent(claimedTask);
|
|
15471
|
+
} catch (err) {
|
|
15472
|
+
process.stderr.write(`[emit] makeOnTurnEvent threw: ${err instanceof Error ? err.message : String(err)}\n`);
|
|
15473
|
+
onTurnEvent = noopTurnEventHandler;
|
|
15474
|
+
}
|
|
15475
|
+
else onTurnEvent = opts.onTurnEvent ?? noopTurnEventHandler;
|
|
15476
|
+
const emit = (kind, payload) => {
|
|
15477
|
+
try {
|
|
15478
|
+
onTurnEvent(kind, summarizePayloadForLog(kind, payload));
|
|
15479
|
+
} catch (err) {
|
|
15480
|
+
process.stderr.write(`[emit] onTurnEvent threw for kind="${kind}": ${err instanceof Error ? err.message : String(err)}\n`);
|
|
15481
|
+
}
|
|
15482
|
+
return reporter.record({
|
|
15483
|
+
kind,
|
|
15484
|
+
payload
|
|
15485
|
+
});
|
|
15486
|
+
};
|
|
14812
15487
|
await emit("info", {
|
|
14813
15488
|
event: "execute_start",
|
|
14814
15489
|
taskType: task.taskType,
|
|
@@ -14818,7 +15493,7 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
14818
15493
|
});
|
|
14819
15494
|
let taskPrompt;
|
|
14820
15495
|
try {
|
|
14821
|
-
taskPrompt =
|
|
15496
|
+
taskPrompt = buildTaskUserPrompt(task, {
|
|
14822
15497
|
diaryId,
|
|
14823
15498
|
taskId: task.id,
|
|
14824
15499
|
extras: opts.promptExtras
|
|
@@ -14831,6 +15506,30 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
14831
15506
|
});
|
|
14832
15507
|
return makeFailedOutput("prompt_build_failed", message);
|
|
14833
15508
|
}
|
|
15509
|
+
const rawContext = task.input.context;
|
|
15510
|
+
let injectedContext;
|
|
15511
|
+
try {
|
|
15512
|
+
const contextArray = rawContext === void 0 ? [] : rawContext;
|
|
15513
|
+
if (!Check(TaskContext, contextArray)) throw new Error(`task.input.context failed TaskContext validation: ${JSON.stringify([...Errors(TaskContext, contextArray)].slice(0, 3))}`);
|
|
15514
|
+
injectedContext = await injectTaskContext({
|
|
15515
|
+
context: contextArray,
|
|
15516
|
+
fs: managed.vm.fs
|
|
15517
|
+
});
|
|
15518
|
+
} catch (err) {
|
|
15519
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
15520
|
+
await emit("error", {
|
|
15521
|
+
message,
|
|
15522
|
+
phase: "context_resolution"
|
|
15523
|
+
});
|
|
15524
|
+
return makeFailedOutput("context_resolution_failed", message);
|
|
15525
|
+
}
|
|
15526
|
+
if (injectedContext.injected.length > 0) await emit("info", {
|
|
15527
|
+
event: "context_injected",
|
|
15528
|
+
count: injectedContext.injected.length,
|
|
15529
|
+
bindings: injectedContext.injected.map((r) => r.binding),
|
|
15530
|
+
slugs: injectedContext.injected.map((r) => r.slug)
|
|
15531
|
+
});
|
|
15532
|
+
if (injectedContext.userInlineSuffix) taskPrompt = `${taskPrompt}\n\n---\n\n${injectedContext.userInlineSuffix}`;
|
|
14834
15533
|
const gondolinCustomTools = [
|
|
14835
15534
|
createReadToolDefinition(mountPath, { operations: createGondolinReadOps(managed.vm, mountPath) }),
|
|
14836
15535
|
createWriteToolDefinition(mountPath, { operations: createGondolinWriteOps(managed.vm, mountPath) }),
|
|
@@ -14859,14 +15558,6 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
14859
15558
|
});
|
|
14860
15559
|
const piAuthDir = process.env.PI_CODING_AGENT_DIR ?? join(homedir(), ".pi", "agent");
|
|
14861
15560
|
const modelHandle = getModel(opts.provider, opts.model);
|
|
14862
|
-
const piOtelExtension = createPiOtelExtension({
|
|
14863
|
-
agentName: opts.agentName,
|
|
14864
|
-
spanAttributes: {
|
|
14865
|
-
"moltnet.task.id": task.id,
|
|
14866
|
-
"moltnet.task.attempt": attemptN,
|
|
14867
|
-
"moltnet.task.type": task.taskType
|
|
14868
|
-
}
|
|
14869
|
-
});
|
|
14870
15561
|
const runtimeInstructor = buildRuntimeInstructor({
|
|
14871
15562
|
taskId: task.id,
|
|
14872
15563
|
taskType: task.taskType,
|
|
@@ -14875,29 +15566,47 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
14875
15566
|
agentName: opts.agentName,
|
|
14876
15567
|
correlationId: task.correlationId ?? null
|
|
14877
15568
|
});
|
|
14878
|
-
const
|
|
14879
|
-
|
|
14880
|
-
|
|
14881
|
-
|
|
14882
|
-
|
|
14883
|
-
|
|
14884
|
-
|
|
14885
|
-
|
|
14886
|
-
|
|
14887
|
-
|
|
14888
|
-
|
|
14889
|
-
|
|
14890
|
-
|
|
14891
|
-
|
|
14892
|
-
|
|
15569
|
+
const appendSystemPrompt = [runtimeInstructor];
|
|
15570
|
+
if (injectedContext.systemPromptPrefix) appendSystemPrompt.push(injectedContext.systemPromptPrefix);
|
|
15571
|
+
const injectedSkills = injectedContext.skills;
|
|
15572
|
+
const parentSubagentTools = [];
|
|
15573
|
+
if (taskTypeUsesSubagents(task.taskType)) {
|
|
15574
|
+
subagentHandle = createSubagentTool({
|
|
15575
|
+
mountPath,
|
|
15576
|
+
piAuthDir,
|
|
15577
|
+
modelHandle,
|
|
15578
|
+
agentName: opts.agentName,
|
|
15579
|
+
inheritedCustomTools: [...gondolinCustomTools, ...moltnetTools],
|
|
15580
|
+
parentRuntimeInstructor: runtimeInstructor,
|
|
15581
|
+
parentTaskId: task.id,
|
|
15582
|
+
parentTaskType: task.taskType,
|
|
15583
|
+
parentAttemptN: attemptN,
|
|
15584
|
+
parentCancelSignal: reporter.cancelSignal
|
|
15585
|
+
});
|
|
15586
|
+
parentSubagentTools.push(subagentHandle.tool);
|
|
15587
|
+
}
|
|
15588
|
+
session = await buildAgentSession({
|
|
15589
|
+
mountPath,
|
|
15590
|
+
piAuthDir,
|
|
15591
|
+
modelHandle,
|
|
15592
|
+
agentName: opts.agentName,
|
|
14893
15593
|
customTools: [
|
|
14894
15594
|
...gondolinCustomTools,
|
|
14895
15595
|
...moltnetTools,
|
|
14896
|
-
...submitTools
|
|
15596
|
+
...submitTools,
|
|
15597
|
+
...parentSubagentTools
|
|
14897
15598
|
],
|
|
14898
|
-
|
|
14899
|
-
|
|
14900
|
-
|
|
15599
|
+
appendSystemPrompt,
|
|
15600
|
+
skillsOverride: () => ({
|
|
15601
|
+
skills: injectedSkills,
|
|
15602
|
+
diagnostics: []
|
|
15603
|
+
}),
|
|
15604
|
+
otelSpanAttrs: {
|
|
15605
|
+
"moltnet.task.id": task.id,
|
|
15606
|
+
"moltnet.task.attempt": attemptN,
|
|
15607
|
+
"moltnet.task.type": task.taskType
|
|
15608
|
+
}
|
|
15609
|
+
});
|
|
14901
15610
|
} catch (err) {
|
|
14902
15611
|
const message = err instanceof Error ? err.message : String(err);
|
|
14903
15612
|
await emit("error", {
|
|
@@ -14968,6 +15677,10 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
14968
15677
|
phase: "session_prompt"
|
|
14969
15678
|
});
|
|
14970
15679
|
}
|
|
15680
|
+
if (subagentHandle && subagentHandle.getCallCount() > 0) await emit("info", {
|
|
15681
|
+
event: "subagent_summary",
|
|
15682
|
+
callCount: subagentHandle.getCallCount()
|
|
15683
|
+
});
|
|
14971
15684
|
await Promise.all(recordingPromise);
|
|
14972
15685
|
const cancelled = reporter.cancelSignal.aborted;
|
|
14973
15686
|
let parsedOutput = null;
|
|
@@ -15106,6 +15819,27 @@ function wireSessionAbort(cancelSignal, session) {
|
|
|
15106
15819
|
* `task_messages.payload` row. Bodies above 4 KiB are replaced with a
|
|
15107
15820
|
* `{ truncated, original_size }` marker so the JSONL/DB size stays bounded.
|
|
15108
15821
|
*/
|
|
15822
|
+
function summarizePayloadForLog(kind, payload) {
|
|
15823
|
+
switch (kind) {
|
|
15824
|
+
case "text_delta": {
|
|
15825
|
+
const delta = payload.delta;
|
|
15826
|
+
return { chars: typeof delta === "string" ? delta.length : 0 };
|
|
15827
|
+
}
|
|
15828
|
+
case "tool_call_start": return { tool: payload.tool_name };
|
|
15829
|
+
case "tool_call_end": return {
|
|
15830
|
+
tool: payload.tool_name,
|
|
15831
|
+
is_error: payload.is_error === true,
|
|
15832
|
+
...payload.is_error === true && payload.result !== void 0 ? { result: payload.result } : {}
|
|
15833
|
+
};
|
|
15834
|
+
case "turn_end": return { stop_reason: payload.stop_reason };
|
|
15835
|
+
case "error": return {
|
|
15836
|
+
phase: payload.phase,
|
|
15837
|
+
message: typeof payload.message === "string" ? payload.message.slice(0, TRUNCATE_LIMIT) : payload.message
|
|
15838
|
+
};
|
|
15839
|
+
case "info": return Object.fromEntries(Object.entries(payload).map(([k, v]) => [k, typeof v === "string" ? v.slice(0, TRUNCATE_LIMIT) : v]));
|
|
15840
|
+
default: return payload;
|
|
15841
|
+
}
|
|
15842
|
+
}
|
|
15109
15843
|
var TRUNCATE_LIMIT = 4 * 1024;
|
|
15110
15844
|
function truncateForWire(value) {
|
|
15111
15845
|
if (value === null || value === void 0) return value;
|
|
@@ -15405,6 +16139,27 @@ function findUp(startDir, filename) {
|
|
|
15405
16139
|
}
|
|
15406
16140
|
}
|
|
15407
16141
|
//#endregion
|
|
16142
|
+
//#region src/lib/turn-event-logger.ts
|
|
16143
|
+
function makeTurnEventHandler(base, context = {}) {
|
|
16144
|
+
const log = base.child({
|
|
16145
|
+
name: "agent-daemon.turn",
|
|
16146
|
+
...context
|
|
16147
|
+
});
|
|
16148
|
+
return (event, summary) => {
|
|
16149
|
+
if (event === "text_delta") return;
|
|
16150
|
+
log[event === "error" ? "warn" : event === "turn_end" ? "info" : "debug"]({
|
|
16151
|
+
event,
|
|
16152
|
+
...summary
|
|
16153
|
+
}, `turn.${event}`);
|
|
16154
|
+
};
|
|
16155
|
+
}
|
|
16156
|
+
function makeTurnEventHandlerFactory(base) {
|
|
16157
|
+
return (claimedTask) => makeTurnEventHandler(base, {
|
|
16158
|
+
taskId: claimedTask.task.id,
|
|
16159
|
+
attemptN: claimedTask.attemptN
|
|
16160
|
+
});
|
|
16161
|
+
}
|
|
16162
|
+
//#endregion
|
|
15408
16163
|
//#region src/cli/poll-shared.ts
|
|
15409
16164
|
async function runPolling(opts) {
|
|
15410
16165
|
if (isHelpFlag(opts.argv)) {
|
|
@@ -15507,7 +16262,8 @@ async function runPolling(opts) {
|
|
|
15507
16262
|
mountPath: sandbox.rootDir,
|
|
15508
16263
|
provider: common.provider,
|
|
15509
16264
|
model: common.model,
|
|
15510
|
-
sandboxConfig: sandbox.config
|
|
16265
|
+
sandboxConfig: sandbox.config,
|
|
16266
|
+
makeOnTurnEvent: makeTurnEventHandlerFactory(rootLogger)
|
|
15511
16267
|
});
|
|
15512
16268
|
runtime = new AgentRuntime({
|
|
15513
16269
|
logger: rootLogger,
|
|
@@ -15515,6 +16271,8 @@ async function runPolling(opts) {
|
|
|
15515
16271
|
agent: ctx.agent,
|
|
15516
16272
|
teamId,
|
|
15517
16273
|
taskTypes: taskTypes.length > 0 ? taskTypes : void 0,
|
|
16274
|
+
provider: common.provider.toLowerCase(),
|
|
16275
|
+
model: common.model.toLowerCase(),
|
|
15518
16276
|
diaryIds: diaryIds.length > 0 ? diaryIds : void 0,
|
|
15519
16277
|
leaseTtlSec: common.leaseTtlSec,
|
|
15520
16278
|
listLimit,
|
|
@@ -15698,7 +16456,8 @@ async function runOnce(argv) {
|
|
|
15698
16456
|
mountPath: sandbox.rootDir,
|
|
15699
16457
|
provider: opts.provider,
|
|
15700
16458
|
model: opts.model,
|
|
15701
|
-
sandboxConfig: sandbox.config
|
|
16459
|
+
sandboxConfig: sandbox.config,
|
|
16460
|
+
onTurnEvent: makeTurnEventHandler(rootLogger, { taskId })
|
|
15702
16461
|
});
|
|
15703
16462
|
const writeCorrelationAnchors = makePrBodyAnchorWriter({
|
|
15704
16463
|
gh: createGhCliClient(),
|