@themoltnet/agent-daemon 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/dist/main.js +666 -165
  2. package/package.json +6 -6
package/dist/main.js CHANGED
@@ -9,9 +9,9 @@ import { createHash as createHash$1 } from "node:crypto";
9
9
  import path, { dirname, isAbsolute, join, resolve } from "node:path";
10
10
  import { homedir } from "node:os";
11
11
  import { execFile, execFileSync } from "node:child_process";
12
- import { DefaultResourceLoader, SessionManager, createAgentSession, createBashToolDefinition, createEditToolDefinition, createReadToolDefinition, createWriteToolDefinition, defineTool } from "@earendil-works/pi-coding-agent";
12
+ import { DefaultResourceLoader, SessionManager, createAgentSession, createBashToolDefinition, createEditToolDefinition, createReadToolDefinition, createSyntheticSourceInfo, createWriteToolDefinition, defineTool, parseFrontmatter } from "@earendil-works/pi-coding-agent";
13
13
  import { Type, getModel } from "@earendil-works/pi-ai";
14
- import { RealFSProvider, ShadowProvider, VM, VmCheckpoint, createHttpHooks, createShadowPathPredicate, ensureImageSelector, loadGuestAssets } from "@earendil-works/gondolin";
14
+ import { MemoryProvider, RealFSProvider, ShadowProvider, VM, VmCheckpoint, createHttpHooks, createShadowPathPredicate, ensureImageSelector, loadGuestAssets } from "@earendil-works/gondolin";
15
15
  import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto";
16
16
  import { resourceFromAttributes } from "@opentelemetry/resources";
17
17
  import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base";
@@ -2886,6 +2886,55 @@ var UUID_RE = /^[0-9a-f]{8}-[0-9a-f]{4}-[1-5][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a
2886
2886
  if (!Has$1("uuid")) Set$1("uuid", (v) => UUID_RE.test(v));
2887
2887
  if (!Has$1("date-time")) Set$1("date-time", (v) => !Number.isNaN(Date.parse(v)));
2888
2888
  //#endregion
2889
+ //#region ../../libs/tasks/src/context.ts
2890
+ /**
2891
+ * How an executor delivers a context entry to its underlying LLM.
2892
+ * V1 bindings only; Tier-2 (reference_file, mcp_resource, imported_file,
2893
+ * tool_response_seed, additional_context_hook) ship in a later slice.
2894
+ */
2895
+ var ContextBinding = Type$2.Union([
2896
+ Type$2.Literal("skill"),
2897
+ Type$2.Literal("prompt_prefix"),
2898
+ Type$2.Literal("user_inline")
2899
+ ], { $id: "ContextBinding" });
2900
+ /**
2901
+ * One context entry. Bytes are inlined: the imposer chose them, and the
2902
+ * task's `inputCid` already pins the entire input — including
2903
+ * `context[]` — so we don't need a separate per-entry hash, fetcher, or
2904
+ * flagged-content gate. Tasks reference rendered packs (or any other
2905
+ * external content) by copying their bytes into `content` at task
2906
+ * creation time.
2907
+ *
2908
+ * - `slug` — short identifier the daemon uses to disambiguate
2909
+ * entries. For `skill` binding it becomes the directory
2910
+ * name under the runtime's skill discovery path. Must be
2911
+ * kebab-case-safe (alphanumeric + dashes/underscores).
2912
+ * - `binding` — how the bytes are delivered to the LLM (see above).
2913
+ * - `content` — the actual bytes (UTF-8 text). Capped at 32 KiB per
2914
+ * entry; total per-task context bytes are bounded by the
2915
+ * soft `maxItems` cap and per-binding daemon limits.
2916
+ */
2917
+ var ContextRef = Type$2.Object({
2918
+ slug: Type$2.String({
2919
+ minLength: 1,
2920
+ maxLength: 64,
2921
+ pattern: "^[a-zA-Z0-9_-]+$"
2922
+ }),
2923
+ binding: ContextBinding,
2924
+ content: Type$2.String({
2925
+ minLength: 1,
2926
+ maxLength: 32768
2927
+ })
2928
+ }, {
2929
+ $id: "ContextRef",
2930
+ additionalProperties: false
2931
+ });
2932
+ /** Reusable input fragment for any task type. Soft cap at 5 items. */
2933
+ var TaskContext = Type$2.Array(ContextRef, {
2934
+ $id: "TaskContext",
2935
+ maxItems: 5
2936
+ });
2937
+ //#endregion
2889
2938
  //#region ../../libs/tasks/src/rubric.ts
2890
2939
  /**
2891
2940
  * Rubric — structured acceptance criteria used by judgment tasks.
@@ -4275,6 +4324,60 @@ var RenderPackOutput = Type$2.Object({
4275
4324
  additionalProperties: false
4276
4325
  });
4277
4326
  //#endregion
4327
+ //#region ../../libs/tasks/src/task-types/run-eval.ts
4328
+ /**
4329
+ * `run_eval` — execute a scenario prompt under a named variant for
4330
+ * later cross-variant grading by `judge_eval_variant` (Slice 2).
4331
+ *
4332
+ * output_kind: artifact
4333
+ * criteria: optional (when set, output.verification is required —
4334
+ * producer self-assessment; the judge is the binding evaluator)
4335
+ * references: not required (scenario lives entirely in input)
4336
+ */
4337
+ var RUN_EVAL_TYPE = "run_eval";
4338
+ var RunEvalInput = Type$2.Object({
4339
+ scenario: Type$2.Object({
4340
+ prompt: Type$2.String({ minLength: 1 }),
4341
+ inputFiles: Type$2.Optional(Type$2.Array(Type$2.String({ minLength: 1 })))
4342
+ }, { additionalProperties: false }),
4343
+ variantLabel: Type$2.String({
4344
+ minLength: 1,
4345
+ maxLength: 64
4346
+ }),
4347
+ context: TaskContext,
4348
+ successCriteria: Type$2.Optional(SuccessCriteria)
4349
+ }, {
4350
+ $id: "RunEvalInput",
4351
+ additionalProperties: false
4352
+ });
4353
+ var RunEvalOutput = Type$2.Object({
4354
+ response: Type$2.String({ minLength: 1 }),
4355
+ artifacts: Type$2.Optional(Type$2.Array(Type$2.Object({
4356
+ path: Type$2.String({ minLength: 1 }),
4357
+ cid: Type$2.String({ minLength: 1 })
4358
+ }, { additionalProperties: false }))),
4359
+ totalTokens: Type$2.Integer({ minimum: 0 }),
4360
+ durationMs: Type$2.Integer({ minimum: 0 }),
4361
+ traceparent: Type$2.String({ minLength: 1 }),
4362
+ verification: Type$2.Optional(VerificationRecord)
4363
+ }, {
4364
+ $id: "RunEvalOutput",
4365
+ additionalProperties: false
4366
+ });
4367
+ /**
4368
+ * Cross-field rule mirroring the `requireVerificationWhenCriteriaPresent`
4369
+ * rule used by the brief task types: when input declares
4370
+ * `successCriteria`, output MUST carry `verification`; when it doesn't,
4371
+ * output MUST NOT carry one.
4372
+ */
4373
+ function validateRunEvalOutput(output, input) {
4374
+ const hasCriteria = input !== null && input !== void 0 && input.successCriteria !== void 0;
4375
+ const hasVerification = output !== null && output !== void 0 && output.verification !== void 0;
4376
+ if (hasCriteria && !hasVerification) return "output.verification is required because input.successCriteria is set; the producer LLM must self-assess against the criteria";
4377
+ if (!hasCriteria && hasVerification) return "output.verification was supplied but input.successCriteria is unset; omit verification when there are no criteria to assess against";
4378
+ return null;
4379
+ }
4380
+ //#endregion
4278
4381
  //#region ../../libs/tasks/src/task-types/index.ts
4279
4382
  /**
4280
4383
  * Validate that a judgment-task input carries a rubric inside its
@@ -4353,6 +4456,14 @@ var BUILT_IN_TASK_TYPES = {
4353
4456
  requiresReferences: true,
4354
4457
  validateInput: validateJudgmentInput,
4355
4458
  validateOutput: validateJudgePackOutput
4459
+ },
4460
+ [RUN_EVAL_TYPE]: {
4461
+ name: RUN_EVAL_TYPE,
4462
+ inputSchema: RunEvalInput,
4463
+ outputSchema: RunEvalOutput,
4464
+ outputKind: "artifact",
4465
+ requiresReferences: false,
4466
+ validateOutput: validateRunEvalOutput
4356
4467
  }
4357
4468
  };
4358
4469
  //#endregion
@@ -5295,6 +5406,14 @@ var ExecutorTrustLevel = Type$2.Union([
5295
5406
  Type$2.Literal("releaseVerifiedTool"),
5296
5407
  Type$2.Literal("sandboxAttested")
5297
5408
  ], { $id: "ExecutorTrustLevel" });
5409
+ /** Identifies a (provider, model) daemon pair allowed to claim a task. */
5410
+ var ExecutorRef = Type$2.Object({
5411
+ provider: Type$2.String({ minLength: 1 }),
5412
+ model: Type$2.String({ minLength: 1 })
5413
+ }, {
5414
+ $id: "ExecutorRef",
5415
+ additionalProperties: false
5416
+ });
5298
5417
  var OutputKind = Type$2.Union([Type$2.Literal("artifact"), Type$2.Literal("judgment")], { $id: "OutputKind" });
5299
5418
  var TaskMessageKind = Type$2.Union([
5300
5419
  Type$2.Literal("text_delta"),
@@ -5387,6 +5506,7 @@ Type$2.Object({
5387
5506
  imposedByHumanId: Type$2.Union([Uuid, Type$2.Null()]),
5388
5507
  acceptedAttemptN: Type$2.Union([Type$2.Number(), Type$2.Null()]),
5389
5508
  requiredExecutorTrustLevel: ExecutorTrustLevel,
5509
+ allowedExecutors: Type$2.Array(ExecutorRef, { maxItems: 16 }),
5390
5510
  status: TaskStatus,
5391
5511
  queuedAt: IsoTimestamp,
5392
5512
  completedAt: Type$2.Union([IsoTimestamp, Type$2.Null()]),
@@ -5608,6 +5728,61 @@ function isHelpFlag(args) {
5608
5728
  return args.includes("--help") || args.includes("-h");
5609
5729
  }
5610
5730
  //#endregion
5731
+ //#region ../../libs/agent-runtime/src/context-bindings.ts
5732
+ var PROMPT_SEPARATOR = "\n\n---\n\n";
5733
+ /**
5734
+ * Resolve `task.input.context[]` into delivered side-effects (skills
5735
+ * persisted via `deliver.skill`) and prompt fragments
5736
+ * (`systemPromptPrefix`, `userInlineSuffix`) the caller weaves into the
5737
+ * built prompt.
5738
+ *
5739
+ * Per-binding semantics (V1):
5740
+ * - `skill` → `deliver.skill({ slug, content })` once per ref.
5741
+ * Slug collisions on distinct contents are
5742
+ * refused loudly.
5743
+ * - `prompt_prefix` → content appended to `systemPromptPrefix` with
5744
+ * the canonical `\n\n---\n\n` separator (in
5745
+ * declared order).
5746
+ * - `user_inline` → content appended to `userInlineSuffix` in
5747
+ * declared order, same separator.
5748
+ *
5749
+ * No fetching, no hashing — bytes are inlined in `ContextRef.content`,
5750
+ * and the task's `inputCid` already pins the entire input. The imposer
5751
+ * chose these bytes; the resolver just dispatches them.
5752
+ *
5753
+ * The function is pure with respect to its arguments: file writes are
5754
+ * confined to the injected `deliver` callback, which makes the
5755
+ * resolver trivial to test.
5756
+ */
5757
+ async function resolveTaskContext(args) {
5758
+ const promptParts = [];
5759
+ const userParts = [];
5760
+ const injected = [];
5761
+ const usedSlugs = /* @__PURE__ */ new Map();
5762
+ for (const ref of args.context) {
5763
+ if (ref.binding === "skill") {
5764
+ const prior = usedSlugs.get(ref.slug);
5765
+ if (prior !== void 0) {
5766
+ if (prior !== ref.content) throw new Error(`slug collision on '${ref.slug}': two skill entries share the same slug but have different content`);
5767
+ injected.push(ref);
5768
+ continue;
5769
+ }
5770
+ usedSlugs.set(ref.slug, ref.content);
5771
+ await args.deliver.skill({
5772
+ slug: ref.slug,
5773
+ content: ref.content
5774
+ });
5775
+ } else if (ref.binding === "prompt_prefix") promptParts.push(ref.content);
5776
+ else userParts.push(ref.content);
5777
+ injected.push(ref);
5778
+ }
5779
+ return {
5780
+ injected,
5781
+ systemPromptPrefix: promptParts.join(PROMPT_SEPARATOR),
5782
+ userInlineSuffix: userParts.join(PROMPT_SEPARATOR)
5783
+ };
5784
+ }
5785
+ //#endregion
5611
5786
  //#region ../../libs/agent-runtime/src/output-tools.ts
5612
5787
  /**
5613
5788
  * Submit-output tool contract.
@@ -5702,7 +5877,7 @@ function buildFinalOutputBlock(opts) {
5702
5877
  //#endregion
5703
5878
  //#region ../../libs/agent-runtime/src/prompts/assess-brief.ts
5704
5879
  /**
5705
- * Build the system prompt for an `assess_brief` judge attempt.
5880
+ * Build the first user-message prompt for an `assess_brief` judge attempt.
5706
5881
  *
5707
5882
  * Design note — no pre-resolved `target` projection
5708
5883
  * --------------------------------------------------
@@ -5723,7 +5898,7 @@ function buildFinalOutputBlock(opts) {
5723
5898
  * future task types whose products are docs / configs / changes /
5724
5899
  * anything) work without any code path here.
5725
5900
  */
5726
- function buildAssessBriefPrompt(input, ctx) {
5901
+ function buildAssessBriefUserPrompt(input, ctx) {
5727
5902
  const rubric = input.successCriteria.rubric;
5728
5903
  const criteriaList = rubric.criteria.map((c, i) => `${i + 1}. **${c.id}** (weight ${c.weight}, scoring: \`${c.scoring}\`) — ${c.description}`).join("\n");
5729
5904
  const preambleSection = rubric.preamble ? [
@@ -5838,7 +6013,7 @@ function buildSelfVerificationBlock(taskId) {
5838
6013
  //#endregion
5839
6014
  //#region ../../libs/agent-runtime/src/prompts/curate-pack.ts
5840
6015
  /**
5841
- * Build the system prompt for a `curate_pack` task.
6016
+ * Build the first user-message prompt for a `curate_pack` task.
5842
6017
  *
5843
6018
  * Design note: this prompt is deliberately NOT a numbered command
5844
6019
  * sequence. The curator's value comes from judgment — inferring scope
@@ -5859,7 +6034,7 @@ function buildSelfVerificationBlock(taskId) {
5859
6034
  * emits pruned state at phase boundaries so a follow-up session can
5860
6035
  * resume without replaying the tool history.
5861
6036
  */
5862
- function buildCuratePackPrompt(input, ctx) {
6037
+ function buildCuratePackUserPrompt(input, ctx) {
5863
6038
  const { diaryId, taskPrompt, entryTypes, tagFilters, tokenBudget, recipe } = input;
5864
6039
  const entryTypesPinned = Boolean(entryTypes);
5865
6040
  const resolvedRecipe = recipe ?? "topic-focused-v1";
@@ -5995,13 +6170,13 @@ function buildCuratePackPrompt(input, ctx) {
5995
6170
  //#endregion
5996
6171
  //#region ../../libs/agent-runtime/src/prompts/fulfill-brief.ts
5997
6172
  /**
5998
- * Build the system prompt for a `fulfill_brief` task.
6173
+ * Build the first user-message prompt for a `fulfill_brief` task.
5999
6174
  *
6000
6175
  * Generalized from the original `resolve-issue` prompt. No longer
6001
6176
  * GitHub-specific; references live on `Task.references[]` and the agent
6002
6177
  * is told to inspect them itself.
6003
6178
  */
6004
- function buildFulfillBriefPrompt(input, ctx) {
6179
+ function buildFulfillBriefUserPrompt(input, ctx) {
6005
6180
  const { brief, title, acceptanceCriteria, seedFiles, scopeHint } = input;
6006
6181
  const criteriaSection = acceptanceCriteria?.length ? [
6007
6182
  "### Acceptance criteria",
@@ -6081,7 +6256,7 @@ function buildFulfillBriefPrompt(input, ctx) {
6081
6256
  }
6082
6257
  //#endregion
6083
6258
  //#region ../../libs/agent-runtime/src/prompts/judge-pack.ts
6084
- function buildJudgePackPrompt(input, ctx) {
6259
+ function buildJudgePackUserPrompt(input, ctx) {
6085
6260
  const { renderedPackId, sourcePackId, successCriteria } = input;
6086
6261
  const rubric = successCriteria.rubric;
6087
6262
  const criteriaList = rubric.criteria.map((c, i) => `${i + 1}. **${c.id}** (weight ${c.weight}, scoring: \`${c.scoring}\`) — ${c.description}`).join("\n");
@@ -6208,10 +6383,10 @@ function buildJudgePackPrompt(input, ctx) {
6208
6383
  //#endregion
6209
6384
  //#region ../../libs/agent-runtime/src/prompts/render-pack.ts
6210
6385
  /**
6211
- * Build the system prompt for a `render_pack` task. Almost mechanical:
6386
+ * Build the first user-message prompt for a `render_pack` task. Almost mechanical:
6212
6387
  * wraps `moltnet_pack_render` and emits the receipt.
6213
6388
  */
6214
- function buildRenderPackPrompt(input, ctx) {
6389
+ function buildRenderPackUserPrompt(input, ctx) {
6215
6390
  const { packId, persist = true, pinned = false } = input;
6216
6391
  return [
6217
6392
  "# Render Pack Agent",
@@ -6265,19 +6440,87 @@ function buildRenderPackPrompt(input, ctx) {
6265
6440
  ].join("\n");
6266
6441
  }
6267
6442
  //#endregion
6443
+ //#region ../../libs/agent-runtime/src/prompts/run-eval.ts
6444
+ /**
6445
+ * Build the first user-message prompt for a `run_eval` task.
6446
+ *
6447
+ * Free-form: no git workflow, no commit ceremony. The executor produces
6448
+ * a textual response (and optional file artifacts) that a later
6449
+ * `judge_eval_variant` task (Slice 2) grades against the rubric.
6450
+ *
6451
+ * Context delivery is handled by `resolveTaskContext` (see
6452
+ * libs/agent-runtime/src/context-bindings.ts) and runs BEFORE this
6453
+ * prompt is rendered: `prompt_prefix` items are concatenated ahead of
6454
+ * the body, `skill` items are persisted at the runtime's skill path,
6455
+ * and `user_inline` items are appended to the first user message. This
6456
+ * builder does NOT inline `input.context[]` itself.
6457
+ */
6458
+ function buildRunEvalUserPrompt(input, ctx) {
6459
+ const { scenario, variantLabel, successCriteria } = input;
6460
+ const inputFilesSection = scenario.inputFiles?.length ? [
6461
+ "### Input files",
6462
+ "",
6463
+ ...scenario.inputFiles.map((f) => `- \`${f}\``),
6464
+ ""
6465
+ ].join("\n") : "";
6466
+ const verificationSection = successCriteria ? buildSelfVerificationBlock(ctx.taskId) : "";
6467
+ const correlationSection = ctx.correlationId ? [
6468
+ "### Correlation",
6469
+ "",
6470
+ `This task carries correlationId \`${ctx.correlationId}\`. It joins`,
6471
+ "this variant to its sibling `run_eval` tasks (other variants of the",
6472
+ "same scenario) and to the eventual `judge_eval_variant` task that",
6473
+ "will grade them together. You do not need to act on it directly —",
6474
+ "it is recorded for cross-variant aggregation at query time.",
6475
+ ""
6476
+ ].join("\n") : "";
6477
+ const finalOutputBlock = buildFinalOutputBlock({
6478
+ taskType: "run_eval",
6479
+ outputSchemaName: "RunEvalOutput",
6480
+ shapeSketch: [
6481
+ "{",
6482
+ " \"response\": \"<your free-form answer>\",",
6483
+ " \"artifacts\": [{ \"path\": \"...\", \"cid\": \"...\" }], // optional",
6484
+ " \"totalTokens\": <int>,",
6485
+ " \"durationMs\": <int>,",
6486
+ " \"traceparent\": \"<from claim>\",",
6487
+ " \"verification\": <required iff input.successCriteria; see Self-verification>",
6488
+ "}"
6489
+ ].join("\n")
6490
+ });
6491
+ return [
6492
+ "# Run Eval Agent\n",
6493
+ `You are running an evaluation scenario as variant \`${variantLabel}\`.\nTask id: \`${ctx.taskId}\`\n`,
6494
+ correlationSection,
6495
+ `### Scenario\n\n${scenario.prompt}\n`,
6496
+ inputFilesSection,
6497
+ verificationSection,
6498
+ finalOutputBlock
6499
+ ].filter((s) => s !== "").join("\n");
6500
+ }
6501
+ //#endregion
6268
6502
  //#region ../../libs/agent-runtime/src/prompts/index.ts
6269
6503
  /**
6270
- * Resolve the correct prompt builder for `task.taskType` and invoke it.
6271
- * Throws if the type is unknown or the input fails TypeBox validation.
6272
- */
6273
- function buildPromptForTask(task, ctx) {
6504
+ * Resolve the correct user-prompt builder for `task.taskType` and
6505
+ * invoke it. Throws if the type is unknown or the input fails TypeBox
6506
+ * validation.
6507
+ *
6508
+ * Role note: the returned string is delivered as the **first user
6509
+ * message** of the agent's session (pi-coding-agent's
6510
+ * `session.prompt(text)` puts text in the user role). The system
6511
+ * prompt is built separately by pi from `appendSystemPrompt` (the
6512
+ * runtime instructor lives there). Builders here are free-form Markdown
6513
+ * for the user turn; they don't replace or prepend to the system
6514
+ * prompt.
6515
+ */
6516
+ function buildTaskUserPrompt(task, ctx) {
6274
6517
  switch (task.taskType) {
6275
6518
  case FULFILL_BRIEF_TYPE:
6276
6519
  if (!Check(FulfillBriefInput, task.input)) {
6277
6520
  const errors = [...Errors(FulfillBriefInput, task.input)];
6278
6521
  throw new Error(`fulfill_brief input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
6279
6522
  }
6280
- return buildFulfillBriefPrompt(task.input, {
6523
+ return buildFulfillBriefUserPrompt(task.input, {
6281
6524
  diaryId: ctx.diaryId,
6282
6525
  taskId: ctx.taskId,
6283
6526
  correlationId: task.correlationId
@@ -6287,7 +6530,7 @@ function buildPromptForTask(task, ctx) {
6287
6530
  const errors = [...Errors(AssessBriefInput, task.input)];
6288
6531
  throw new Error(`assess_brief input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
6289
6532
  }
6290
- return buildAssessBriefPrompt(task.input, {
6533
+ return buildAssessBriefUserPrompt(task.input, {
6291
6534
  diaryId: ctx.diaryId,
6292
6535
  taskId: ctx.taskId
6293
6536
  });
@@ -6296,7 +6539,7 @@ function buildPromptForTask(task, ctx) {
6296
6539
  const errors = [...Errors(CuratePackInput, task.input)];
6297
6540
  throw new Error(`curate_pack input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
6298
6541
  }
6299
- return buildCuratePackPrompt(task.input, {
6542
+ return buildCuratePackUserPrompt(task.input, {
6300
6543
  diaryId: ctx.diaryId,
6301
6544
  taskId: ctx.taskId
6302
6545
  });
@@ -6305,7 +6548,7 @@ function buildPromptForTask(task, ctx) {
6305
6548
  const errors = [...Errors(RenderPackInput, task.input)];
6306
6549
  throw new Error(`render_pack input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
6307
6550
  }
6308
- return buildRenderPackPrompt(task.input, {
6551
+ return buildRenderPackUserPrompt(task.input, {
6309
6552
  diaryId: ctx.diaryId,
6310
6553
  taskId: ctx.taskId
6311
6554
  });
@@ -6314,10 +6557,20 @@ function buildPromptForTask(task, ctx) {
6314
6557
  const errors = [...Errors(JudgePackInput, task.input)];
6315
6558
  throw new Error(`judge_pack input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
6316
6559
  }
6317
- return buildJudgePackPrompt(task.input, {
6560
+ return buildJudgePackUserPrompt(task.input, {
6318
6561
  diaryId: ctx.diaryId,
6319
6562
  taskId: ctx.taskId
6320
6563
  });
6564
+ case RUN_EVAL_TYPE:
6565
+ if (!Check(RunEvalInput, task.input)) {
6566
+ const errors = [...Errors(RunEvalInput, task.input)];
6567
+ throw new Error(`run_eval input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
6568
+ }
6569
+ return buildRunEvalUserPrompt(task.input, {
6570
+ diaryId: ctx.diaryId,
6571
+ taskId: ctx.taskId,
6572
+ correlationId: task.correlationId
6573
+ });
6321
6574
  default: throw new Error(`No prompt builder registered for taskType="${task.taskType}"`);
6322
6575
  }
6323
6576
  }
@@ -9098,13 +9351,31 @@ function problemToError(problem, statusCode) {
9098
9351
  //#endregion
9099
9352
  //#region ../../libs/sdk/src/agent-context.ts
9100
9353
  function unwrapResult(result) {
9101
- if (result.error) {
9354
+ if (result.error !== void 0 && result.error !== null) {
9102
9355
  const error = result.error;
9103
- throw problemToError(error, error.status ?? 500);
9356
+ if (isProblemDetails(error)) throw problemToError(error, error.status);
9357
+ if (error instanceof Error && result.response === void 0) {
9358
+ const networkError = new NetworkError(error.message, { detail: error.cause ? stringifyUnknown(error.cause) : void 0 });
9359
+ networkError.stack = error.stack;
9360
+ throw networkError;
9361
+ }
9362
+ throw new MoltNetError(`Unexpected error from MoltNet API: ${stringifyUnknown(error)}`, { code: "UNKNOWN" });
9104
9363
  }
9105
9364
  if (result.data === void 0) throw new MoltNetError("Unexpected empty response from MoltNet API", { code: "EMPTY_RESPONSE" });
9106
9365
  return result.data;
9107
9366
  }
9367
+ function isProblemDetails(error) {
9368
+ if (!error || typeof error !== "object") return false;
9369
+ return typeof error.status === "number" && ("title" in error || "detail" in error);
9370
+ }
9371
+ function stringifyUnknown(value) {
9372
+ if (value instanceof Error) return `${value.name}: ${value.message}`;
9373
+ try {
9374
+ return JSON.stringify(value) ?? String(value);
9375
+ } catch {
9376
+ return String(value);
9377
+ }
9378
+ }
9108
9379
  function unwrapRequired(result, message, code) {
9109
9380
  if (result.error || !result.data) throw new MoltNetError(message, { code });
9110
9381
  return result.data;
@@ -12870,6 +13141,7 @@ var PollingApiTaskSource = class {
12870
13141
  this.minBackoffMs = opts.pollIntervalMs ?? DEFAULT_POLL_INTERVAL_MS;
12871
13142
  this.maxBackoffMs = opts.maxPollIntervalMs ?? DEFAULT_MAX_POLL_INTERVAL_MS;
12872
13143
  if (this.maxBackoffMs < this.minBackoffMs) throw new Error(`PollingApiTaskSource: maxPollIntervalMs (${this.maxBackoffMs}) must be >= pollIntervalMs (${this.minBackoffMs})`);
13144
+ if (Boolean(opts.provider) !== Boolean(opts.model)) throw new Error("PollingApiTaskSource: provider and model must be set together");
12873
13145
  this.listLimit = opts.listLimit ?? DEFAULT_LIST_LIMIT;
12874
13146
  this.currentBackoffMs = this.minBackoffMs;
12875
13147
  this.logger = (opts.logger ?? pino({ name: "polling-api-source" })).child({ teamId: opts.teamId });
@@ -12903,6 +13175,10 @@ var PollingApiTaskSource = class {
12903
13175
  teamId: this.opts.teamId,
12904
13176
  status: "queued",
12905
13177
  ...taskType ? { taskType } : {},
13178
+ ...this.opts.provider && this.opts.model ? {
13179
+ provider: this.opts.provider,
13180
+ model: this.opts.model
13181
+ } : {},
12906
13182
  limit: this.listLimit
12907
13183
  });
12908
13184
  if (this.opts.debug) this.logger.debug({
@@ -12913,6 +13189,10 @@ var PollingApiTaskSource = class {
12913
13189
  for (const item of result.items) {
12914
13190
  if (seen.has(item.id)) continue;
12915
13191
  if (this.opts.diaryIds && this.opts.diaryIds.length > 0 && (item.diaryId === null || !this.opts.diaryIds.includes(item.diaryId))) continue;
13192
+ if (this.opts.provider && this.opts.model) {
13193
+ const allowed = item.allowedExecutors ?? [];
13194
+ if (allowed.length > 0 && !allowed.some((e) => e.provider === this.opts.provider && e.model === this.opts.model)) continue;
13195
+ }
12916
13196
  if (item.status !== "queued") continue;
12917
13197
  seen.add(item.id);
12918
13198
  out.push(item);
@@ -13906,138 +14186,29 @@ function pruneOldSnapshots(maxCached, currentDir) {
13906
14186
  });
13907
14187
  }
13908
14188
  //#endregion
13909
- //#region ../../libs/pi-extension/src/tool-operations.ts
13910
- /**
13911
- * Gondolin tool operations: redirect pi's built-in tool operations
13912
- * (read, write, edit, bash) to execute inside the VM.
13913
- *
13914
- * Follows the same pattern as upstream pi-gondolin.ts — pi's tool factories
13915
- * accept an `operations` object that provides the underlying I/O.
13916
- */
14189
+ //#region ../../libs/pi-extension/src/vm-manager.ts
13917
14190
  var GUEST_WORKSPACE$1 = "/workspace";
13918
- function shQuote(s) {
13919
- return "'" + s.replace(/'/g, "'\\''") + "'";
13920
- }
13921
14191
  /**
13922
- * Map a host-side absolute path to a guest-side /workspace path.
13923
- * Throws if the path escapes the workspace.
13924
- */
13925
- function toGuestPath(localCwd, localPath) {
13926
- if (localPath === GUEST_WORKSPACE$1 || localPath.startsWith(`${GUEST_WORKSPACE$1}/`)) return localPath;
13927
- const rel = path.relative(localCwd, localPath);
13928
- if (rel === "") return GUEST_WORKSPACE$1;
13929
- if (rel.startsWith("..") || path.isAbsolute(rel)) throw new Error(`path escapes workspace: ${localPath}`);
13930
- const posixRel = rel.split(path.sep).join(path.posix.sep);
13931
- return path.posix.join(GUEST_WORKSPACE$1, posixRel);
13932
- }
13933
- function createGondolinReadOps(vm, localCwd) {
13934
- return {
13935
- readFile: async (p) => {
13936
- const r = await vm.exec(["/bin/cat", toGuestPath(localCwd, p)]);
13937
- if (!r.ok) throw new Error(`cat failed (${r.exitCode}): ${r.stderr}`);
13938
- return r.stdoutBuffer;
13939
- },
13940
- access: async (p) => {
13941
- if (!(await vm.exec([
13942
- "/bin/sh",
13943
- "-lc",
13944
- `test -r ${shQuote(toGuestPath(localCwd, p))}`
13945
- ])).ok) throw new Error(`not readable: ${p}`);
13946
- },
13947
- detectImageMimeType: async (p) => {
13948
- try {
13949
- const r = await vm.exec([
13950
- "/bin/sh",
13951
- "-lc",
13952
- `file --mime-type -b ${shQuote(toGuestPath(localCwd, p))}`
13953
- ]);
13954
- if (!r.ok) return null;
13955
- const m = r.stdout.trim();
13956
- return [
13957
- "image/jpeg",
13958
- "image/png",
13959
- "image/gif",
13960
- "image/webp"
13961
- ].includes(m) ? m : null;
13962
- } catch {
13963
- return null;
13964
- }
13965
- }
13966
- };
13967
- }
13968
- function createGondolinWriteOps(vm, localCwd) {
13969
- return {
13970
- writeFile: async (p, content) => {
13971
- const guestPath = toGuestPath(localCwd, p);
13972
- const dir = path.posix.dirname(guestPath);
13973
- const b64 = Buffer.from(content, "utf8").toString("base64");
13974
- const r = await vm.exec([
13975
- "/bin/sh",
13976
- "-lc",
13977
- [
13978
- "set -eu",
13979
- `mkdir -p ${shQuote(dir)}`,
13980
- `echo ${shQuote(b64)} | base64 -d > ${shQuote(guestPath)}`
13981
- ].join("\n")
13982
- ]);
13983
- if (!r.ok) throw new Error(`write failed (${r.exitCode}): ${r.stderr}`);
13984
- },
13985
- mkdir: async (dir) => {
13986
- const r = await vm.exec([
13987
- "/bin/mkdir",
13988
- "-p",
13989
- toGuestPath(localCwd, dir)
13990
- ]);
13991
- if (!r.ok) throw new Error(`mkdir failed (${r.exitCode}): ${r.stderr}`);
13992
- }
13993
- };
13994
- }
13995
- function createGondolinEditOps(vm, localCwd) {
13996
- const r = createGondolinReadOps(vm, localCwd);
13997
- const w = createGondolinWriteOps(vm, localCwd);
13998
- return {
13999
- readFile: r.readFile,
14000
- access: r.access,
14001
- writeFile: w.writeFile
14002
- };
14003
- }
14004
- function createGondolinBashOps(vm, localCwd) {
14005
- return { exec: async (command, cwd, { onData, signal, timeout, env }) => {
14006
- const guestCwd = toGuestPath(localCwd, cwd);
14007
- const ac = new AbortController();
14008
- const onAbort = () => ac.abort();
14009
- signal?.addEventListener("abort", onAbort, { once: true });
14010
- let timedOut = false;
14011
- const timer = timeout && timeout > 0 ? setTimeout(() => {
14012
- timedOut = true;
14013
- ac.abort();
14014
- }, timeout * 1e3) : void 0;
14015
- try {
14016
- const proc = vm.exec([
14017
- "/bin/sh",
14018
- "-lc",
14019
- command
14020
- ], {
14021
- cwd: guestCwd,
14022
- signal: ac.signal,
14023
- stdout: "pipe",
14024
- stderr: "pipe"
14025
- });
14026
- for await (const chunk of proc.output()) onData(typeof chunk.data === "string" ? Buffer.from(chunk.data, "utf8") : chunk.data);
14027
- return { exitCode: (await proc).exitCode };
14028
- } catch (err) {
14029
- if (signal?.aborted) throw new Error("aborted");
14030
- if (timedOut) throw new Error(`timeout:${timeout}`);
14031
- throw err;
14032
- } finally {
14033
- if (timer) clearTimeout(timer);
14034
- signal?.removeEventListener("abort", onAbort);
14035
- }
14036
- } };
14037
- }
14038
- //#endregion
14039
- //#region ../../libs/pi-extension/src/vm-manager.ts
14040
- var GUEST_WORKSPACE = "/workspace";
14192
+ * Memory-backed VFS mount used by the daemon to inject task-context
14193
+ * skills (#943 slice 1.5). Sibling of /workspace, NOT a sub-path —
14194
+ * Gondolin mounts can't nest. The agent's Gondolin-bound Read tool
14195
+ * accepts paths under this prefix (see toGuestPath in tool-operations.ts).
14196
+ *
14197
+ * Why MemoryProvider rather than a path under /workspace:
14198
+ * - Injected skills are ephemeral by intent: per-task-attempt input
14199
+ * scoped to the VM lifetime. MemoryProvider models that exactly —
14200
+ * in-memory, per-VM-instance, zero host artefacts, automatic
14201
+ * cleanup on VM close.
14202
+ * - Writing under /workspace fails in worktrees because we symlink
14203
+ * `.moltnet/` to the main repo (so credentials are reachable from
14204
+ * worktrees), and Gondolin's RealFSProvider correctly refuses to
14205
+ * create paths whose ancestors' realpath escapes the mount root.
14206
+ * That refusal is a deliberate sandbox-escape protection, not a
14207
+ * bug. See diary semantic entry cd27d9d3-efdc-4aec-ac0d-5fd8ce258d1f
14208
+ * and episodic 7affbfeb-18a2-4963-aeac-c177eb2afa2d for the full
14209
+ * investigation and the alternatives we rejected.
14210
+ */
14211
+ var GUEST_TASK_SKILLS_MOUNT = "/moltnet-task-skills";
14041
14212
  /**
14042
14213
  * Resolve the main worktree root (where .moltnet/ lives — it's untracked,
14043
14214
  * only exists in the main worktree, not in git worktrees).
@@ -14166,7 +14337,10 @@ async function resumeVm(config) {
14166
14337
  env: vmEnv,
14167
14338
  ...resources?.memory && { memory: resources.memory },
14168
14339
  ...resources?.cpus && { cpus: resources.cpus },
14169
- vfs: { mounts: { [GUEST_WORKSPACE]: workspaceProvider } }
14340
+ vfs: { mounts: {
14341
+ [GUEST_WORKSPACE$1]: workspaceProvider,
14342
+ [GUEST_TASK_SKILLS_MOUNT]: new MemoryProvider()
14343
+ } }
14170
14344
  });
14171
14345
  await vm.exec(`sh -c '
14172
14346
  cp /etc/gondolin/mitm/ca.crt /usr/local/share/ca-certificates/gondolin-mitm.crt
@@ -14196,7 +14370,7 @@ nameserver 1.1.1.1" > /etc/resolv.conf'`);
14196
14370
  vm,
14197
14371
  credentials: creds,
14198
14372
  mountPath: config.mountPath,
14199
- guestWorkspace: GUEST_WORKSPACE,
14373
+ guestWorkspace: GUEST_WORKSPACE$1,
14200
14374
  agentDir
14201
14375
  };
14202
14376
  }
@@ -14249,6 +14423,137 @@ function ensureRelativeWorktreePaths(gitconfig) {
14249
14423
  return `${gitconfig}${gitconfig.endsWith("\n") ? "" : "\n"}[worktree]\n\tuseRelativePaths = true\n`;
14250
14424
  }
14251
14425
  //#endregion
14426
+ //#region ../../libs/pi-extension/src/tool-operations.ts
14427
+ /**
14428
+ * Gondolin tool operations: redirect pi's built-in tool operations
14429
+ * (read, write, edit, bash) to execute inside the VM.
14430
+ *
14431
+ * Follows the same pattern as upstream pi-gondolin.ts — pi's tool factories
14432
+ * accept an `operations` object that provides the underlying I/O.
14433
+ */
14434
+ var GUEST_WORKSPACE = "/workspace";
14435
+ function shQuote(s) {
14436
+ return "'" + s.replace(/'/g, "'\\''") + "'";
14437
+ }
14438
+ /**
14439
+ * Map a host-side absolute path to a guest-side /workspace path.
14440
+ * Throws if the path escapes the workspace.
14441
+ */
14442
+ function toGuestPath(localCwd, localPath) {
14443
+ if (localPath === GUEST_WORKSPACE || localPath.startsWith(`${GUEST_WORKSPACE}/`)) return localPath;
14444
+ if (localPath === "/moltnet-task-skills" || localPath.startsWith(`/moltnet-task-skills/`)) return localPath;
14445
+ const rel = path.relative(localCwd, localPath);
14446
+ if (rel === "") return GUEST_WORKSPACE;
14447
+ if (rel.startsWith("..") || path.isAbsolute(rel)) throw new Error(`path escapes workspace: ${localPath}`);
14448
+ const posixRel = rel.split(path.sep).join(path.posix.sep);
14449
+ return path.posix.join(GUEST_WORKSPACE, posixRel);
14450
+ }
14451
+ function createGondolinReadOps(vm, localCwd) {
14452
+ return {
14453
+ readFile: async (p) => {
14454
+ const r = await vm.exec(["/bin/cat", toGuestPath(localCwd, p)]);
14455
+ if (!r.ok) throw new Error(`cat failed (${r.exitCode}): ${r.stderr}`);
14456
+ return r.stdoutBuffer;
14457
+ },
14458
+ access: async (p) => {
14459
+ if (!(await vm.exec([
14460
+ "/bin/sh",
14461
+ "-lc",
14462
+ `test -r ${shQuote(toGuestPath(localCwd, p))}`
14463
+ ])).ok) throw new Error(`not readable: ${p}`);
14464
+ },
14465
+ detectImageMimeType: async (p) => {
14466
+ try {
14467
+ const r = await vm.exec([
14468
+ "/bin/sh",
14469
+ "-lc",
14470
+ `file --mime-type -b ${shQuote(toGuestPath(localCwd, p))}`
14471
+ ]);
14472
+ if (!r.ok) return null;
14473
+ const m = r.stdout.trim();
14474
+ return [
14475
+ "image/jpeg",
14476
+ "image/png",
14477
+ "image/gif",
14478
+ "image/webp"
14479
+ ].includes(m) ? m : null;
14480
+ } catch {
14481
+ return null;
14482
+ }
14483
+ }
14484
+ };
14485
+ }
14486
+ function createGondolinWriteOps(vm, localCwd) {
14487
+ return {
14488
+ writeFile: async (p, content) => {
14489
+ const guestPath = toGuestPath(localCwd, p);
14490
+ const dir = path.posix.dirname(guestPath);
14491
+ const b64 = Buffer.from(content, "utf8").toString("base64");
14492
+ const r = await vm.exec([
14493
+ "/bin/sh",
14494
+ "-lc",
14495
+ [
14496
+ "set -eu",
14497
+ `mkdir -p ${shQuote(dir)}`,
14498
+ `echo ${shQuote(b64)} | base64 -d > ${shQuote(guestPath)}`
14499
+ ].join("\n")
14500
+ ]);
14501
+ if (!r.ok) throw new Error(`write failed (${r.exitCode}): ${r.stderr}`);
14502
+ },
14503
+ mkdir: async (dir) => {
14504
+ const r = await vm.exec([
14505
+ "/bin/mkdir",
14506
+ "-p",
14507
+ toGuestPath(localCwd, dir)
14508
+ ]);
14509
+ if (!r.ok) throw new Error(`mkdir failed (${r.exitCode}): ${r.stderr}`);
14510
+ }
14511
+ };
14512
+ }
14513
+ function createGondolinEditOps(vm, localCwd) {
14514
+ const r = createGondolinReadOps(vm, localCwd);
14515
+ const w = createGondolinWriteOps(vm, localCwd);
14516
+ return {
14517
+ readFile: r.readFile,
14518
+ access: r.access,
14519
+ writeFile: w.writeFile
14520
+ };
14521
+ }
14522
+ function createGondolinBashOps(vm, localCwd) {
14523
+ return { exec: async (command, cwd, { onData, signal, timeout, env }) => {
14524
+ const guestCwd = toGuestPath(localCwd, cwd);
14525
+ const ac = new AbortController();
14526
+ const onAbort = () => ac.abort();
14527
+ signal?.addEventListener("abort", onAbort, { once: true });
14528
+ let timedOut = false;
14529
+ const timer = timeout && timeout > 0 ? setTimeout(() => {
14530
+ timedOut = true;
14531
+ ac.abort();
14532
+ }, timeout * 1e3) : void 0;
14533
+ try {
14534
+ const proc = vm.exec([
14535
+ "/bin/sh",
14536
+ "-lc",
14537
+ command
14538
+ ], {
14539
+ cwd: guestCwd,
14540
+ signal: ac.signal,
14541
+ stdout: "pipe",
14542
+ stderr: "pipe"
14543
+ });
14544
+ for await (const chunk of proc.output()) onData(typeof chunk.data === "string" ? Buffer.from(chunk.data, "utf8") : chunk.data);
14545
+ return { exitCode: (await proc).exitCode };
14546
+ } catch (err) {
14547
+ if (signal?.aborted) throw new Error("aborted");
14548
+ if (timedOut) throw new Error(`timeout:${timeout}`);
14549
+ throw err;
14550
+ } finally {
14551
+ if (timer) clearTimeout(timer);
14552
+ signal?.removeEventListener("abort", onAbort);
14553
+ }
14554
+ } };
14555
+ }
14556
+ //#endregion
14252
14557
  //#region ../../libs/pi-extension/src/otel/index.ts
14253
14558
  var TRACER_NAME = "@themoltnet/pi-extension/otel";
14254
14559
  function stripReservedAttrs(attrs) {
@@ -14386,6 +14691,114 @@ function extractUsage(message) {
14386
14691
  };
14387
14692
  }
14388
14693
  //#endregion
14694
+ //#region ../../libs/pi-extension/src/runtime/inject-task-context.ts
14695
+ /**
14696
+ * Slice 1.5 of #943 — wire the agent-runtime resolver into the
14697
+ * pi-extension execution path.
14698
+ *
14699
+ * `resolveTaskContext` is a pure dispatcher; this module provides the
14700
+ * Gondolin-aware deliverer and the post-resolution shape the
14701
+ * `execute-pi-task` caller needs to splice into pi's setup:
14702
+ *
14703
+ * - `systemPromptPrefix` → fed into `appendSystemPrompt` alongside
14704
+ * the runtime instructor (it IS a system-prompt fragment).
14705
+ * - `userInlineSuffix` → appended to the `buildTaskUserPrompt`
14706
+ * output BEFORE `session.prompt(text)`.
14707
+ * - `skills` → spliced into the `skillsOverride` callback's
14708
+ * return value. pi includes them in `<available_skills>` in the
14709
+ * system prompt; the agent fetches the body on demand via the
14710
+ * Read tool.
14711
+ *
14712
+ * Skill files are written into the VM at
14713
+ * `/workspace/.moltnet/skills/<slug>/SKILL.md`. The agent's
14714
+ * Gondolin-bound Read tool is scoped to `/workspace`, so that path is
14715
+ * the only location the agent can actually read at runtime. pi only
14716
+ * reads `<available_skills>` metadata (name, description, location),
14717
+ * never the file body, so we construct synthetic `Skill` objects
14718
+ * pointing at the in-VM path without ever materialising the file on
14719
+ * the host.
14720
+ */
14721
+ /**
14722
+ * Where in the VM we write skill bodies — the memory-backed mount
14723
+ * declared in `vm-manager.ts`. See the comment on
14724
+ * `GUEST_TASK_SKILLS_MOUNT` there for the full rationale (ephemeral
14725
+ * by intent + the worktree symlink interaction with Gondolin's
14726
+ * sandbox-escape protection). The agent's Gondolin Read tool accepts
14727
+ * paths under this mount via `toGuestPath` in `tool-operations.ts`.
14728
+ */
14729
+ var SKILL_ROOT_IN_VM = GUEST_TASK_SKILLS_MOUNT;
14730
+ /** Bounds borrowed from pi's skill validation; conservative caps so a
14731
+ * malformed SKILL.md doesn't bloat the system prompt. */
14732
+ var MAX_SKILL_NAME = 64;
14733
+ var MAX_SKILL_DESCRIPTION = 1024;
14734
+ /**
14735
+ * Resolve a task's `input.context[]` and inject the side effects pi
14736
+ * needs. Safe to call with an empty array — returns an inert result.
14737
+ */
14738
+ async function injectTaskContext(args) {
14739
+ const skills = [];
14740
+ const resolved = await resolveTaskContext({
14741
+ context: args.context,
14742
+ deliver: { skill: async ({ slug, content }) => {
14743
+ const dir = `${SKILL_ROOT_IN_VM}/${slug}`;
14744
+ const filePath = `${dir}/SKILL.md`;
14745
+ await args.fs.mkdir(dir, { recursive: true });
14746
+ await args.fs.writeFile(filePath, content, { mode: 420 });
14747
+ skills.push(buildSyntheticSkill({
14748
+ slug,
14749
+ content,
14750
+ filePath,
14751
+ dir
14752
+ }));
14753
+ } }
14754
+ });
14755
+ return {
14756
+ injected: resolved.injected,
14757
+ skills,
14758
+ systemPromptPrefix: resolved.systemPromptPrefix,
14759
+ userInlineSuffix: resolved.userInlineSuffix
14760
+ };
14761
+ }
14762
+ /**
14763
+ * Build a `Skill` object pi will faithfully render in
14764
+ * `<available_skills>`. We extract `name` and `description` from the
14765
+ * skill content's YAML frontmatter using pi's own `parseFrontmatter`
14766
+ * helper (proper YAML, not a regex hack) and fall back to the slug +
14767
+ * a generic description so a SKILL.md without frontmatter still
14768
+ * renders something meaningful.
14769
+ *
14770
+ * Frontmatter parsing is best-effort: a malformed YAML block is
14771
+ * optional metadata, not a reason to fail the task. We swallow parser
14772
+ * errors and fall back to the slug-derived metadata; the skill body
14773
+ * is unaffected.
14774
+ *
14775
+ * pi's `formatSkillsForPrompt` only reads `name`, `description`, and
14776
+ * `filePath` — `sourceInfo`/`baseDir` exist on the type but never
14777
+ * surface in the prompt, so a synthetic `SourceInfo` is enough.
14778
+ */
14779
+ function buildSyntheticSkill(args) {
14780
+ let fm = {};
14781
+ try {
14782
+ fm = parseFrontmatter(args.content).frontmatter;
14783
+ } catch {}
14784
+ return {
14785
+ name: clip(typeof fm.name === "string" && fm.name.trim().length > 0 ? fm.name.trim() : args.slug, MAX_SKILL_NAME),
14786
+ description: clip(typeof fm.description === "string" && fm.description.trim().length > 0 ? fm.description.trim() : `Task-injected context skill (${args.slug})`, MAX_SKILL_DESCRIPTION),
14787
+ filePath: args.filePath,
14788
+ baseDir: args.dir,
14789
+ sourceInfo: createSyntheticSourceInfo(args.filePath, {
14790
+ source: "moltnet:task-context",
14791
+ scope: "temporary",
14792
+ origin: "top-level",
14793
+ baseDir: args.dir
14794
+ }),
14795
+ disableModelInvocation: fm["disable-model-invocation"] === true
14796
+ };
14797
+ }
14798
+ function clip(s, max) {
14799
+ return s.length > max ? s.slice(0, max) : s;
14800
+ }
14801
+ //#endregion
14389
14802
  //#region ../../libs/pi-extension/src/runtime/runtime-instructor.ts
14390
14803
  /**
14391
14804
  * Build the daemon-controlled invariant prose injected into the system prompt
@@ -14709,6 +15122,7 @@ function resolveSubmitTools(taskType, opts = {}) {
14709
15122
  * Anthropic-SDK one) plug in via the `executeTask` function injected into
14710
15123
  * `AgentRuntime`.
14711
15124
  */
15125
+ var noopTurnEventHandler = () => {};
14712
15126
  /**
14713
15127
  * Factory that builds a pi-specific `executeTask` function suitable for
14714
15128
  * injection into `AgentRuntime`. The returned function caches the resolved
@@ -14805,10 +15219,25 @@ async function executePiTask(claimedTask, reporter, opts) {
14805
15219
  attemptN
14806
15220
  });
14807
15221
  reporterOpen = true;
14808
- const emit = (kind, payload) => reporter.record({
14809
- kind,
14810
- payload
14811
- });
15222
+ let onTurnEvent;
15223
+ if (opts.makeOnTurnEvent) try {
15224
+ onTurnEvent = opts.makeOnTurnEvent(claimedTask);
15225
+ } catch (err) {
15226
+ process.stderr.write(`[emit] makeOnTurnEvent threw: ${err instanceof Error ? err.message : String(err)}\n`);
15227
+ onTurnEvent = noopTurnEventHandler;
15228
+ }
15229
+ else onTurnEvent = opts.onTurnEvent ?? noopTurnEventHandler;
15230
+ const emit = (kind, payload) => {
15231
+ try {
15232
+ onTurnEvent(kind, summarizePayloadForLog(kind, payload));
15233
+ } catch (err) {
15234
+ process.stderr.write(`[emit] onTurnEvent threw for kind="${kind}": ${err instanceof Error ? err.message : String(err)}\n`);
15235
+ }
15236
+ return reporter.record({
15237
+ kind,
15238
+ payload
15239
+ });
15240
+ };
14812
15241
  await emit("info", {
14813
15242
  event: "execute_start",
14814
15243
  taskType: task.taskType,
@@ -14818,7 +15247,7 @@ async function executePiTask(claimedTask, reporter, opts) {
14818
15247
  });
14819
15248
  let taskPrompt;
14820
15249
  try {
14821
- taskPrompt = buildPromptForTask(task, {
15250
+ taskPrompt = buildTaskUserPrompt(task, {
14822
15251
  diaryId,
14823
15252
  taskId: task.id,
14824
15253
  extras: opts.promptExtras
@@ -14831,6 +15260,30 @@ async function executePiTask(claimedTask, reporter, opts) {
14831
15260
  });
14832
15261
  return makeFailedOutput("prompt_build_failed", message);
14833
15262
  }
15263
+ const rawContext = task.input.context;
15264
+ let injectedContext;
15265
+ try {
15266
+ const contextArray = rawContext === void 0 ? [] : rawContext;
15267
+ if (!Check(TaskContext, contextArray)) throw new Error(`task.input.context failed TaskContext validation: ${JSON.stringify([...Errors(TaskContext, contextArray)].slice(0, 3))}`);
15268
+ injectedContext = await injectTaskContext({
15269
+ context: contextArray,
15270
+ fs: managed.vm.fs
15271
+ });
15272
+ } catch (err) {
15273
+ const message = err instanceof Error ? err.message : String(err);
15274
+ await emit("error", {
15275
+ message,
15276
+ phase: "context_resolution"
15277
+ });
15278
+ return makeFailedOutput("context_resolution_failed", message);
15279
+ }
15280
+ if (injectedContext.injected.length > 0) await emit("info", {
15281
+ event: "context_injected",
15282
+ count: injectedContext.injected.length,
15283
+ bindings: injectedContext.injected.map((r) => r.binding),
15284
+ slugs: injectedContext.injected.map((r) => r.slug)
15285
+ });
15286
+ if (injectedContext.userInlineSuffix) taskPrompt = `${taskPrompt}\n\n---\n\n${injectedContext.userInlineSuffix}`;
14834
15287
  const gondolinCustomTools = [
14835
15288
  createReadToolDefinition(mountPath, { operations: createGondolinReadOps(managed.vm, mountPath) }),
14836
15289
  createWriteToolDefinition(mountPath, { operations: createGondolinWriteOps(managed.vm, mountPath) }),
@@ -14867,21 +15320,23 @@ async function executePiTask(claimedTask, reporter, opts) {
14867
15320
  "moltnet.task.type": task.taskType
14868
15321
  }
14869
15322
  });
14870
- const runtimeInstructor = buildRuntimeInstructor({
15323
+ const appendSystemPrompt = [buildRuntimeInstructor({
14871
15324
  taskId: task.id,
14872
15325
  taskType: task.taskType,
14873
15326
  attemptN,
14874
15327
  diaryId,
14875
15328
  agentName: opts.agentName,
14876
15329
  correlationId: task.correlationId ?? null
14877
- });
15330
+ })];
15331
+ if (injectedContext.systemPromptPrefix) appendSystemPrompt.push(injectedContext.systemPromptPrefix);
15332
+ const injectedSkills = injectedContext.skills;
14878
15333
  const resourceLoader = new DefaultResourceLoader({
14879
15334
  cwd: mountPath,
14880
15335
  agentDir: piAuthDir,
14881
15336
  extensionFactories: [piOtelExtension],
14882
- appendSystemPrompt: [runtimeInstructor],
15337
+ appendSystemPrompt,
14883
15338
  skillsOverride: () => ({
14884
- skills: [],
15339
+ skills: injectedSkills,
14885
15340
  diagnostics: []
14886
15341
  })
14887
15342
  });
@@ -15106,6 +15561,27 @@ function wireSessionAbort(cancelSignal, session) {
15106
15561
  * `task_messages.payload` row. Bodies above 4 KiB are replaced with a
15107
15562
  * `{ truncated, original_size }` marker so the JSONL/DB size stays bounded.
15108
15563
  */
15564
+ function summarizePayloadForLog(kind, payload) {
15565
+ switch (kind) {
15566
+ case "text_delta": {
15567
+ const delta = payload.delta;
15568
+ return { chars: typeof delta === "string" ? delta.length : 0 };
15569
+ }
15570
+ case "tool_call_start": return { tool: payload.tool_name };
15571
+ case "tool_call_end": return {
15572
+ tool: payload.tool_name,
15573
+ is_error: payload.is_error === true,
15574
+ ...payload.is_error === true && payload.result !== void 0 ? { result: payload.result } : {}
15575
+ };
15576
+ case "turn_end": return { stop_reason: payload.stop_reason };
15577
+ case "error": return {
15578
+ phase: payload.phase,
15579
+ message: typeof payload.message === "string" ? payload.message.slice(0, TRUNCATE_LIMIT) : payload.message
15580
+ };
15581
+ case "info": return Object.fromEntries(Object.entries(payload).map(([k, v]) => [k, typeof v === "string" ? v.slice(0, TRUNCATE_LIMIT) : v]));
15582
+ default: return payload;
15583
+ }
15584
+ }
15109
15585
  var TRUNCATE_LIMIT = 4 * 1024;
15110
15586
  function truncateForWire(value) {
15111
15587
  if (value === null || value === void 0) return value;
@@ -15405,6 +15881,27 @@ function findUp(startDir, filename) {
15405
15881
  }
15406
15882
  }
15407
15883
  //#endregion
15884
+ //#region src/lib/turn-event-logger.ts
15885
+ function makeTurnEventHandler(base, context = {}) {
15886
+ const log = base.child({
15887
+ name: "agent-daemon.turn",
15888
+ ...context
15889
+ });
15890
+ return (event, summary) => {
15891
+ if (event === "text_delta") return;
15892
+ log[event === "error" ? "warn" : event === "turn_end" ? "info" : "debug"]({
15893
+ event,
15894
+ ...summary
15895
+ }, `turn.${event}`);
15896
+ };
15897
+ }
15898
+ function makeTurnEventHandlerFactory(base) {
15899
+ return (claimedTask) => makeTurnEventHandler(base, {
15900
+ taskId: claimedTask.task.id,
15901
+ attemptN: claimedTask.attemptN
15902
+ });
15903
+ }
15904
+ //#endregion
15408
15905
  //#region src/cli/poll-shared.ts
15409
15906
  async function runPolling(opts) {
15410
15907
  if (isHelpFlag(opts.argv)) {
@@ -15507,7 +16004,8 @@ async function runPolling(opts) {
15507
16004
  mountPath: sandbox.rootDir,
15508
16005
  provider: common.provider,
15509
16006
  model: common.model,
15510
- sandboxConfig: sandbox.config
16007
+ sandboxConfig: sandbox.config,
16008
+ makeOnTurnEvent: makeTurnEventHandlerFactory(rootLogger)
15511
16009
  });
15512
16010
  runtime = new AgentRuntime({
15513
16011
  logger: rootLogger,
@@ -15515,6 +16013,8 @@ async function runPolling(opts) {
15515
16013
  agent: ctx.agent,
15516
16014
  teamId,
15517
16015
  taskTypes: taskTypes.length > 0 ? taskTypes : void 0,
16016
+ provider: common.provider.toLowerCase(),
16017
+ model: common.model.toLowerCase(),
15518
16018
  diaryIds: diaryIds.length > 0 ? diaryIds : void 0,
15519
16019
  leaseTtlSec: common.leaseTtlSec,
15520
16020
  listLimit,
@@ -15698,7 +16198,8 @@ async function runOnce(argv) {
15698
16198
  mountPath: sandbox.rootDir,
15699
16199
  provider: opts.provider,
15700
16200
  model: opts.model,
15701
- sandboxConfig: sandbox.config
16201
+ sandboxConfig: sandbox.config,
16202
+ onTurnEvent: makeTurnEventHandler(rootLogger, { taskId })
15702
16203
  });
15703
16204
  const writeCorrelationAnchors = makePrBodyAnchorWriter({
15704
16205
  gh: createGhCliClient(),
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@themoltnet/agent-daemon",
3
- "version": "0.4.0",
3
+ "version": "0.5.0",
4
4
  "license": "AGPL-3.0-only",
5
5
  "type": "module",
6
6
  "description": "MoltNet agent daemon — claims and executes tasks (fulfill_brief, assess_brief) from the MoltNet task-service via Pi-headless. CLI: moltnet-agent.",
@@ -33,9 +33,9 @@
33
33
  "@opentelemetry/semantic-conventions": "^1.39.0",
34
34
  "pino": "^10.3.1",
35
35
  "pino-pretty": "^13.1.3",
36
- "@themoltnet/agent-runtime": "0.11.0",
37
- "@themoltnet/sdk": "0.99.0",
38
- "@themoltnet/pi-extension": "0.13.5"
36
+ "@themoltnet/agent-runtime": "0.12.0",
37
+ "@themoltnet/pi-extension": "0.14.0",
38
+ "@themoltnet/sdk": "0.100.0"
39
39
  },
40
40
  "devDependencies": {
41
41
  "tsx": "^4.7.0",
@@ -43,9 +43,9 @@
43
43
  "vite": "^8.0.0",
44
44
  "vitest": "^3.0.0",
45
45
  "@moltnet/bootstrap": "0.1.0",
46
- "@moltnet/database": "0.1.0",
46
+ "@moltnet/crypto-service": "0.1.0",
47
47
  "@moltnet/tasks": "0.1.0",
48
- "@moltnet/crypto-service": "0.1.0"
48
+ "@moltnet/database": "0.1.0"
49
49
  },
50
50
  "nx": {
51
51
  "tags": [