@themoltnet/agent-daemon 0.3.1 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/main.js +758 -171
- package/package.json +6 -6
package/dist/main.js
CHANGED
|
@@ -9,9 +9,9 @@ import { createHash as createHash$1 } from "node:crypto";
|
|
|
9
9
|
import path, { dirname, isAbsolute, join, resolve } from "node:path";
|
|
10
10
|
import { homedir } from "node:os";
|
|
11
11
|
import { execFile, execFileSync } from "node:child_process";
|
|
12
|
-
import { DefaultResourceLoader, SessionManager, createAgentSession, createBashToolDefinition, createEditToolDefinition, createReadToolDefinition, createWriteToolDefinition, defineTool } from "@earendil-works/pi-coding-agent";
|
|
12
|
+
import { DefaultResourceLoader, SessionManager, createAgentSession, createBashToolDefinition, createEditToolDefinition, createReadToolDefinition, createSyntheticSourceInfo, createWriteToolDefinition, defineTool, parseFrontmatter } from "@earendil-works/pi-coding-agent";
|
|
13
13
|
import { Type, getModel } from "@earendil-works/pi-ai";
|
|
14
|
-
import { RealFSProvider, ShadowProvider, VM, VmCheckpoint, createHttpHooks, createShadowPathPredicate, ensureImageSelector, loadGuestAssets } from "@earendil-works/gondolin";
|
|
14
|
+
import { MemoryProvider, RealFSProvider, ShadowProvider, VM, VmCheckpoint, createHttpHooks, createShadowPathPredicate, ensureImageSelector, loadGuestAssets } from "@earendil-works/gondolin";
|
|
15
15
|
import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto";
|
|
16
16
|
import { resourceFromAttributes } from "@opentelemetry/resources";
|
|
17
17
|
import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base";
|
|
@@ -2886,6 +2886,55 @@ var UUID_RE = /^[0-9a-f]{8}-[0-9a-f]{4}-[1-5][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a
|
|
|
2886
2886
|
if (!Has$1("uuid")) Set$1("uuid", (v) => UUID_RE.test(v));
|
|
2887
2887
|
if (!Has$1("date-time")) Set$1("date-time", (v) => !Number.isNaN(Date.parse(v)));
|
|
2888
2888
|
//#endregion
|
|
2889
|
+
//#region ../../libs/tasks/src/context.ts
|
|
2890
|
+
/**
|
|
2891
|
+
* How an executor delivers a context entry to its underlying LLM.
|
|
2892
|
+
* V1 bindings only; Tier-2 (reference_file, mcp_resource, imported_file,
|
|
2893
|
+
* tool_response_seed, additional_context_hook) ship in a later slice.
|
|
2894
|
+
*/
|
|
2895
|
+
var ContextBinding = Type$2.Union([
|
|
2896
|
+
Type$2.Literal("skill"),
|
|
2897
|
+
Type$2.Literal("prompt_prefix"),
|
|
2898
|
+
Type$2.Literal("user_inline")
|
|
2899
|
+
], { $id: "ContextBinding" });
|
|
2900
|
+
/**
|
|
2901
|
+
* One context entry. Bytes are inlined: the imposer chose them, and the
|
|
2902
|
+
* task's `inputCid` already pins the entire input — including
|
|
2903
|
+
* `context[]` — so we don't need a separate per-entry hash, fetcher, or
|
|
2904
|
+
* flagged-content gate. Tasks reference rendered packs (or any other
|
|
2905
|
+
* external content) by copying their bytes into `content` at task
|
|
2906
|
+
* creation time.
|
|
2907
|
+
*
|
|
2908
|
+
* - `slug` — short identifier the daemon uses to disambiguate
|
|
2909
|
+
* entries. For `skill` binding it becomes the directory
|
|
2910
|
+
* name under the runtime's skill discovery path. Must be
|
|
2911
|
+
* kebab-case-safe (alphanumeric + dashes/underscores).
|
|
2912
|
+
* - `binding` — how the bytes are delivered to the LLM (see above).
|
|
2913
|
+
* - `content` — the actual bytes (UTF-8 text). Capped at 32 KiB per
|
|
2914
|
+
* entry; total per-task context bytes are bounded by the
|
|
2915
|
+
* soft `maxItems` cap and per-binding daemon limits.
|
|
2916
|
+
*/
|
|
2917
|
+
var ContextRef = Type$2.Object({
|
|
2918
|
+
slug: Type$2.String({
|
|
2919
|
+
minLength: 1,
|
|
2920
|
+
maxLength: 64,
|
|
2921
|
+
pattern: "^[a-zA-Z0-9_-]+$"
|
|
2922
|
+
}),
|
|
2923
|
+
binding: ContextBinding,
|
|
2924
|
+
content: Type$2.String({
|
|
2925
|
+
minLength: 1,
|
|
2926
|
+
maxLength: 32768
|
|
2927
|
+
})
|
|
2928
|
+
}, {
|
|
2929
|
+
$id: "ContextRef",
|
|
2930
|
+
additionalProperties: false
|
|
2931
|
+
});
|
|
2932
|
+
/** Reusable input fragment for any task type. Soft cap at 5 items. */
|
|
2933
|
+
var TaskContext = Type$2.Array(ContextRef, {
|
|
2934
|
+
$id: "TaskContext",
|
|
2935
|
+
maxItems: 5
|
|
2936
|
+
});
|
|
2937
|
+
//#endregion
|
|
2889
2938
|
//#region ../../libs/tasks/src/rubric.ts
|
|
2890
2939
|
/**
|
|
2891
2940
|
* Rubric — structured acceptance criteria used by judgment tasks.
|
|
@@ -4275,6 +4324,60 @@ var RenderPackOutput = Type$2.Object({
|
|
|
4275
4324
|
additionalProperties: false
|
|
4276
4325
|
});
|
|
4277
4326
|
//#endregion
|
|
4327
|
+
//#region ../../libs/tasks/src/task-types/run-eval.ts
|
|
4328
|
+
/**
|
|
4329
|
+
* `run_eval` — execute a scenario prompt under a named variant for
|
|
4330
|
+
* later cross-variant grading by `judge_eval_variant` (Slice 2).
|
|
4331
|
+
*
|
|
4332
|
+
* output_kind: artifact
|
|
4333
|
+
* criteria: optional (when set, output.verification is required —
|
|
4334
|
+
* producer self-assessment; the judge is the binding evaluator)
|
|
4335
|
+
* references: not required (scenario lives entirely in input)
|
|
4336
|
+
*/
|
|
4337
|
+
var RUN_EVAL_TYPE = "run_eval";
|
|
4338
|
+
var RunEvalInput = Type$2.Object({
|
|
4339
|
+
scenario: Type$2.Object({
|
|
4340
|
+
prompt: Type$2.String({ minLength: 1 }),
|
|
4341
|
+
inputFiles: Type$2.Optional(Type$2.Array(Type$2.String({ minLength: 1 })))
|
|
4342
|
+
}, { additionalProperties: false }),
|
|
4343
|
+
variantLabel: Type$2.String({
|
|
4344
|
+
minLength: 1,
|
|
4345
|
+
maxLength: 64
|
|
4346
|
+
}),
|
|
4347
|
+
context: TaskContext,
|
|
4348
|
+
successCriteria: Type$2.Optional(SuccessCriteria)
|
|
4349
|
+
}, {
|
|
4350
|
+
$id: "RunEvalInput",
|
|
4351
|
+
additionalProperties: false
|
|
4352
|
+
});
|
|
4353
|
+
var RunEvalOutput = Type$2.Object({
|
|
4354
|
+
response: Type$2.String({ minLength: 1 }),
|
|
4355
|
+
artifacts: Type$2.Optional(Type$2.Array(Type$2.Object({
|
|
4356
|
+
path: Type$2.String({ minLength: 1 }),
|
|
4357
|
+
cid: Type$2.String({ minLength: 1 })
|
|
4358
|
+
}, { additionalProperties: false }))),
|
|
4359
|
+
totalTokens: Type$2.Integer({ minimum: 0 }),
|
|
4360
|
+
durationMs: Type$2.Integer({ minimum: 0 }),
|
|
4361
|
+
traceparent: Type$2.String({ minLength: 1 }),
|
|
4362
|
+
verification: Type$2.Optional(VerificationRecord)
|
|
4363
|
+
}, {
|
|
4364
|
+
$id: "RunEvalOutput",
|
|
4365
|
+
additionalProperties: false
|
|
4366
|
+
});
|
|
4367
|
+
/**
|
|
4368
|
+
* Cross-field rule mirroring the `requireVerificationWhenCriteriaPresent`
|
|
4369
|
+
* rule used by the brief task types: when input declares
|
|
4370
|
+
* `successCriteria`, output MUST carry `verification`; when it doesn't,
|
|
4371
|
+
* output MUST NOT carry one.
|
|
4372
|
+
*/
|
|
4373
|
+
function validateRunEvalOutput(output, input) {
|
|
4374
|
+
const hasCriteria = input !== null && input !== void 0 && input.successCriteria !== void 0;
|
|
4375
|
+
const hasVerification = output !== null && output !== void 0 && output.verification !== void 0;
|
|
4376
|
+
if (hasCriteria && !hasVerification) return "output.verification is required because input.successCriteria is set; the producer LLM must self-assess against the criteria";
|
|
4377
|
+
if (!hasCriteria && hasVerification) return "output.verification was supplied but input.successCriteria is unset; omit verification when there are no criteria to assess against";
|
|
4378
|
+
return null;
|
|
4379
|
+
}
|
|
4380
|
+
//#endregion
|
|
4278
4381
|
//#region ../../libs/tasks/src/task-types/index.ts
|
|
4279
4382
|
/**
|
|
4280
4383
|
* Validate that a judgment-task input carries a rubric inside its
|
|
@@ -4353,6 +4456,14 @@ var BUILT_IN_TASK_TYPES = {
|
|
|
4353
4456
|
requiresReferences: true,
|
|
4354
4457
|
validateInput: validateJudgmentInput,
|
|
4355
4458
|
validateOutput: validateJudgePackOutput
|
|
4459
|
+
},
|
|
4460
|
+
[RUN_EVAL_TYPE]: {
|
|
4461
|
+
name: RUN_EVAL_TYPE,
|
|
4462
|
+
inputSchema: RunEvalInput,
|
|
4463
|
+
outputSchema: RunEvalOutput,
|
|
4464
|
+
outputKind: "artifact",
|
|
4465
|
+
requiresReferences: false,
|
|
4466
|
+
validateOutput: validateRunEvalOutput
|
|
4356
4467
|
}
|
|
4357
4468
|
};
|
|
4358
4469
|
//#endregion
|
|
@@ -5295,6 +5406,14 @@ var ExecutorTrustLevel = Type$2.Union([
|
|
|
5295
5406
|
Type$2.Literal("releaseVerifiedTool"),
|
|
5296
5407
|
Type$2.Literal("sandboxAttested")
|
|
5297
5408
|
], { $id: "ExecutorTrustLevel" });
|
|
5409
|
+
/** Identifies a (provider, model) daemon pair allowed to claim a task. */
|
|
5410
|
+
var ExecutorRef = Type$2.Object({
|
|
5411
|
+
provider: Type$2.String({ minLength: 1 }),
|
|
5412
|
+
model: Type$2.String({ minLength: 1 })
|
|
5413
|
+
}, {
|
|
5414
|
+
$id: "ExecutorRef",
|
|
5415
|
+
additionalProperties: false
|
|
5416
|
+
});
|
|
5298
5417
|
var OutputKind = Type$2.Union([Type$2.Literal("artifact"), Type$2.Literal("judgment")], { $id: "OutputKind" });
|
|
5299
5418
|
var TaskMessageKind = Type$2.Union([
|
|
5300
5419
|
Type$2.Literal("text_delta"),
|
|
@@ -5387,6 +5506,7 @@ Type$2.Object({
|
|
|
5387
5506
|
imposedByHumanId: Type$2.Union([Uuid, Type$2.Null()]),
|
|
5388
5507
|
acceptedAttemptN: Type$2.Union([Type$2.Number(), Type$2.Null()]),
|
|
5389
5508
|
requiredExecutorTrustLevel: ExecutorTrustLevel,
|
|
5509
|
+
allowedExecutors: Type$2.Array(ExecutorRef, { maxItems: 16 }),
|
|
5390
5510
|
status: TaskStatus,
|
|
5391
5511
|
queuedAt: IsoTimestamp,
|
|
5392
5512
|
completedAt: Type$2.Union([IsoTimestamp, Type$2.Null()]),
|
|
@@ -5608,6 +5728,61 @@ function isHelpFlag(args) {
|
|
|
5608
5728
|
return args.includes("--help") || args.includes("-h");
|
|
5609
5729
|
}
|
|
5610
5730
|
//#endregion
|
|
5731
|
+
//#region ../../libs/agent-runtime/src/context-bindings.ts
|
|
5732
|
+
var PROMPT_SEPARATOR = "\n\n---\n\n";
|
|
5733
|
+
/**
|
|
5734
|
+
* Resolve `task.input.context[]` into delivered side-effects (skills
|
|
5735
|
+
* persisted via `deliver.skill`) and prompt fragments
|
|
5736
|
+
* (`systemPromptPrefix`, `userInlineSuffix`) the caller weaves into the
|
|
5737
|
+
* built prompt.
|
|
5738
|
+
*
|
|
5739
|
+
* Per-binding semantics (V1):
|
|
5740
|
+
* - `skill` → `deliver.skill({ slug, content })` once per ref.
|
|
5741
|
+
* Slug collisions on distinct contents are
|
|
5742
|
+
* refused loudly.
|
|
5743
|
+
* - `prompt_prefix` → content appended to `systemPromptPrefix` with
|
|
5744
|
+
* the canonical `\n\n---\n\n` separator (in
|
|
5745
|
+
* declared order).
|
|
5746
|
+
* - `user_inline` → content appended to `userInlineSuffix` in
|
|
5747
|
+
* declared order, same separator.
|
|
5748
|
+
*
|
|
5749
|
+
* No fetching, no hashing — bytes are inlined in `ContextRef.content`,
|
|
5750
|
+
* and the task's `inputCid` already pins the entire input. The imposer
|
|
5751
|
+
* chose these bytes; the resolver just dispatches them.
|
|
5752
|
+
*
|
|
5753
|
+
* The function is pure with respect to its arguments: file writes are
|
|
5754
|
+
* confined to the injected `deliver` callback, which makes the
|
|
5755
|
+
* resolver trivial to test.
|
|
5756
|
+
*/
|
|
5757
|
+
async function resolveTaskContext(args) {
|
|
5758
|
+
const promptParts = [];
|
|
5759
|
+
const userParts = [];
|
|
5760
|
+
const injected = [];
|
|
5761
|
+
const usedSlugs = /* @__PURE__ */ new Map();
|
|
5762
|
+
for (const ref of args.context) {
|
|
5763
|
+
if (ref.binding === "skill") {
|
|
5764
|
+
const prior = usedSlugs.get(ref.slug);
|
|
5765
|
+
if (prior !== void 0) {
|
|
5766
|
+
if (prior !== ref.content) throw new Error(`slug collision on '${ref.slug}': two skill entries share the same slug but have different content`);
|
|
5767
|
+
injected.push(ref);
|
|
5768
|
+
continue;
|
|
5769
|
+
}
|
|
5770
|
+
usedSlugs.set(ref.slug, ref.content);
|
|
5771
|
+
await args.deliver.skill({
|
|
5772
|
+
slug: ref.slug,
|
|
5773
|
+
content: ref.content
|
|
5774
|
+
});
|
|
5775
|
+
} else if (ref.binding === "prompt_prefix") promptParts.push(ref.content);
|
|
5776
|
+
else userParts.push(ref.content);
|
|
5777
|
+
injected.push(ref);
|
|
5778
|
+
}
|
|
5779
|
+
return {
|
|
5780
|
+
injected,
|
|
5781
|
+
systemPromptPrefix: promptParts.join(PROMPT_SEPARATOR),
|
|
5782
|
+
userInlineSuffix: userParts.join(PROMPT_SEPARATOR)
|
|
5783
|
+
};
|
|
5784
|
+
}
|
|
5785
|
+
//#endregion
|
|
5611
5786
|
//#region ../../libs/agent-runtime/src/output-tools.ts
|
|
5612
5787
|
/**
|
|
5613
5788
|
* Submit-output tool contract.
|
|
@@ -5702,7 +5877,7 @@ function buildFinalOutputBlock(opts) {
|
|
|
5702
5877
|
//#endregion
|
|
5703
5878
|
//#region ../../libs/agent-runtime/src/prompts/assess-brief.ts
|
|
5704
5879
|
/**
|
|
5705
|
-
* Build the
|
|
5880
|
+
* Build the first user-message prompt for an `assess_brief` judge attempt.
|
|
5706
5881
|
*
|
|
5707
5882
|
* Design note — no pre-resolved `target` projection
|
|
5708
5883
|
* --------------------------------------------------
|
|
@@ -5723,7 +5898,7 @@ function buildFinalOutputBlock(opts) {
|
|
|
5723
5898
|
* future task types whose products are docs / configs / changes /
|
|
5724
5899
|
* anything) work without any code path here.
|
|
5725
5900
|
*/
|
|
5726
|
-
function
|
|
5901
|
+
function buildAssessBriefUserPrompt(input, ctx) {
|
|
5727
5902
|
const rubric = input.successCriteria.rubric;
|
|
5728
5903
|
const criteriaList = rubric.criteria.map((c, i) => `${i + 1}. **${c.id}** (weight ${c.weight}, scoring: \`${c.scoring}\`) — ${c.description}`).join("\n");
|
|
5729
5904
|
const preambleSection = rubric.preamble ? [
|
|
@@ -5838,7 +6013,7 @@ function buildSelfVerificationBlock(taskId) {
|
|
|
5838
6013
|
//#endregion
|
|
5839
6014
|
//#region ../../libs/agent-runtime/src/prompts/curate-pack.ts
|
|
5840
6015
|
/**
|
|
5841
|
-
* Build the
|
|
6016
|
+
* Build the first user-message prompt for a `curate_pack` task.
|
|
5842
6017
|
*
|
|
5843
6018
|
* Design note: this prompt is deliberately NOT a numbered command
|
|
5844
6019
|
* sequence. The curator's value comes from judgment — inferring scope
|
|
@@ -5859,7 +6034,7 @@ function buildSelfVerificationBlock(taskId) {
|
|
|
5859
6034
|
* emits pruned state at phase boundaries so a follow-up session can
|
|
5860
6035
|
* resume without replaying the tool history.
|
|
5861
6036
|
*/
|
|
5862
|
-
function
|
|
6037
|
+
function buildCuratePackUserPrompt(input, ctx) {
|
|
5863
6038
|
const { diaryId, taskPrompt, entryTypes, tagFilters, tokenBudget, recipe } = input;
|
|
5864
6039
|
const entryTypesPinned = Boolean(entryTypes);
|
|
5865
6040
|
const resolvedRecipe = recipe ?? "topic-focused-v1";
|
|
@@ -5995,13 +6170,13 @@ function buildCuratePackPrompt(input, ctx) {
|
|
|
5995
6170
|
//#endregion
|
|
5996
6171
|
//#region ../../libs/agent-runtime/src/prompts/fulfill-brief.ts
|
|
5997
6172
|
/**
|
|
5998
|
-
* Build the
|
|
6173
|
+
* Build the first user-message prompt for a `fulfill_brief` task.
|
|
5999
6174
|
*
|
|
6000
6175
|
* Generalized from the original `resolve-issue` prompt. No longer
|
|
6001
6176
|
* GitHub-specific; references live on `Task.references[]` and the agent
|
|
6002
6177
|
* is told to inspect them itself.
|
|
6003
6178
|
*/
|
|
6004
|
-
function
|
|
6179
|
+
function buildFulfillBriefUserPrompt(input, ctx) {
|
|
6005
6180
|
const { brief, title, acceptanceCriteria, seedFiles, scopeHint } = input;
|
|
6006
6181
|
const criteriaSection = acceptanceCriteria?.length ? [
|
|
6007
6182
|
"### Acceptance criteria",
|
|
@@ -6081,7 +6256,7 @@ function buildFulfillBriefPrompt(input, ctx) {
|
|
|
6081
6256
|
}
|
|
6082
6257
|
//#endregion
|
|
6083
6258
|
//#region ../../libs/agent-runtime/src/prompts/judge-pack.ts
|
|
6084
|
-
function
|
|
6259
|
+
function buildJudgePackUserPrompt(input, ctx) {
|
|
6085
6260
|
const { renderedPackId, sourcePackId, successCriteria } = input;
|
|
6086
6261
|
const rubric = successCriteria.rubric;
|
|
6087
6262
|
const criteriaList = rubric.criteria.map((c, i) => `${i + 1}. **${c.id}** (weight ${c.weight}, scoring: \`${c.scoring}\`) — ${c.description}`).join("\n");
|
|
@@ -6208,10 +6383,10 @@ function buildJudgePackPrompt(input, ctx) {
|
|
|
6208
6383
|
//#endregion
|
|
6209
6384
|
//#region ../../libs/agent-runtime/src/prompts/render-pack.ts
|
|
6210
6385
|
/**
|
|
6211
|
-
* Build the
|
|
6386
|
+
* Build the first user-message prompt for a `render_pack` task. Almost mechanical:
|
|
6212
6387
|
* wraps `moltnet_pack_render` and emits the receipt.
|
|
6213
6388
|
*/
|
|
6214
|
-
function
|
|
6389
|
+
function buildRenderPackUserPrompt(input, ctx) {
|
|
6215
6390
|
const { packId, persist = true, pinned = false } = input;
|
|
6216
6391
|
return [
|
|
6217
6392
|
"# Render Pack Agent",
|
|
@@ -6265,19 +6440,87 @@ function buildRenderPackPrompt(input, ctx) {
|
|
|
6265
6440
|
].join("\n");
|
|
6266
6441
|
}
|
|
6267
6442
|
//#endregion
|
|
6443
|
+
//#region ../../libs/agent-runtime/src/prompts/run-eval.ts
|
|
6444
|
+
/**
|
|
6445
|
+
* Build the first user-message prompt for a `run_eval` task.
|
|
6446
|
+
*
|
|
6447
|
+
* Free-form: no git workflow, no commit ceremony. The executor produces
|
|
6448
|
+
* a textual response (and optional file artifacts) that a later
|
|
6449
|
+
* `judge_eval_variant` task (Slice 2) grades against the rubric.
|
|
6450
|
+
*
|
|
6451
|
+
* Context delivery is handled by `resolveTaskContext` (see
|
|
6452
|
+
* libs/agent-runtime/src/context-bindings.ts) and runs BEFORE this
|
|
6453
|
+
* prompt is rendered: `prompt_prefix` items are concatenated ahead of
|
|
6454
|
+
* the body, `skill` items are persisted at the runtime's skill path,
|
|
6455
|
+
* and `user_inline` items are appended to the first user message. This
|
|
6456
|
+
* builder does NOT inline `input.context[]` itself.
|
|
6457
|
+
*/
|
|
6458
|
+
function buildRunEvalUserPrompt(input, ctx) {
|
|
6459
|
+
const { scenario, variantLabel, successCriteria } = input;
|
|
6460
|
+
const inputFilesSection = scenario.inputFiles?.length ? [
|
|
6461
|
+
"### Input files",
|
|
6462
|
+
"",
|
|
6463
|
+
...scenario.inputFiles.map((f) => `- \`${f}\``),
|
|
6464
|
+
""
|
|
6465
|
+
].join("\n") : "";
|
|
6466
|
+
const verificationSection = successCriteria ? buildSelfVerificationBlock(ctx.taskId) : "";
|
|
6467
|
+
const correlationSection = ctx.correlationId ? [
|
|
6468
|
+
"### Correlation",
|
|
6469
|
+
"",
|
|
6470
|
+
`This task carries correlationId \`${ctx.correlationId}\`. It joins`,
|
|
6471
|
+
"this variant to its sibling `run_eval` tasks (other variants of the",
|
|
6472
|
+
"same scenario) and to the eventual `judge_eval_variant` task that",
|
|
6473
|
+
"will grade them together. You do not need to act on it directly —",
|
|
6474
|
+
"it is recorded for cross-variant aggregation at query time.",
|
|
6475
|
+
""
|
|
6476
|
+
].join("\n") : "";
|
|
6477
|
+
const finalOutputBlock = buildFinalOutputBlock({
|
|
6478
|
+
taskType: "run_eval",
|
|
6479
|
+
outputSchemaName: "RunEvalOutput",
|
|
6480
|
+
shapeSketch: [
|
|
6481
|
+
"{",
|
|
6482
|
+
" \"response\": \"<your free-form answer>\",",
|
|
6483
|
+
" \"artifacts\": [{ \"path\": \"...\", \"cid\": \"...\" }], // optional",
|
|
6484
|
+
" \"totalTokens\": <int>,",
|
|
6485
|
+
" \"durationMs\": <int>,",
|
|
6486
|
+
" \"traceparent\": \"<from claim>\",",
|
|
6487
|
+
" \"verification\": <required iff input.successCriteria; see Self-verification>",
|
|
6488
|
+
"}"
|
|
6489
|
+
].join("\n")
|
|
6490
|
+
});
|
|
6491
|
+
return [
|
|
6492
|
+
"# Run Eval Agent\n",
|
|
6493
|
+
`You are running an evaluation scenario as variant \`${variantLabel}\`.\nTask id: \`${ctx.taskId}\`\n`,
|
|
6494
|
+
correlationSection,
|
|
6495
|
+
`### Scenario\n\n${scenario.prompt}\n`,
|
|
6496
|
+
inputFilesSection,
|
|
6497
|
+
verificationSection,
|
|
6498
|
+
finalOutputBlock
|
|
6499
|
+
].filter((s) => s !== "").join("\n");
|
|
6500
|
+
}
|
|
6501
|
+
//#endregion
|
|
6268
6502
|
//#region ../../libs/agent-runtime/src/prompts/index.ts
|
|
6269
6503
|
/**
|
|
6270
|
-
* Resolve the correct prompt builder for `task.taskType` and
|
|
6271
|
-
* Throws if the type is unknown or the input fails TypeBox
|
|
6272
|
-
|
|
6273
|
-
|
|
6504
|
+
* Resolve the correct user-prompt builder for `task.taskType` and
|
|
6505
|
+
* invoke it. Throws if the type is unknown or the input fails TypeBox
|
|
6506
|
+
* validation.
|
|
6507
|
+
*
|
|
6508
|
+
* Role note: the returned string is delivered as the **first user
|
|
6509
|
+
* message** of the agent's session (pi-coding-agent's
|
|
6510
|
+
* `session.prompt(text)` puts text in the user role). The system
|
|
6511
|
+
* prompt is built separately by pi from `appendSystemPrompt` (the
|
|
6512
|
+
* runtime instructor lives there). Builders here are free-form Markdown
|
|
6513
|
+
* for the user turn; they don't replace or prepend to the system
|
|
6514
|
+
* prompt.
|
|
6515
|
+
*/
|
|
6516
|
+
function buildTaskUserPrompt(task, ctx) {
|
|
6274
6517
|
switch (task.taskType) {
|
|
6275
6518
|
case FULFILL_BRIEF_TYPE:
|
|
6276
6519
|
if (!Check(FulfillBriefInput, task.input)) {
|
|
6277
6520
|
const errors = [...Errors(FulfillBriefInput, task.input)];
|
|
6278
6521
|
throw new Error(`fulfill_brief input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
|
|
6279
6522
|
}
|
|
6280
|
-
return
|
|
6523
|
+
return buildFulfillBriefUserPrompt(task.input, {
|
|
6281
6524
|
diaryId: ctx.diaryId,
|
|
6282
6525
|
taskId: ctx.taskId,
|
|
6283
6526
|
correlationId: task.correlationId
|
|
@@ -6287,7 +6530,7 @@ function buildPromptForTask(task, ctx) {
|
|
|
6287
6530
|
const errors = [...Errors(AssessBriefInput, task.input)];
|
|
6288
6531
|
throw new Error(`assess_brief input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
|
|
6289
6532
|
}
|
|
6290
|
-
return
|
|
6533
|
+
return buildAssessBriefUserPrompt(task.input, {
|
|
6291
6534
|
diaryId: ctx.diaryId,
|
|
6292
6535
|
taskId: ctx.taskId
|
|
6293
6536
|
});
|
|
@@ -6296,7 +6539,7 @@ function buildPromptForTask(task, ctx) {
|
|
|
6296
6539
|
const errors = [...Errors(CuratePackInput, task.input)];
|
|
6297
6540
|
throw new Error(`curate_pack input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
|
|
6298
6541
|
}
|
|
6299
|
-
return
|
|
6542
|
+
return buildCuratePackUserPrompt(task.input, {
|
|
6300
6543
|
diaryId: ctx.diaryId,
|
|
6301
6544
|
taskId: ctx.taskId
|
|
6302
6545
|
});
|
|
@@ -6305,7 +6548,7 @@ function buildPromptForTask(task, ctx) {
|
|
|
6305
6548
|
const errors = [...Errors(RenderPackInput, task.input)];
|
|
6306
6549
|
throw new Error(`render_pack input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
|
|
6307
6550
|
}
|
|
6308
|
-
return
|
|
6551
|
+
return buildRenderPackUserPrompt(task.input, {
|
|
6309
6552
|
diaryId: ctx.diaryId,
|
|
6310
6553
|
taskId: ctx.taskId
|
|
6311
6554
|
});
|
|
@@ -6314,16 +6557,45 @@ function buildPromptForTask(task, ctx) {
|
|
|
6314
6557
|
const errors = [...Errors(JudgePackInput, task.input)];
|
|
6315
6558
|
throw new Error(`judge_pack input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
|
|
6316
6559
|
}
|
|
6317
|
-
return
|
|
6560
|
+
return buildJudgePackUserPrompt(task.input, {
|
|
6318
6561
|
diaryId: ctx.diaryId,
|
|
6319
6562
|
taskId: ctx.taskId
|
|
6320
6563
|
});
|
|
6564
|
+
case RUN_EVAL_TYPE:
|
|
6565
|
+
if (!Check(RunEvalInput, task.input)) {
|
|
6566
|
+
const errors = [...Errors(RunEvalInput, task.input)];
|
|
6567
|
+
throw new Error(`run_eval input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
|
|
6568
|
+
}
|
|
6569
|
+
return buildRunEvalUserPrompt(task.input, {
|
|
6570
|
+
diaryId: ctx.diaryId,
|
|
6571
|
+
taskId: ctx.taskId,
|
|
6572
|
+
correlationId: task.correlationId
|
|
6573
|
+
});
|
|
6321
6574
|
default: throw new Error(`No prompt builder registered for taskType="${task.taskType}"`);
|
|
6322
6575
|
}
|
|
6323
6576
|
}
|
|
6324
6577
|
//#endregion
|
|
6325
6578
|
//#region ../../libs/agent-runtime/src/reporters/api.ts
|
|
6326
6579
|
/**
|
|
6580
|
+
* True when an error looks like the Keto "tuple lagged the read API"
|
|
6581
|
+
* race: HTTP 403 plus a server message starting with "Not authorized".
|
|
6582
|
+
* Used to gate the first-append retry to the narrow recoverable case.
|
|
6583
|
+
*
|
|
6584
|
+
* The server's task.service.ts returns errors of the form
|
|
6585
|
+
* `Not authorized to <verb> ...`
|
|
6586
|
+
* for every claimant-relation check (append messages, list messages,
|
|
6587
|
+
* heartbeat, complete, fail). Other Fastify 403s — auth-plugin
|
|
6588
|
+
* rejection, route-level guards — use different message shapes and
|
|
6589
|
+
* are not consistency-window flakes; retrying them just delays the
|
|
6590
|
+
* permanent failure surfacing.
|
|
6591
|
+
*/
|
|
6592
|
+
function isKetoConsistencyLag403(err) {
|
|
6593
|
+
if (!err || typeof err !== "object") return false;
|
|
6594
|
+
if (err.statusCode !== 403) return false;
|
|
6595
|
+
const message = err.message ?? "";
|
|
6596
|
+
return /Not authorized/i.test(message);
|
|
6597
|
+
}
|
|
6598
|
+
/**
|
|
6327
6599
|
* TaskReporter backed by the Tasks API via the SDK's TasksNamespace.
|
|
6328
6600
|
*
|
|
6329
6601
|
* - `open()` fires an immediate heartbeat (satisfies DBOS recv('started', 300s))
|
|
@@ -6349,6 +6621,15 @@ var ApiTaskReporter = class {
|
|
|
6349
6621
|
* failures are never silently dropped by the batching layer.
|
|
6350
6622
|
*/
|
|
6351
6623
|
pendingError = null;
|
|
6624
|
+
/**
|
|
6625
|
+
* Flips to `true` after the first successful `appendMessages` call.
|
|
6626
|
+
* Used to gate the 403-retry window in `flush` to the very first
|
|
6627
|
+
* append (the only point where the Keto `Task:claimant#Agent` tuple
|
|
6628
|
+
* can lag the read API). Once any append has succeeded, Keto is
|
|
6629
|
+
* known consistent for this task and a 403 thereafter is a real
|
|
6630
|
+
* authorization failure that should surface immediately.
|
|
6631
|
+
*/
|
|
6632
|
+
firstAppendSucceeded = false;
|
|
6352
6633
|
cancelController = new AbortController();
|
|
6353
6634
|
observedCancelReason = null;
|
|
6354
6635
|
maxBatchSize;
|
|
@@ -6376,6 +6657,7 @@ var ApiTaskReporter = class {
|
|
|
6376
6657
|
}
|
|
6377
6658
|
this.taskId = ctx.taskId;
|
|
6378
6659
|
this.attemptN = ctx.attemptN;
|
|
6660
|
+
this.firstAppendSucceeded = false;
|
|
6379
6661
|
await this.sendInitialHeartbeat();
|
|
6380
6662
|
const intervalMs = this.opts.heartbeatIntervalMs ?? 6e4;
|
|
6381
6663
|
if (intervalMs > 0) this.heartbeatTimer = setInterval(() => {
|
|
@@ -6421,7 +6703,8 @@ var ApiTaskReporter = class {
|
|
|
6421
6703
|
const batch = this.buffer.splice(0, this.buffer.length);
|
|
6422
6704
|
this.inFlight = (async () => {
|
|
6423
6705
|
try {
|
|
6424
|
-
await this.
|
|
6706
|
+
await this.appendWithFirstCallRetry(batch);
|
|
6707
|
+
this.firstAppendSucceeded = true;
|
|
6425
6708
|
} catch (err) {
|
|
6426
6709
|
const overflowCap = this.maxBatchSize * 3;
|
|
6427
6710
|
const restoredCount = batch.length;
|
|
@@ -6498,18 +6781,53 @@ var ApiTaskReporter = class {
|
|
|
6498
6781
|
async sendInitialHeartbeat() {
|
|
6499
6782
|
const maxAttempts = 5;
|
|
6500
6783
|
const baseDelayMs = 100;
|
|
6501
|
-
let lastErr;
|
|
6502
6784
|
for (let attempt = 1; attempt <= maxAttempts; attempt += 1) try {
|
|
6503
6785
|
await this.sendHeartbeat();
|
|
6504
6786
|
return;
|
|
6505
6787
|
} catch (err) {
|
|
6506
|
-
lastErr = err;
|
|
6507
6788
|
if ((err && typeof err === "object" && "statusCode" in err ? err.statusCode : void 0) !== 403 || attempt === maxAttempts) throw err;
|
|
6508
6789
|
await new Promise((resolve) => {
|
|
6509
6790
|
setTimeout(resolve, baseDelayMs * attempt);
|
|
6510
6791
|
});
|
|
6511
6792
|
}
|
|
6512
|
-
|
|
6793
|
+
}
|
|
6794
|
+
/**
|
|
6795
|
+
* Wrap `tasks.appendMessages` with a bounded 403-retry on the first
|
|
6796
|
+
* call only. Mirrors `sendInitialHeartbeat` — the same Keto
|
|
6797
|
+
* `Task:claimant#Agent` tuple race covers both endpoints, and a
|
|
6798
|
+
* batched-up first flush triggered immediately after `open()` (e.g.
|
|
6799
|
+
* a fast executor that records a `task_started` info message before
|
|
6800
|
+
* any heartbeat round-trip completes) can hit a 403 even though the
|
|
6801
|
+
* heartbeat itself eventually succeeds.
|
|
6802
|
+
*
|
|
6803
|
+
* Once `firstAppendSucceeded` is set we know Keto is consistent for
|
|
6804
|
+
* this task and any subsequent 403 is a real authorization failure
|
|
6805
|
+
* (e.g. another agent stole the claim) that must surface immediately.
|
|
6806
|
+
*
|
|
6807
|
+
* Retry budget is intentionally tight: the documented Keto
|
|
6808
|
+
* consistency window is "tens of milliseconds," so 100 ms + 200 ms
|
|
6809
|
+
* (3 attempts total) covers the realistic window with margin without
|
|
6810
|
+
* adding meaningful latency to the permanent-failure path (wrong
|
|
6811
|
+
* task ID, claim revoked, etc.). We also string-match the server's
|
|
6812
|
+
* `Not authorized` message to avoid retrying 403s that mean
|
|
6813
|
+
* something else (Fastify routes can 403 for non-auth reasons too).
|
|
6814
|
+
*/
|
|
6815
|
+
async appendWithFirstCallRetry(batch) {
|
|
6816
|
+
if (this.firstAppendSucceeded) {
|
|
6817
|
+
await this.opts.tasks.appendMessages(this.taskId, this.attemptN, { messages: batch });
|
|
6818
|
+
return;
|
|
6819
|
+
}
|
|
6820
|
+
const maxAttempts = 3;
|
|
6821
|
+
const baseDelayMs = 100;
|
|
6822
|
+
for (let attempt = 1; attempt <= maxAttempts; attempt += 1) try {
|
|
6823
|
+
await this.opts.tasks.appendMessages(this.taskId, this.attemptN, { messages: batch });
|
|
6824
|
+
return;
|
|
6825
|
+
} catch (err) {
|
|
6826
|
+
if (attempt === maxAttempts || !isKetoConsistencyLag403(err)) throw err;
|
|
6827
|
+
await new Promise((resolve) => {
|
|
6828
|
+
setTimeout(resolve, baseDelayMs * attempt);
|
|
6829
|
+
});
|
|
6830
|
+
}
|
|
6513
6831
|
}
|
|
6514
6832
|
async sendHeartbeat() {
|
|
6515
6833
|
const body = this.opts.leaseTtlSec ? { leaseTtlSec: this.opts.leaseTtlSec } : {};
|
|
@@ -9033,13 +9351,31 @@ function problemToError(problem, statusCode) {
|
|
|
9033
9351
|
//#endregion
|
|
9034
9352
|
//#region ../../libs/sdk/src/agent-context.ts
|
|
9035
9353
|
function unwrapResult(result) {
|
|
9036
|
-
if (result.error) {
|
|
9354
|
+
if (result.error !== void 0 && result.error !== null) {
|
|
9037
9355
|
const error = result.error;
|
|
9038
|
-
throw problemToError(error, error.status
|
|
9356
|
+
if (isProblemDetails(error)) throw problemToError(error, error.status);
|
|
9357
|
+
if (error instanceof Error && result.response === void 0) {
|
|
9358
|
+
const networkError = new NetworkError(error.message, { detail: error.cause ? stringifyUnknown(error.cause) : void 0 });
|
|
9359
|
+
networkError.stack = error.stack;
|
|
9360
|
+
throw networkError;
|
|
9361
|
+
}
|
|
9362
|
+
throw new MoltNetError(`Unexpected error from MoltNet API: ${stringifyUnknown(error)}`, { code: "UNKNOWN" });
|
|
9039
9363
|
}
|
|
9040
9364
|
if (result.data === void 0) throw new MoltNetError("Unexpected empty response from MoltNet API", { code: "EMPTY_RESPONSE" });
|
|
9041
9365
|
return result.data;
|
|
9042
9366
|
}
|
|
9367
|
+
function isProblemDetails(error) {
|
|
9368
|
+
if (!error || typeof error !== "object") return false;
|
|
9369
|
+
return typeof error.status === "number" && ("title" in error || "detail" in error);
|
|
9370
|
+
}
|
|
9371
|
+
function stringifyUnknown(value) {
|
|
9372
|
+
if (value instanceof Error) return `${value.name}: ${value.message}`;
|
|
9373
|
+
try {
|
|
9374
|
+
return JSON.stringify(value) ?? String(value);
|
|
9375
|
+
} catch {
|
|
9376
|
+
return String(value);
|
|
9377
|
+
}
|
|
9378
|
+
}
|
|
9043
9379
|
function unwrapRequired(result, message, code) {
|
|
9044
9380
|
if (result.error || !result.data) throw new MoltNetError(message, { code });
|
|
9045
9381
|
return result.data;
|
|
@@ -12805,6 +13141,7 @@ var PollingApiTaskSource = class {
|
|
|
12805
13141
|
this.minBackoffMs = opts.pollIntervalMs ?? DEFAULT_POLL_INTERVAL_MS;
|
|
12806
13142
|
this.maxBackoffMs = opts.maxPollIntervalMs ?? DEFAULT_MAX_POLL_INTERVAL_MS;
|
|
12807
13143
|
if (this.maxBackoffMs < this.minBackoffMs) throw new Error(`PollingApiTaskSource: maxPollIntervalMs (${this.maxBackoffMs}) must be >= pollIntervalMs (${this.minBackoffMs})`);
|
|
13144
|
+
if (Boolean(opts.provider) !== Boolean(opts.model)) throw new Error("PollingApiTaskSource: provider and model must be set together");
|
|
12808
13145
|
this.listLimit = opts.listLimit ?? DEFAULT_LIST_LIMIT;
|
|
12809
13146
|
this.currentBackoffMs = this.minBackoffMs;
|
|
12810
13147
|
this.logger = (opts.logger ?? pino({ name: "polling-api-source" })).child({ teamId: opts.teamId });
|
|
@@ -12838,6 +13175,10 @@ var PollingApiTaskSource = class {
|
|
|
12838
13175
|
teamId: this.opts.teamId,
|
|
12839
13176
|
status: "queued",
|
|
12840
13177
|
...taskType ? { taskType } : {},
|
|
13178
|
+
...this.opts.provider && this.opts.model ? {
|
|
13179
|
+
provider: this.opts.provider,
|
|
13180
|
+
model: this.opts.model
|
|
13181
|
+
} : {},
|
|
12841
13182
|
limit: this.listLimit
|
|
12842
13183
|
});
|
|
12843
13184
|
if (this.opts.debug) this.logger.debug({
|
|
@@ -12848,6 +13189,10 @@ var PollingApiTaskSource = class {
|
|
|
12848
13189
|
for (const item of result.items) {
|
|
12849
13190
|
if (seen.has(item.id)) continue;
|
|
12850
13191
|
if (this.opts.diaryIds && this.opts.diaryIds.length > 0 && (item.diaryId === null || !this.opts.diaryIds.includes(item.diaryId))) continue;
|
|
13192
|
+
if (this.opts.provider && this.opts.model) {
|
|
13193
|
+
const allowed = item.allowedExecutors ?? [];
|
|
13194
|
+
if (allowed.length > 0 && !allowed.some((e) => e.provider === this.opts.provider && e.model === this.opts.model)) continue;
|
|
13195
|
+
}
|
|
12851
13196
|
if (item.status !== "queued") continue;
|
|
12852
13197
|
seen.add(item.id);
|
|
12853
13198
|
out.push(item);
|
|
@@ -13841,138 +14186,29 @@ function pruneOldSnapshots(maxCached, currentDir) {
|
|
|
13841
14186
|
});
|
|
13842
14187
|
}
|
|
13843
14188
|
//#endregion
|
|
13844
|
-
//#region ../../libs/pi-extension/src/
|
|
13845
|
-
/**
|
|
13846
|
-
* Gondolin tool operations: redirect pi's built-in tool operations
|
|
13847
|
-
* (read, write, edit, bash) to execute inside the VM.
|
|
13848
|
-
*
|
|
13849
|
-
* Follows the same pattern as upstream pi-gondolin.ts — pi's tool factories
|
|
13850
|
-
* accept an `operations` object that provides the underlying I/O.
|
|
13851
|
-
*/
|
|
14189
|
+
//#region ../../libs/pi-extension/src/vm-manager.ts
|
|
13852
14190
|
var GUEST_WORKSPACE$1 = "/workspace";
|
|
13853
|
-
function shQuote(s) {
|
|
13854
|
-
return "'" + s.replace(/'/g, "'\\''") + "'";
|
|
13855
|
-
}
|
|
13856
14191
|
/**
|
|
13857
|
-
*
|
|
13858
|
-
*
|
|
13859
|
-
|
|
13860
|
-
|
|
13861
|
-
|
|
13862
|
-
|
|
13863
|
-
|
|
13864
|
-
|
|
13865
|
-
|
|
13866
|
-
|
|
13867
|
-
|
|
13868
|
-
|
|
13869
|
-
|
|
13870
|
-
|
|
13871
|
-
|
|
13872
|
-
|
|
13873
|
-
|
|
13874
|
-
|
|
13875
|
-
|
|
13876
|
-
|
|
13877
|
-
"/bin/sh",
|
|
13878
|
-
"-lc",
|
|
13879
|
-
`test -r ${shQuote(toGuestPath(localCwd, p))}`
|
|
13880
|
-
])).ok) throw new Error(`not readable: ${p}`);
|
|
13881
|
-
},
|
|
13882
|
-
detectImageMimeType: async (p) => {
|
|
13883
|
-
try {
|
|
13884
|
-
const r = await vm.exec([
|
|
13885
|
-
"/bin/sh",
|
|
13886
|
-
"-lc",
|
|
13887
|
-
`file --mime-type -b ${shQuote(toGuestPath(localCwd, p))}`
|
|
13888
|
-
]);
|
|
13889
|
-
if (!r.ok) return null;
|
|
13890
|
-
const m = r.stdout.trim();
|
|
13891
|
-
return [
|
|
13892
|
-
"image/jpeg",
|
|
13893
|
-
"image/png",
|
|
13894
|
-
"image/gif",
|
|
13895
|
-
"image/webp"
|
|
13896
|
-
].includes(m) ? m : null;
|
|
13897
|
-
} catch {
|
|
13898
|
-
return null;
|
|
13899
|
-
}
|
|
13900
|
-
}
|
|
13901
|
-
};
|
|
13902
|
-
}
|
|
13903
|
-
function createGondolinWriteOps(vm, localCwd) {
|
|
13904
|
-
return {
|
|
13905
|
-
writeFile: async (p, content) => {
|
|
13906
|
-
const guestPath = toGuestPath(localCwd, p);
|
|
13907
|
-
const dir = path.posix.dirname(guestPath);
|
|
13908
|
-
const b64 = Buffer.from(content, "utf8").toString("base64");
|
|
13909
|
-
const r = await vm.exec([
|
|
13910
|
-
"/bin/sh",
|
|
13911
|
-
"-lc",
|
|
13912
|
-
[
|
|
13913
|
-
"set -eu",
|
|
13914
|
-
`mkdir -p ${shQuote(dir)}`,
|
|
13915
|
-
`echo ${shQuote(b64)} | base64 -d > ${shQuote(guestPath)}`
|
|
13916
|
-
].join("\n")
|
|
13917
|
-
]);
|
|
13918
|
-
if (!r.ok) throw new Error(`write failed (${r.exitCode}): ${r.stderr}`);
|
|
13919
|
-
},
|
|
13920
|
-
mkdir: async (dir) => {
|
|
13921
|
-
const r = await vm.exec([
|
|
13922
|
-
"/bin/mkdir",
|
|
13923
|
-
"-p",
|
|
13924
|
-
toGuestPath(localCwd, dir)
|
|
13925
|
-
]);
|
|
13926
|
-
if (!r.ok) throw new Error(`mkdir failed (${r.exitCode}): ${r.stderr}`);
|
|
13927
|
-
}
|
|
13928
|
-
};
|
|
13929
|
-
}
|
|
13930
|
-
function createGondolinEditOps(vm, localCwd) {
|
|
13931
|
-
const r = createGondolinReadOps(vm, localCwd);
|
|
13932
|
-
const w = createGondolinWriteOps(vm, localCwd);
|
|
13933
|
-
return {
|
|
13934
|
-
readFile: r.readFile,
|
|
13935
|
-
access: r.access,
|
|
13936
|
-
writeFile: w.writeFile
|
|
13937
|
-
};
|
|
13938
|
-
}
|
|
13939
|
-
function createGondolinBashOps(vm, localCwd) {
|
|
13940
|
-
return { exec: async (command, cwd, { onData, signal, timeout, env }) => {
|
|
13941
|
-
const guestCwd = toGuestPath(localCwd, cwd);
|
|
13942
|
-
const ac = new AbortController();
|
|
13943
|
-
const onAbort = () => ac.abort();
|
|
13944
|
-
signal?.addEventListener("abort", onAbort, { once: true });
|
|
13945
|
-
let timedOut = false;
|
|
13946
|
-
const timer = timeout && timeout > 0 ? setTimeout(() => {
|
|
13947
|
-
timedOut = true;
|
|
13948
|
-
ac.abort();
|
|
13949
|
-
}, timeout * 1e3) : void 0;
|
|
13950
|
-
try {
|
|
13951
|
-
const proc = vm.exec([
|
|
13952
|
-
"/bin/sh",
|
|
13953
|
-
"-lc",
|
|
13954
|
-
command
|
|
13955
|
-
], {
|
|
13956
|
-
cwd: guestCwd,
|
|
13957
|
-
signal: ac.signal,
|
|
13958
|
-
stdout: "pipe",
|
|
13959
|
-
stderr: "pipe"
|
|
13960
|
-
});
|
|
13961
|
-
for await (const chunk of proc.output()) onData(typeof chunk.data === "string" ? Buffer.from(chunk.data, "utf8") : chunk.data);
|
|
13962
|
-
return { exitCode: (await proc).exitCode };
|
|
13963
|
-
} catch (err) {
|
|
13964
|
-
if (signal?.aborted) throw new Error("aborted");
|
|
13965
|
-
if (timedOut) throw new Error(`timeout:${timeout}`);
|
|
13966
|
-
throw err;
|
|
13967
|
-
} finally {
|
|
13968
|
-
if (timer) clearTimeout(timer);
|
|
13969
|
-
signal?.removeEventListener("abort", onAbort);
|
|
13970
|
-
}
|
|
13971
|
-
} };
|
|
13972
|
-
}
|
|
13973
|
-
//#endregion
|
|
13974
|
-
//#region ../../libs/pi-extension/src/vm-manager.ts
|
|
13975
|
-
var GUEST_WORKSPACE = "/workspace";
|
|
14192
|
+
* Memory-backed VFS mount used by the daemon to inject task-context
|
|
14193
|
+
* skills (#943 slice 1.5). Sibling of /workspace, NOT a sub-path —
|
|
14194
|
+
* Gondolin mounts can't nest. The agent's Gondolin-bound Read tool
|
|
14195
|
+
* accepts paths under this prefix (see toGuestPath in tool-operations.ts).
|
|
14196
|
+
*
|
|
14197
|
+
* Why MemoryProvider rather than a path under /workspace:
|
|
14198
|
+
* - Injected skills are ephemeral by intent: per-task-attempt input
|
|
14199
|
+
* scoped to the VM lifetime. MemoryProvider models that exactly —
|
|
14200
|
+
* in-memory, per-VM-instance, zero host artefacts, automatic
|
|
14201
|
+
* cleanup on VM close.
|
|
14202
|
+
* - Writing under /workspace fails in worktrees because we symlink
|
|
14203
|
+
* `.moltnet/` to the main repo (so credentials are reachable from
|
|
14204
|
+
* worktrees), and Gondolin's RealFSProvider correctly refuses to
|
|
14205
|
+
* create paths whose ancestors' realpath escapes the mount root.
|
|
14206
|
+
* That refusal is a deliberate sandbox-escape protection, not a
|
|
14207
|
+
* bug. See diary semantic entry cd27d9d3-efdc-4aec-ac0d-5fd8ce258d1f
|
|
14208
|
+
* and episodic 7affbfeb-18a2-4963-aeac-c177eb2afa2d for the full
|
|
14209
|
+
* investigation and the alternatives we rejected.
|
|
14210
|
+
*/
|
|
14211
|
+
var GUEST_TASK_SKILLS_MOUNT = "/moltnet-task-skills";
|
|
13976
14212
|
/**
|
|
13977
14213
|
* Resolve the main worktree root (where .moltnet/ lives — it's untracked,
|
|
13978
14214
|
* only exists in the main worktree, not in git worktrees).
|
|
@@ -14101,7 +14337,10 @@ async function resumeVm(config) {
|
|
|
14101
14337
|
env: vmEnv,
|
|
14102
14338
|
...resources?.memory && { memory: resources.memory },
|
|
14103
14339
|
...resources?.cpus && { cpus: resources.cpus },
|
|
14104
|
-
vfs: { mounts: {
|
|
14340
|
+
vfs: { mounts: {
|
|
14341
|
+
[GUEST_WORKSPACE$1]: workspaceProvider,
|
|
14342
|
+
[GUEST_TASK_SKILLS_MOUNT]: new MemoryProvider()
|
|
14343
|
+
} }
|
|
14105
14344
|
});
|
|
14106
14345
|
await vm.exec(`sh -c '
|
|
14107
14346
|
cp /etc/gondolin/mitm/ca.crt /usr/local/share/ca-certificates/gondolin-mitm.crt
|
|
@@ -14131,7 +14370,7 @@ nameserver 1.1.1.1" > /etc/resolv.conf'`);
|
|
|
14131
14370
|
vm,
|
|
14132
14371
|
credentials: creds,
|
|
14133
14372
|
mountPath: config.mountPath,
|
|
14134
|
-
guestWorkspace: GUEST_WORKSPACE,
|
|
14373
|
+
guestWorkspace: GUEST_WORKSPACE$1,
|
|
14135
14374
|
agentDir
|
|
14136
14375
|
};
|
|
14137
14376
|
}
|
|
@@ -14184,6 +14423,137 @@ function ensureRelativeWorktreePaths(gitconfig) {
|
|
|
14184
14423
|
return `${gitconfig}${gitconfig.endsWith("\n") ? "" : "\n"}[worktree]\n\tuseRelativePaths = true\n`;
|
|
14185
14424
|
}
|
|
14186
14425
|
//#endregion
|
|
14426
|
+
//#region ../../libs/pi-extension/src/tool-operations.ts
|
|
14427
|
+
/**
|
|
14428
|
+
* Gondolin tool operations: redirect pi's built-in tool operations
|
|
14429
|
+
* (read, write, edit, bash) to execute inside the VM.
|
|
14430
|
+
*
|
|
14431
|
+
* Follows the same pattern as upstream pi-gondolin.ts — pi's tool factories
|
|
14432
|
+
* accept an `operations` object that provides the underlying I/O.
|
|
14433
|
+
*/
|
|
14434
|
+
var GUEST_WORKSPACE = "/workspace";
|
|
14435
|
+
function shQuote(s) {
|
|
14436
|
+
return "'" + s.replace(/'/g, "'\\''") + "'";
|
|
14437
|
+
}
|
|
14438
|
+
/**
|
|
14439
|
+
* Map a host-side absolute path to a guest-side /workspace path.
|
|
14440
|
+
* Throws if the path escapes the workspace.
|
|
14441
|
+
*/
|
|
14442
|
+
function toGuestPath(localCwd, localPath) {
|
|
14443
|
+
if (localPath === GUEST_WORKSPACE || localPath.startsWith(`${GUEST_WORKSPACE}/`)) return localPath;
|
|
14444
|
+
if (localPath === "/moltnet-task-skills" || localPath.startsWith(`/moltnet-task-skills/`)) return localPath;
|
|
14445
|
+
const rel = path.relative(localCwd, localPath);
|
|
14446
|
+
if (rel === "") return GUEST_WORKSPACE;
|
|
14447
|
+
if (rel.startsWith("..") || path.isAbsolute(rel)) throw new Error(`path escapes workspace: ${localPath}`);
|
|
14448
|
+
const posixRel = rel.split(path.sep).join(path.posix.sep);
|
|
14449
|
+
return path.posix.join(GUEST_WORKSPACE, posixRel);
|
|
14450
|
+
}
|
|
14451
|
+
function createGondolinReadOps(vm, localCwd) {
|
|
14452
|
+
return {
|
|
14453
|
+
readFile: async (p) => {
|
|
14454
|
+
const r = await vm.exec(["/bin/cat", toGuestPath(localCwd, p)]);
|
|
14455
|
+
if (!r.ok) throw new Error(`cat failed (${r.exitCode}): ${r.stderr}`);
|
|
14456
|
+
return r.stdoutBuffer;
|
|
14457
|
+
},
|
|
14458
|
+
access: async (p) => {
|
|
14459
|
+
if (!(await vm.exec([
|
|
14460
|
+
"/bin/sh",
|
|
14461
|
+
"-lc",
|
|
14462
|
+
`test -r ${shQuote(toGuestPath(localCwd, p))}`
|
|
14463
|
+
])).ok) throw new Error(`not readable: ${p}`);
|
|
14464
|
+
},
|
|
14465
|
+
detectImageMimeType: async (p) => {
|
|
14466
|
+
try {
|
|
14467
|
+
const r = await vm.exec([
|
|
14468
|
+
"/bin/sh",
|
|
14469
|
+
"-lc",
|
|
14470
|
+
`file --mime-type -b ${shQuote(toGuestPath(localCwd, p))}`
|
|
14471
|
+
]);
|
|
14472
|
+
if (!r.ok) return null;
|
|
14473
|
+
const m = r.stdout.trim();
|
|
14474
|
+
return [
|
|
14475
|
+
"image/jpeg",
|
|
14476
|
+
"image/png",
|
|
14477
|
+
"image/gif",
|
|
14478
|
+
"image/webp"
|
|
14479
|
+
].includes(m) ? m : null;
|
|
14480
|
+
} catch {
|
|
14481
|
+
return null;
|
|
14482
|
+
}
|
|
14483
|
+
}
|
|
14484
|
+
};
|
|
14485
|
+
}
|
|
14486
|
+
function createGondolinWriteOps(vm, localCwd) {
|
|
14487
|
+
return {
|
|
14488
|
+
writeFile: async (p, content) => {
|
|
14489
|
+
const guestPath = toGuestPath(localCwd, p);
|
|
14490
|
+
const dir = path.posix.dirname(guestPath);
|
|
14491
|
+
const b64 = Buffer.from(content, "utf8").toString("base64");
|
|
14492
|
+
const r = await vm.exec([
|
|
14493
|
+
"/bin/sh",
|
|
14494
|
+
"-lc",
|
|
14495
|
+
[
|
|
14496
|
+
"set -eu",
|
|
14497
|
+
`mkdir -p ${shQuote(dir)}`,
|
|
14498
|
+
`echo ${shQuote(b64)} | base64 -d > ${shQuote(guestPath)}`
|
|
14499
|
+
].join("\n")
|
|
14500
|
+
]);
|
|
14501
|
+
if (!r.ok) throw new Error(`write failed (${r.exitCode}): ${r.stderr}`);
|
|
14502
|
+
},
|
|
14503
|
+
mkdir: async (dir) => {
|
|
14504
|
+
const r = await vm.exec([
|
|
14505
|
+
"/bin/mkdir",
|
|
14506
|
+
"-p",
|
|
14507
|
+
toGuestPath(localCwd, dir)
|
|
14508
|
+
]);
|
|
14509
|
+
if (!r.ok) throw new Error(`mkdir failed (${r.exitCode}): ${r.stderr}`);
|
|
14510
|
+
}
|
|
14511
|
+
};
|
|
14512
|
+
}
|
|
14513
|
+
function createGondolinEditOps(vm, localCwd) {
|
|
14514
|
+
const r = createGondolinReadOps(vm, localCwd);
|
|
14515
|
+
const w = createGondolinWriteOps(vm, localCwd);
|
|
14516
|
+
return {
|
|
14517
|
+
readFile: r.readFile,
|
|
14518
|
+
access: r.access,
|
|
14519
|
+
writeFile: w.writeFile
|
|
14520
|
+
};
|
|
14521
|
+
}
|
|
14522
|
+
function createGondolinBashOps(vm, localCwd) {
|
|
14523
|
+
return { exec: async (command, cwd, { onData, signal, timeout, env }) => {
|
|
14524
|
+
const guestCwd = toGuestPath(localCwd, cwd);
|
|
14525
|
+
const ac = new AbortController();
|
|
14526
|
+
const onAbort = () => ac.abort();
|
|
14527
|
+
signal?.addEventListener("abort", onAbort, { once: true });
|
|
14528
|
+
let timedOut = false;
|
|
14529
|
+
const timer = timeout && timeout > 0 ? setTimeout(() => {
|
|
14530
|
+
timedOut = true;
|
|
14531
|
+
ac.abort();
|
|
14532
|
+
}, timeout * 1e3) : void 0;
|
|
14533
|
+
try {
|
|
14534
|
+
const proc = vm.exec([
|
|
14535
|
+
"/bin/sh",
|
|
14536
|
+
"-lc",
|
|
14537
|
+
command
|
|
14538
|
+
], {
|
|
14539
|
+
cwd: guestCwd,
|
|
14540
|
+
signal: ac.signal,
|
|
14541
|
+
stdout: "pipe",
|
|
14542
|
+
stderr: "pipe"
|
|
14543
|
+
});
|
|
14544
|
+
for await (const chunk of proc.output()) onData(typeof chunk.data === "string" ? Buffer.from(chunk.data, "utf8") : chunk.data);
|
|
14545
|
+
return { exitCode: (await proc).exitCode };
|
|
14546
|
+
} catch (err) {
|
|
14547
|
+
if (signal?.aborted) throw new Error("aborted");
|
|
14548
|
+
if (timedOut) throw new Error(`timeout:${timeout}`);
|
|
14549
|
+
throw err;
|
|
14550
|
+
} finally {
|
|
14551
|
+
if (timer) clearTimeout(timer);
|
|
14552
|
+
signal?.removeEventListener("abort", onAbort);
|
|
14553
|
+
}
|
|
14554
|
+
} };
|
|
14555
|
+
}
|
|
14556
|
+
//#endregion
|
|
14187
14557
|
//#region ../../libs/pi-extension/src/otel/index.ts
|
|
14188
14558
|
var TRACER_NAME = "@themoltnet/pi-extension/otel";
|
|
14189
14559
|
function stripReservedAttrs(attrs) {
|
|
@@ -14321,6 +14691,114 @@ function extractUsage(message) {
|
|
|
14321
14691
|
};
|
|
14322
14692
|
}
|
|
14323
14693
|
//#endregion
|
|
14694
|
+
//#region ../../libs/pi-extension/src/runtime/inject-task-context.ts
|
|
14695
|
+
/**
|
|
14696
|
+
* Slice 1.5 of #943 — wire the agent-runtime resolver into the
|
|
14697
|
+
* pi-extension execution path.
|
|
14698
|
+
*
|
|
14699
|
+
* `resolveTaskContext` is a pure dispatcher; this module provides the
|
|
14700
|
+
* Gondolin-aware deliverer and the post-resolution shape the
|
|
14701
|
+
* `execute-pi-task` caller needs to splice into pi's setup:
|
|
14702
|
+
*
|
|
14703
|
+
* - `systemPromptPrefix` → fed into `appendSystemPrompt` alongside
|
|
14704
|
+
* the runtime instructor (it IS a system-prompt fragment).
|
|
14705
|
+
* - `userInlineSuffix` → appended to the `buildTaskUserPrompt`
|
|
14706
|
+
* output BEFORE `session.prompt(text)`.
|
|
14707
|
+
* - `skills` → spliced into the `skillsOverride` callback's
|
|
14708
|
+
* return value. pi includes them in `<available_skills>` in the
|
|
14709
|
+
* system prompt; the agent fetches the body on demand via the
|
|
14710
|
+
* Read tool.
|
|
14711
|
+
*
|
|
14712
|
+
* Skill files are written into the VM at
|
|
14713
|
+
* `/workspace/.moltnet/skills/<slug>/SKILL.md`. The agent's
|
|
14714
|
+
* Gondolin-bound Read tool is scoped to `/workspace`, so that path is
|
|
14715
|
+
* the only location the agent can actually read at runtime. pi only
|
|
14716
|
+
* reads `<available_skills>` metadata (name, description, location),
|
|
14717
|
+
* never the file body, so we construct synthetic `Skill` objects
|
|
14718
|
+
* pointing at the in-VM path without ever materialising the file on
|
|
14719
|
+
* the host.
|
|
14720
|
+
*/
|
|
14721
|
+
/**
|
|
14722
|
+
* Where in the VM we write skill bodies — the memory-backed mount
|
|
14723
|
+
* declared in `vm-manager.ts`. See the comment on
|
|
14724
|
+
* `GUEST_TASK_SKILLS_MOUNT` there for the full rationale (ephemeral
|
|
14725
|
+
* by intent + the worktree symlink interaction with Gondolin's
|
|
14726
|
+
* sandbox-escape protection). The agent's Gondolin Read tool accepts
|
|
14727
|
+
* paths under this mount via `toGuestPath` in `tool-operations.ts`.
|
|
14728
|
+
*/
|
|
14729
|
+
var SKILL_ROOT_IN_VM = GUEST_TASK_SKILLS_MOUNT;
|
|
14730
|
+
/** Bounds borrowed from pi's skill validation; conservative caps so a
|
|
14731
|
+
* malformed SKILL.md doesn't bloat the system prompt. */
|
|
14732
|
+
var MAX_SKILL_NAME = 64;
|
|
14733
|
+
var MAX_SKILL_DESCRIPTION = 1024;
|
|
14734
|
+
/**
|
|
14735
|
+
* Resolve a task's `input.context[]` and inject the side effects pi
|
|
14736
|
+
* needs. Safe to call with an empty array — returns an inert result.
|
|
14737
|
+
*/
|
|
14738
|
+
async function injectTaskContext(args) {
|
|
14739
|
+
const skills = [];
|
|
14740
|
+
const resolved = await resolveTaskContext({
|
|
14741
|
+
context: args.context,
|
|
14742
|
+
deliver: { skill: async ({ slug, content }) => {
|
|
14743
|
+
const dir = `${SKILL_ROOT_IN_VM}/${slug}`;
|
|
14744
|
+
const filePath = `${dir}/SKILL.md`;
|
|
14745
|
+
await args.fs.mkdir(dir, { recursive: true });
|
|
14746
|
+
await args.fs.writeFile(filePath, content, { mode: 420 });
|
|
14747
|
+
skills.push(buildSyntheticSkill({
|
|
14748
|
+
slug,
|
|
14749
|
+
content,
|
|
14750
|
+
filePath,
|
|
14751
|
+
dir
|
|
14752
|
+
}));
|
|
14753
|
+
} }
|
|
14754
|
+
});
|
|
14755
|
+
return {
|
|
14756
|
+
injected: resolved.injected,
|
|
14757
|
+
skills,
|
|
14758
|
+
systemPromptPrefix: resolved.systemPromptPrefix,
|
|
14759
|
+
userInlineSuffix: resolved.userInlineSuffix
|
|
14760
|
+
};
|
|
14761
|
+
}
|
|
14762
|
+
/**
|
|
14763
|
+
* Build a `Skill` object pi will faithfully render in
|
|
14764
|
+
* `<available_skills>`. We extract `name` and `description` from the
|
|
14765
|
+
* skill content's YAML frontmatter using pi's own `parseFrontmatter`
|
|
14766
|
+
* helper (proper YAML, not a regex hack) and fall back to the slug +
|
|
14767
|
+
* a generic description so a SKILL.md without frontmatter still
|
|
14768
|
+
* renders something meaningful.
|
|
14769
|
+
*
|
|
14770
|
+
* Frontmatter parsing is best-effort: a malformed YAML block is
|
|
14771
|
+
* optional metadata, not a reason to fail the task. We swallow parser
|
|
14772
|
+
* errors and fall back to the slug-derived metadata; the skill body
|
|
14773
|
+
* is unaffected.
|
|
14774
|
+
*
|
|
14775
|
+
* pi's `formatSkillsForPrompt` only reads `name`, `description`, and
|
|
14776
|
+
* `filePath` — `sourceInfo`/`baseDir` exist on the type but never
|
|
14777
|
+
* surface in the prompt, so a synthetic `SourceInfo` is enough.
|
|
14778
|
+
*/
|
|
14779
|
+
function buildSyntheticSkill(args) {
|
|
14780
|
+
let fm = {};
|
|
14781
|
+
try {
|
|
14782
|
+
fm = parseFrontmatter(args.content).frontmatter;
|
|
14783
|
+
} catch {}
|
|
14784
|
+
return {
|
|
14785
|
+
name: clip(typeof fm.name === "string" && fm.name.trim().length > 0 ? fm.name.trim() : args.slug, MAX_SKILL_NAME),
|
|
14786
|
+
description: clip(typeof fm.description === "string" && fm.description.trim().length > 0 ? fm.description.trim() : `Task-injected context skill (${args.slug})`, MAX_SKILL_DESCRIPTION),
|
|
14787
|
+
filePath: args.filePath,
|
|
14788
|
+
baseDir: args.dir,
|
|
14789
|
+
sourceInfo: createSyntheticSourceInfo(args.filePath, {
|
|
14790
|
+
source: "moltnet:task-context",
|
|
14791
|
+
scope: "temporary",
|
|
14792
|
+
origin: "top-level",
|
|
14793
|
+
baseDir: args.dir
|
|
14794
|
+
}),
|
|
14795
|
+
disableModelInvocation: fm["disable-model-invocation"] === true
|
|
14796
|
+
};
|
|
14797
|
+
}
|
|
14798
|
+
function clip(s, max) {
|
|
14799
|
+
return s.length > max ? s.slice(0, max) : s;
|
|
14800
|
+
}
|
|
14801
|
+
//#endregion
|
|
14324
14802
|
//#region ../../libs/pi-extension/src/runtime/runtime-instructor.ts
|
|
14325
14803
|
/**
|
|
14326
14804
|
* Build the daemon-controlled invariant prose injected into the system prompt
|
|
@@ -14644,6 +15122,7 @@ function resolveSubmitTools(taskType, opts = {}) {
|
|
|
14644
15122
|
* Anthropic-SDK one) plug in via the `executeTask` function injected into
|
|
14645
15123
|
* `AgentRuntime`.
|
|
14646
15124
|
*/
|
|
15125
|
+
var noopTurnEventHandler = () => {};
|
|
14647
15126
|
/**
|
|
14648
15127
|
* Factory that builds a pi-specific `executeTask` function suitable for
|
|
14649
15128
|
* injection into `AgentRuntime`. The returned function caches the resolved
|
|
@@ -14740,10 +15219,25 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
14740
15219
|
attemptN
|
|
14741
15220
|
});
|
|
14742
15221
|
reporterOpen = true;
|
|
14743
|
-
|
|
14744
|
-
|
|
14745
|
-
|
|
14746
|
-
})
|
|
15222
|
+
let onTurnEvent;
|
|
15223
|
+
if (opts.makeOnTurnEvent) try {
|
|
15224
|
+
onTurnEvent = opts.makeOnTurnEvent(claimedTask);
|
|
15225
|
+
} catch (err) {
|
|
15226
|
+
process.stderr.write(`[emit] makeOnTurnEvent threw: ${err instanceof Error ? err.message : String(err)}\n`);
|
|
15227
|
+
onTurnEvent = noopTurnEventHandler;
|
|
15228
|
+
}
|
|
15229
|
+
else onTurnEvent = opts.onTurnEvent ?? noopTurnEventHandler;
|
|
15230
|
+
const emit = (kind, payload) => {
|
|
15231
|
+
try {
|
|
15232
|
+
onTurnEvent(kind, summarizePayloadForLog(kind, payload));
|
|
15233
|
+
} catch (err) {
|
|
15234
|
+
process.stderr.write(`[emit] onTurnEvent threw for kind="${kind}": ${err instanceof Error ? err.message : String(err)}\n`);
|
|
15235
|
+
}
|
|
15236
|
+
return reporter.record({
|
|
15237
|
+
kind,
|
|
15238
|
+
payload
|
|
15239
|
+
});
|
|
15240
|
+
};
|
|
14747
15241
|
await emit("info", {
|
|
14748
15242
|
event: "execute_start",
|
|
14749
15243
|
taskType: task.taskType,
|
|
@@ -14753,7 +15247,7 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
14753
15247
|
});
|
|
14754
15248
|
let taskPrompt;
|
|
14755
15249
|
try {
|
|
14756
|
-
taskPrompt =
|
|
15250
|
+
taskPrompt = buildTaskUserPrompt(task, {
|
|
14757
15251
|
diaryId,
|
|
14758
15252
|
taskId: task.id,
|
|
14759
15253
|
extras: opts.promptExtras
|
|
@@ -14766,6 +15260,30 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
14766
15260
|
});
|
|
14767
15261
|
return makeFailedOutput("prompt_build_failed", message);
|
|
14768
15262
|
}
|
|
15263
|
+
const rawContext = task.input.context;
|
|
15264
|
+
let injectedContext;
|
|
15265
|
+
try {
|
|
15266
|
+
const contextArray = rawContext === void 0 ? [] : rawContext;
|
|
15267
|
+
if (!Check(TaskContext, contextArray)) throw new Error(`task.input.context failed TaskContext validation: ${JSON.stringify([...Errors(TaskContext, contextArray)].slice(0, 3))}`);
|
|
15268
|
+
injectedContext = await injectTaskContext({
|
|
15269
|
+
context: contextArray,
|
|
15270
|
+
fs: managed.vm.fs
|
|
15271
|
+
});
|
|
15272
|
+
} catch (err) {
|
|
15273
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
15274
|
+
await emit("error", {
|
|
15275
|
+
message,
|
|
15276
|
+
phase: "context_resolution"
|
|
15277
|
+
});
|
|
15278
|
+
return makeFailedOutput("context_resolution_failed", message);
|
|
15279
|
+
}
|
|
15280
|
+
if (injectedContext.injected.length > 0) await emit("info", {
|
|
15281
|
+
event: "context_injected",
|
|
15282
|
+
count: injectedContext.injected.length,
|
|
15283
|
+
bindings: injectedContext.injected.map((r) => r.binding),
|
|
15284
|
+
slugs: injectedContext.injected.map((r) => r.slug)
|
|
15285
|
+
});
|
|
15286
|
+
if (injectedContext.userInlineSuffix) taskPrompt = `${taskPrompt}\n\n---\n\n${injectedContext.userInlineSuffix}`;
|
|
14769
15287
|
const gondolinCustomTools = [
|
|
14770
15288
|
createReadToolDefinition(mountPath, { operations: createGondolinReadOps(managed.vm, mountPath) }),
|
|
14771
15289
|
createWriteToolDefinition(mountPath, { operations: createGondolinWriteOps(managed.vm, mountPath) }),
|
|
@@ -14802,21 +15320,23 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
14802
15320
|
"moltnet.task.type": task.taskType
|
|
14803
15321
|
}
|
|
14804
15322
|
});
|
|
14805
|
-
const
|
|
15323
|
+
const appendSystemPrompt = [buildRuntimeInstructor({
|
|
14806
15324
|
taskId: task.id,
|
|
14807
15325
|
taskType: task.taskType,
|
|
14808
15326
|
attemptN,
|
|
14809
15327
|
diaryId,
|
|
14810
15328
|
agentName: opts.agentName,
|
|
14811
15329
|
correlationId: task.correlationId ?? null
|
|
14812
|
-
});
|
|
15330
|
+
})];
|
|
15331
|
+
if (injectedContext.systemPromptPrefix) appendSystemPrompt.push(injectedContext.systemPromptPrefix);
|
|
15332
|
+
const injectedSkills = injectedContext.skills;
|
|
14813
15333
|
const resourceLoader = new DefaultResourceLoader({
|
|
14814
15334
|
cwd: mountPath,
|
|
14815
15335
|
agentDir: piAuthDir,
|
|
14816
15336
|
extensionFactories: [piOtelExtension],
|
|
14817
|
-
appendSystemPrompt
|
|
15337
|
+
appendSystemPrompt,
|
|
14818
15338
|
skillsOverride: () => ({
|
|
14819
|
-
skills:
|
|
15339
|
+
skills: injectedSkills,
|
|
14820
15340
|
diagnostics: []
|
|
14821
15341
|
})
|
|
14822
15342
|
});
|
|
@@ -15041,6 +15561,27 @@ function wireSessionAbort(cancelSignal, session) {
|
|
|
15041
15561
|
* `task_messages.payload` row. Bodies above 4 KiB are replaced with a
|
|
15042
15562
|
* `{ truncated, original_size }` marker so the JSONL/DB size stays bounded.
|
|
15043
15563
|
*/
|
|
15564
|
+
function summarizePayloadForLog(kind, payload) {
|
|
15565
|
+
switch (kind) {
|
|
15566
|
+
case "text_delta": {
|
|
15567
|
+
const delta = payload.delta;
|
|
15568
|
+
return { chars: typeof delta === "string" ? delta.length : 0 };
|
|
15569
|
+
}
|
|
15570
|
+
case "tool_call_start": return { tool: payload.tool_name };
|
|
15571
|
+
case "tool_call_end": return {
|
|
15572
|
+
tool: payload.tool_name,
|
|
15573
|
+
is_error: payload.is_error === true,
|
|
15574
|
+
...payload.is_error === true && payload.result !== void 0 ? { result: payload.result } : {}
|
|
15575
|
+
};
|
|
15576
|
+
case "turn_end": return { stop_reason: payload.stop_reason };
|
|
15577
|
+
case "error": return {
|
|
15578
|
+
phase: payload.phase,
|
|
15579
|
+
message: typeof payload.message === "string" ? payload.message.slice(0, TRUNCATE_LIMIT) : payload.message
|
|
15580
|
+
};
|
|
15581
|
+
case "info": return Object.fromEntries(Object.entries(payload).map(([k, v]) => [k, typeof v === "string" ? v.slice(0, TRUNCATE_LIMIT) : v]));
|
|
15582
|
+
default: return payload;
|
|
15583
|
+
}
|
|
15584
|
+
}
|
|
15044
15585
|
var TRUNCATE_LIMIT = 4 * 1024;
|
|
15045
15586
|
function truncateForWire(value) {
|
|
15046
15587
|
if (value === null || value === void 0) return value;
|
|
@@ -15340,6 +15881,27 @@ function findUp(startDir, filename) {
|
|
|
15340
15881
|
}
|
|
15341
15882
|
}
|
|
15342
15883
|
//#endregion
|
|
15884
|
+
//#region src/lib/turn-event-logger.ts
|
|
15885
|
+
function makeTurnEventHandler(base, context = {}) {
|
|
15886
|
+
const log = base.child({
|
|
15887
|
+
name: "agent-daemon.turn",
|
|
15888
|
+
...context
|
|
15889
|
+
});
|
|
15890
|
+
return (event, summary) => {
|
|
15891
|
+
if (event === "text_delta") return;
|
|
15892
|
+
log[event === "error" ? "warn" : event === "turn_end" ? "info" : "debug"]({
|
|
15893
|
+
event,
|
|
15894
|
+
...summary
|
|
15895
|
+
}, `turn.${event}`);
|
|
15896
|
+
};
|
|
15897
|
+
}
|
|
15898
|
+
function makeTurnEventHandlerFactory(base) {
|
|
15899
|
+
return (claimedTask) => makeTurnEventHandler(base, {
|
|
15900
|
+
taskId: claimedTask.task.id,
|
|
15901
|
+
attemptN: claimedTask.attemptN
|
|
15902
|
+
});
|
|
15903
|
+
}
|
|
15904
|
+
//#endregion
|
|
15343
15905
|
//#region src/cli/poll-shared.ts
|
|
15344
15906
|
async function runPolling(opts) {
|
|
15345
15907
|
if (isHelpFlag(opts.argv)) {
|
|
@@ -15442,7 +16004,8 @@ async function runPolling(opts) {
|
|
|
15442
16004
|
mountPath: sandbox.rootDir,
|
|
15443
16005
|
provider: common.provider,
|
|
15444
16006
|
model: common.model,
|
|
15445
|
-
sandboxConfig: sandbox.config
|
|
16007
|
+
sandboxConfig: sandbox.config,
|
|
16008
|
+
makeOnTurnEvent: makeTurnEventHandlerFactory(rootLogger)
|
|
15446
16009
|
});
|
|
15447
16010
|
runtime = new AgentRuntime({
|
|
15448
16011
|
logger: rootLogger,
|
|
@@ -15450,6 +16013,8 @@ async function runPolling(opts) {
|
|
|
15450
16013
|
agent: ctx.agent,
|
|
15451
16014
|
teamId,
|
|
15452
16015
|
taskTypes: taskTypes.length > 0 ? taskTypes : void 0,
|
|
16016
|
+
provider: common.provider.toLowerCase(),
|
|
16017
|
+
model: common.model.toLowerCase(),
|
|
15453
16018
|
diaryIds: diaryIds.length > 0 ? diaryIds : void 0,
|
|
15454
16019
|
leaseTtlSec: common.leaseTtlSec,
|
|
15455
16020
|
listLimit,
|
|
@@ -15607,19 +16172,40 @@ async function runOnce(argv) {
|
|
|
15607
16172
|
sandbox: sandbox.path,
|
|
15608
16173
|
taskId
|
|
15609
16174
|
}, "agent-daemon.starting");
|
|
16175
|
+
let runtime = null;
|
|
16176
|
+
const onSignal = (sig) => {
|
|
16177
|
+
rootLogger.warn({
|
|
16178
|
+
signal: sig,
|
|
16179
|
+
taskId
|
|
16180
|
+
}, "agent-daemon.draining");
|
|
16181
|
+
runtime?.stop();
|
|
16182
|
+
ctx.agent.tasks.cancel(taskId, { reason: `runner_${sig.toLowerCase()}` }).catch((err) => {
|
|
16183
|
+
rootLogger.warn({
|
|
16184
|
+
err: err instanceof Error ? err.message : String(err),
|
|
16185
|
+
taskId
|
|
16186
|
+
}, "agent-daemon.cancel_on_signal_failed");
|
|
16187
|
+
});
|
|
16188
|
+
};
|
|
16189
|
+
process.on("SIGINT", () => {
|
|
16190
|
+
onSignal("SIGINT");
|
|
16191
|
+
});
|
|
16192
|
+
process.on("SIGTERM", () => {
|
|
16193
|
+
onSignal("SIGTERM");
|
|
16194
|
+
});
|
|
15610
16195
|
try {
|
|
15611
16196
|
const executeTask = createPiTaskExecutor({
|
|
15612
16197
|
agentName: opts.agent,
|
|
15613
16198
|
mountPath: sandbox.rootDir,
|
|
15614
16199
|
provider: opts.provider,
|
|
15615
16200
|
model: opts.model,
|
|
15616
|
-
sandboxConfig: sandbox.config
|
|
16201
|
+
sandboxConfig: sandbox.config,
|
|
16202
|
+
onTurnEvent: makeTurnEventHandler(rootLogger, { taskId })
|
|
15617
16203
|
});
|
|
15618
16204
|
const writeCorrelationAnchors = makePrBodyAnchorWriter({
|
|
15619
16205
|
gh: createGhCliClient(),
|
|
15620
16206
|
logger: rootLogger
|
|
15621
16207
|
});
|
|
15622
|
-
|
|
16208
|
+
runtime = new AgentRuntime({
|
|
15623
16209
|
logger: rootLogger,
|
|
15624
16210
|
source: new ApiTaskSource({
|
|
15625
16211
|
agent: ctx.agent,
|
|
@@ -15639,7 +16225,8 @@ async function runOnce(argv) {
|
|
|
15639
16225
|
log: (msg, err) => rootLogger.warn({ err }, msg)
|
|
15640
16226
|
}),
|
|
15641
16227
|
executeTask
|
|
15642
|
-
})
|
|
16228
|
+
});
|
|
16229
|
+
const [output] = await runtime.start();
|
|
15643
16230
|
if (!output) {
|
|
15644
16231
|
rootLogger.error({}, "agent-daemon.no_output");
|
|
15645
16232
|
return 1;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@themoltnet/agent-daemon",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.5.0",
|
|
4
4
|
"license": "AGPL-3.0-only",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"description": "MoltNet agent daemon — claims and executes tasks (fulfill_brief, assess_brief) from the MoltNet task-service via Pi-headless. CLI: moltnet-agent.",
|
|
@@ -33,19 +33,19 @@
|
|
|
33
33
|
"@opentelemetry/semantic-conventions": "^1.39.0",
|
|
34
34
|
"pino": "^10.3.1",
|
|
35
35
|
"pino-pretty": "^13.1.3",
|
|
36
|
-
"@themoltnet/
|
|
37
|
-
"@themoltnet/
|
|
38
|
-
"@themoltnet/sdk": "0.
|
|
36
|
+
"@themoltnet/agent-runtime": "0.12.0",
|
|
37
|
+
"@themoltnet/pi-extension": "0.14.0",
|
|
38
|
+
"@themoltnet/sdk": "0.100.0"
|
|
39
39
|
},
|
|
40
40
|
"devDependencies": {
|
|
41
41
|
"tsx": "^4.7.0",
|
|
42
42
|
"typescript": "^5.3.3",
|
|
43
43
|
"vite": "^8.0.0",
|
|
44
44
|
"vitest": "^3.0.0",
|
|
45
|
+
"@moltnet/bootstrap": "0.1.0",
|
|
45
46
|
"@moltnet/crypto-service": "0.1.0",
|
|
46
|
-
"@moltnet/database": "0.1.0",
|
|
47
47
|
"@moltnet/tasks": "0.1.0",
|
|
48
|
-
"@moltnet/
|
|
48
|
+
"@moltnet/database": "0.1.0"
|
|
49
49
|
},
|
|
50
50
|
"nx": {
|
|
51
51
|
"tags": [
|