@themoltnet/agent-daemon 0.5.0 → 0.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/README.md +1 -1
  2. package/dist/main.js +847 -36
  3. package/package.json +7 -7
package/README.md CHANGED
@@ -44,7 +44,7 @@ All config flows from environment variables. The daemon reads them in
44
44
 
45
45
  The agent's `moltnet.json` and gitconfig live next to each other in
46
46
  `.moltnet/<agent>/`. Provision them once via
47
- [`legreffier init`](../../docs/getting-started.md).
47
+ [`legreffier init`](../../docs/start/install-and-initialize.md).
48
48
 
49
49
  ### Pi provider auth
50
50
 
package/dist/main.js CHANGED
@@ -3088,7 +3088,7 @@ unchanged" is.
3088
3088
  * (server-side schema check). Self-assessment is a truthful self-rating,
3089
3089
  * NOT enforcement — `verification.passed=false` does not block /complete
3090
3090
  * and does not affect `acceptedAttemptN`. See
3091
- * `docs/agent-runtime.md` for the full producer/judge flow.
3091
+ * `docs/understand/agent-runtime.md` for the full producer/judge flow.
3092
3092
  *
3093
3093
  * **Binding evaluation** (judgment tasks: `assess_brief`, `judge_pack`).
3094
3094
  * A separate task whose IS the application of `successCriteria` to
@@ -4086,6 +4086,39 @@ var AssessBriefOutput = Type$2.Object({
4086
4086
  $id: "AssessBriefOutput",
4087
4087
  additionalProperties: false
4088
4088
  });
4089
+ /**
4090
+ * Async preflight (#1096):
4091
+ * - `targetTaskId` resolves to a real task the caller can see.
4092
+ * - The target is a `fulfill_brief` (you cannot grade an arbitrary
4093
+ * task type as if it were a brief fulfillment).
4094
+ * - The target is `completed` with an accepted attempt — grading
4095
+ * an in-flight or failed task would either race or grade nothing.
4096
+ *
4097
+ * Agent-distinctness ("assessor ≠ producer") is a runtime / auth-
4098
+ * layer concern and intentionally NOT checked here. It belongs in
4099
+ * an auth-aware claim-time check.
4100
+ */
4101
+ async function validateAssessBriefInputAsync(input, ctx) {
4102
+ const { targetTaskId } = input;
4103
+ const errors = [];
4104
+ const target = await ctx.resolveTask(targetTaskId);
4105
+ if (!target) {
4106
+ errors.push({
4107
+ field: "targetTaskId",
4108
+ message: `targetTaskId ${targetTaskId} does not resolve to a task you can read`
4109
+ });
4110
+ return errors;
4111
+ }
4112
+ if (target.taskType !== "fulfill_brief") errors.push({
4113
+ field: "targetTaskId",
4114
+ message: `targetTaskId ${targetTaskId} is a ${target.taskType}, not a fulfill_brief`
4115
+ });
4116
+ if (target.status !== "completed" || target.acceptedAttemptN === null) errors.push({
4117
+ field: "targetTaskId",
4118
+ message: `targetTaskId ${targetTaskId} is not completed with an accepted attempt (status=${target.status}, acceptedAttemptN=${target.acceptedAttemptN})`
4119
+ });
4120
+ return errors;
4121
+ }
4089
4122
  //#endregion
4090
4123
  //#region ../../libs/tasks/src/task-types/curate-pack.ts
4091
4124
  /**
@@ -4284,6 +4317,311 @@ function validateJudgePackOutput(output) {
4284
4317
  }
4285
4318
  return null;
4286
4319
  }
4320
+ /**
4321
+ * Async preflight (#1096):
4322
+ * - `renderedPackId` resolves to a rendered_packs row.
4323
+ * - `sourcePackId` resolves to a context_packs row.
4324
+ * - The rendered pack actually came from the claimed source pack —
4325
+ * `renderedPack.sourcePackId === input.sourcePackId`. Without
4326
+ * this check a judge can be tricked into grading rendering A as
4327
+ * if it came from source B.
4328
+ */
4329
+ async function validateJudgePackInputAsync(input, ctx) {
4330
+ const { renderedPackId, sourcePackId } = input;
4331
+ const errors = [];
4332
+ const [rendered, source] = await Promise.all([ctx.resolveRenderedPack(renderedPackId), ctx.resolveContextPack(sourcePackId)]);
4333
+ if (!rendered) errors.push({
4334
+ field: "renderedPackId",
4335
+ message: `renderedPackId ${renderedPackId} does not resolve to a rendered pack you can read`
4336
+ });
4337
+ if (!source) errors.push({
4338
+ field: "sourcePackId",
4339
+ message: `sourcePackId ${sourcePackId} does not resolve to a context pack you can read`
4340
+ });
4341
+ if (rendered && source && rendered.sourcePackId !== source.id) errors.push({
4342
+ field: "sourcePackId",
4343
+ message: `renderedPack ${renderedPackId} was produced from source ${rendered.sourcePackId}, not from sourcePackId=${sourcePackId}`
4344
+ });
4345
+ return errors;
4346
+ }
4347
+ //#endregion
4348
+ //#region ../../libs/tasks/src/task-types/judge-eval-variant.ts
4349
+ /**
4350
+ * `judge_eval_variant` — score N variants of a `run_eval` scenario
4351
+ * against a single rubric, in one pass, with per-variant subagent
4352
+ * isolation.
4353
+ *
4354
+ * output_kind: judgment
4355
+ * criteria: required (`successCriteria.rubric` — same envelope shape as
4356
+ * `judge_pack` / `assess_brief`)
4357
+ * references: not required at the input layer — `runTaskIds` already
4358
+ * pin the targets being graded.
4359
+ *
4360
+ * Slice 2 of #943. The parent task carries the rubric and the list of
4361
+ * variant `run_eval` task ids. The pi executor registers the generic
4362
+ * `subagent` custom tool (#1087), and the parent LLM calls
4363
+ * `subagent({ task, output_schema: 'judge_eval_variant_result' })` once
4364
+ * per variant — each child session has fresh context, fetches the
4365
+ * variant's accepted attempt output via `moltnet_get_task` /
4366
+ * `moltnet_list_task_attempts`, and grades against the rubric.
4367
+ *
4368
+ * Reuses `JudgePackScore` from `judge_pack` for per-criterion scoring
4369
+ * (Lane 1 binary via `llm_checklist`, Lane 2 graded via `llm_score`,
4370
+ * deterministic_*) — the score shape is the same across judgment
4371
+ * tasks; only the wrapping (per-variant grouping + deltas) differs.
4372
+ *
4373
+ * Cross-task input invariants — "all targets share the same
4374
+ * correlation_id, all are `run_eval`, all are completed with an
4375
+ * accepted attempt, all share byte-identical `input.successCriteria`"
4376
+ * — REQUIRE async DB lookups and live in `validateInputAsync` below,
4377
+ * which the task service runs at create time (#1096 wiring). The
4378
+ * TypeBox layer here only enforces shape: UUID format,
4379
+ * minItems/maxItems, rubric presence + weight invariant.
4380
+ */
4381
+ var JUDGE_EVAL_VARIANT_TYPE = "judge_eval_variant";
4382
+ var JudgeEvalVariantInput = Type$2.Object({
4383
+ runTaskIds: Type$2.Array(Type$2.String({ format: "uuid" }), {
4384
+ minItems: 2,
4385
+ maxItems: 10
4386
+ }),
4387
+ successCriteria: SuccessCriteria
4388
+ }, {
4389
+ $id: "JudgeEvalVariantInput",
4390
+ additionalProperties: false
4391
+ });
4392
+ /**
4393
+ * Per-variant grading. `scores[]` shape is identical to `JudgePackScore`
4394
+ * (mode-aware: binary via `llm_checklist`, graded via `llm_score`,
4395
+ * deterministic_*). Reuse the type rather than re-declare.
4396
+ *
4397
+ * This is also the **subagent output contract** — the parent's
4398
+ * `subagent` tool resolves the contract name `judge_eval_variant_result`
4399
+ * to this schema. See `agent-runtime`'s subagent contract registry.
4400
+ */
4401
+ var JudgeEvalVariantResult = Type$2.Object({
4402
+ runTaskId: Type$2.String({ format: "uuid" }),
4403
+ variantLabel: Type$2.String({
4404
+ minLength: 1,
4405
+ maxLength: 64,
4406
+ pattern: "^(?!.* - ).*$"
4407
+ }),
4408
+ scores: Type$2.Array(JudgePackScore, { minItems: 1 }),
4409
+ composite: Type$2.Number({
4410
+ minimum: 0,
4411
+ maximum: 1
4412
+ }),
4413
+ verdict: Type$2.String({ minLength: 1 })
4414
+ }, {
4415
+ $id: "JudgeEvalVariantResult",
4416
+ additionalProperties: false
4417
+ });
4418
+ var JudgeEvalVariantOutput = Type$2.Object({
4419
+ results: Type$2.Array(JudgeEvalVariantResult, { minItems: 2 }),
4420
+ deltas: Type$2.Optional(Type$2.Record(Type$2.String(), Type$2.Number({
4421
+ minimum: -1,
4422
+ maximum: 1
4423
+ }))),
4424
+ judgeModel: Type$2.Optional(Type$2.String({ minLength: 1 })),
4425
+ traceparent: Type$2.String({ minLength: 1 })
4426
+ }, {
4427
+ $id: "JudgeEvalVariantOutput",
4428
+ additionalProperties: false
4429
+ });
4430
+ /**
4431
+ * Synchronous input invariants beyond TypeBox shape: rubric must be
4432
+ * present (already required by the schema, but the rubric body has
4433
+ * its own per-criterion weight invariant) and the rubric's weights
4434
+ * must sum to 1.
4435
+ *
4436
+ * Cross-task invariants (all targets are `run_eval`, all completed,
4437
+ * share `correlation_id`, byte-identical `input.successCriteria`)
4438
+ * are NOT checked here — they require async DB lookups against
4439
+ * `runTaskIds` and live in `validateJudgeEvalVariantInputAsync`
4440
+ * below, invoked by the task service at create time (#1096).
4441
+ */
4442
+ function validateJudgeEvalVariantInput(input) {
4443
+ const sc = input.successCriteria;
4444
+ if (!sc) return "successCriteria is required for judge_eval_variant";
4445
+ if (!sc.rubric) return "successCriteria.rubric is required for judge_eval_variant";
4446
+ return validateRubricWeights(sc.rubric);
4447
+ }
4448
+ /**
4449
+ * Output cross-field invariants the schema cannot express:
4450
+ *
4451
+ * 1. `results.length === input.runTaskIds.length` — every variant
4452
+ * the imposer asked for must be graded. Partial grading
4453
+ * invalidates cross-variant comparison; fail the whole task
4454
+ * rather than silently report a subset.
4455
+ *
4456
+ * 2. `results[i].runTaskId === input.runTaskIds[i]` — order is
4457
+ * load-bearing for downstream consumers (e.g. deltas keyed by
4458
+ * adjacent pairs). Mismatch is an LLM bug; reject loudly.
4459
+ *
4460
+ * 3. Each `result.scores` follows the same `llm_checklist` rule
4461
+ * `judge_pack` enforces (#999): if a score has an `assertions`
4462
+ * array, the numeric score MUST be `1` iff every assertion
4463
+ * passes. Inconsistent payloads pollute attestations.
4464
+ *
4465
+ * 4. Each `result.composite` MUST equal the rubric-weighted sum
4466
+ * `Σ(weight_j × scores[j].score)`. The parent (and any subagent
4467
+ * it delegated to) is supposed to compute this; surfacing a
4468
+ * drift here catches LLMs that hand-wave the arithmetic.
4469
+ *
4470
+ * 5. Optional `deltas` keys MUST be of the form `"A - B"` where
4471
+ * both `A` and `B` are variantLabels present in `results`.
4472
+ * Values are not range-checked (any float in [-1, 1] is
4473
+ * arithmetically possible).
4474
+ */
4475
+ function validateJudgeEvalVariantOutput(output, input) {
4476
+ const out = output;
4477
+ const inp = input;
4478
+ if (inp) {
4479
+ if (out.results.length !== inp.runTaskIds.length) return `results.length (${out.results.length}) does not match input.runTaskIds.length (${inp.runTaskIds.length}). Every variant must be graded; partial grading is rejected.`;
4480
+ for (let i = 0; i < out.results.length; i++) if (out.results[i].runTaskId !== inp.runTaskIds[i]) return `results[${i}].runTaskId (${out.results[i].runTaskId}) does not match input.runTaskIds[${i}] (${inp.runTaskIds[i]}). Order must align with input for downstream delta computation.`;
4481
+ }
4482
+ for (let r = 0; r < out.results.length; r++) {
4483
+ const result = out.results[r];
4484
+ for (let s = 0; s < result.scores.length; s++) {
4485
+ const sc = result.scores[s];
4486
+ if (!sc.assertions) continue;
4487
+ const allPassed = sc.assertions.every((a) => a.passed);
4488
+ const expected = allPassed ? 1 : 0;
4489
+ if (sc.score !== expected) return `results[${r}].scores[${s}] (criterionId="${sc.criterionId}"): assertions ${allPassed ? "all pass" : "have at least one fail"} but score=${sc.score}. Score must be derived: 1 iff every assertion passes, else 0 (#999 llm_checklist rule).`;
4490
+ }
4491
+ }
4492
+ if (inp?.successCriteria?.rubric) {
4493
+ const criteria = inp.successCriteria.rubric.criteria;
4494
+ const weightById = new Map(criteria.map((c) => [c.id, c.weight]));
4495
+ for (let r = 0; r < out.results.length; r++) {
4496
+ const result = out.results[r];
4497
+ let sum = 0;
4498
+ for (const sc of result.scores) {
4499
+ const w = weightById.get(sc.criterionId);
4500
+ if (w === void 0) return `results[${r}].scores: criterionId "${sc.criterionId}" is not in the input rubric (known: ${Array.from(weightById.keys()).join(", ")}). Score every rubric criterion exactly once; do not invent new ids.`;
4501
+ sum += w * sc.score;
4502
+ }
4503
+ if (Math.abs(sum - result.composite) > .001) return `results[${r}].composite (${result.composite}) does not match Σ(weight × score) (${sum.toFixed(6)}). Composite must be the rubric-weighted sum of per-criterion scores (drift > 0.001).`;
4504
+ }
4505
+ }
4506
+ if (out.deltas) {
4507
+ const labels = new Set(out.results.map((r) => r.variantLabel));
4508
+ for (const key of Object.keys(out.deltas)) {
4509
+ const m = /^(.+?) - (.+)$/.exec(key);
4510
+ if (!m) return `deltas key "${key}" is not of the form "<variantLabel-A> - <variantLabel-B>". Use a single space-hyphen-space separator between labels.`;
4511
+ const [, a, b] = m;
4512
+ if (!labels.has(a) || !labels.has(b)) return `deltas key "${key}" references variantLabel(s) not present in results: ${!labels.has(a) ? `"${a}" missing` : ""}${!labels.has(a) && !labels.has(b) ? ", " : ""}${!labels.has(b) ? `"${b}" missing` : ""}`;
4513
+ }
4514
+ }
4515
+ return null;
4516
+ }
4517
+ /**
4518
+ * Local stable-stringify for cross-variant `successCriteria` byte-
4519
+ * equality. Recursively sorts object keys; arrays preserve order
4520
+ * (intentional — rubric criteria order is semantically meaningful).
4521
+ * Mirrors the canonical-JSON shape `crypto-service` uses for CIDs,
4522
+ * without taking on a crypto-service dep just for this comparison.
4523
+ */
4524
+ function stableStringify(value) {
4525
+ if (value === null || typeof value !== "object") return JSON.stringify(value);
4526
+ if (Array.isArray(value)) return "[" + value.map(stableStringify).join(",") + "]";
4527
+ const obj = value;
4528
+ return "{" + Object.keys(obj).sort().map((k) => JSON.stringify(k) + ":" + stableStringify(obj[k])).join(",") + "}";
4529
+ }
4530
+ /**
4531
+ * Async preflight for `judge_eval_variant` (#1096 + #943):
4532
+ *
4533
+ * 1. Every `runTaskIds[i]` resolves to a task the caller can read.
4534
+ * 2. Every resolved task is `taskType === 'run_eval'`.
4535
+ * 3. Every resolved task is `status === 'completed'` with a
4536
+ * non-null `acceptedAttemptN` — grading an unaccepted attempt
4537
+ * races with re-attempts and pollutes the judge attestation.
4538
+ * 4. Every resolved task shares a non-null `correlationId`, and all
4539
+ * `correlationId`s are equal. Without this an imposer could
4540
+ * fabricate a "variant set" by stapling unrelated runs together.
4541
+ * 5. The shared `correlationId` is NOT already sealed. A previous
4542
+ * judge_eval_variant against the same group is final; produce a
4543
+ * fresh correlation_id for a new judging round rather than
4544
+ * adding contradictory verdicts to a sealed group.
4545
+ * 6. Every variant's `input.successCriteria` is byte-identical (via
4546
+ * stable-stringify). Different rubrics across "variants" makes
4547
+ * the comparison meaningless.
4548
+ */
4549
+ async function validateJudgeEvalVariantInputAsync(input, ctx) {
4550
+ const { runTaskIds } = input;
4551
+ const errors = [];
4552
+ const resolved = await Promise.all(runTaskIds.map((id) => ctx.resolveTask(id)));
4553
+ let missingTargets = false;
4554
+ const presentTargets = [];
4555
+ for (let i = 0; i < runTaskIds.length; i++) {
4556
+ const t = resolved[i];
4557
+ if (!t) {
4558
+ missingTargets = true;
4559
+ errors.push({
4560
+ field: `runTaskIds[${i}]`,
4561
+ message: `runTaskIds[${i}]=${runTaskIds[i]} does not resolve to a task you can read`
4562
+ });
4563
+ continue;
4564
+ }
4565
+ presentTargets.push(t);
4566
+ if (t.taskType !== "run_eval") errors.push({
4567
+ field: `runTaskIds[${i}]`,
4568
+ message: `runTaskIds[${i}]=${runTaskIds[i]} is a ${t.taskType}, not a run_eval`
4569
+ });
4570
+ if (t.status !== "completed" || t.acceptedAttemptN === null) errors.push({
4571
+ field: `runTaskIds[${i}]`,
4572
+ message: `runTaskIds[${i}]=${runTaskIds[i]} is not completed with an accepted attempt (status=${t.status}, acceptedAttemptN=${t.acceptedAttemptN})`
4573
+ });
4574
+ }
4575
+ if (missingTargets || presentTargets.length === 0) return errors;
4576
+ const correlationIds = new Set(presentTargets.map((t) => t.correlationId ?? "__null__"));
4577
+ if (correlationIds.has("__null__")) errors.push({
4578
+ field: "runTaskIds",
4579
+ message: "one or more run_eval targets have no correlation_id; cannot group as variants"
4580
+ });
4581
+ if (correlationIds.size > 1) errors.push({
4582
+ field: "runTaskIds",
4583
+ message: `run_eval targets span multiple correlation_ids (${Array.from(correlationIds).join(", ")}); variants must share one`
4584
+ });
4585
+ if (errors.length > 0) return errors;
4586
+ const correlationId = presentTargets[0].correlationId;
4587
+ if (!correlationId) return errors;
4588
+ const seal = await ctx.findCorrelationSeal(correlationId);
4589
+ if (seal) errors.push({
4590
+ field: "runTaskIds",
4591
+ message: `correlation_id ${correlationId} is already sealed by ${seal.sealedByTaskType}/${seal.sealedByTaskId} at ${seal.sealedAt}; use a fresh correlation_id for a new judging round`
4592
+ });
4593
+ const first = stableStringify(presentTargets[0].input.successCriteria);
4594
+ for (let i = 1; i < presentTargets.length; i++) if (stableStringify(presentTargets[i].input.successCriteria) !== first) {
4595
+ errors.push({
4596
+ field: `runTaskIds[${i}]`,
4597
+ message: `runTaskIds[${i}] has a different input.successCriteria than runTaskIds[0]; all variants must share the rubric and gates`
4598
+ });
4599
+ break;
4600
+ }
4601
+ return errors;
4602
+ }
4603
+ /**
4604
+ * Side effect emitted on successful `judge_eval_variant` create:
4605
+ * seal the shared correlation_id atomically with the insert. The
4606
+ * task service applies the seal in the same transaction; a
4607
+ * concurrent second `judge_eval_variant` against the same group
4608
+ * loses the race and is rejected with a clean conflict error.
4609
+ *
4610
+ * The seal applies to the SHARED correlation_id of the targets —
4611
+ * NOT to the judge task's own correlationId (which is typically
4612
+ * null or distinct). The task service derives the correlationId
4613
+ * for the effect from the resolved targets, not from the judge
4614
+ * task row.
4615
+ */
4616
+ async function onCreateJudgeEvalVariant(input, ctx) {
4617
+ const { runTaskIds } = input;
4618
+ const first = await ctx.resolveTask(runTaskIds[0]);
4619
+ if (!first?.correlationId) return [];
4620
+ return [{
4621
+ kind: "sealCorrelation",
4622
+ correlationId: first.correlationId
4623
+ }];
4624
+ }
4287
4625
  //#endregion
4288
4626
  //#region ../../libs/tasks/src/task-types/render-pack.ts
4289
4627
  /**
@@ -4323,6 +4661,18 @@ var RenderPackOutput = Type$2.Object({
4323
4661
  $id: "RenderPackOutput",
4324
4662
  additionalProperties: false
4325
4663
  });
4664
+ /**
4665
+ * Async preflight (#1096): `packId` resolves to a context_packs row
4666
+ * the caller can read.
4667
+ */
4668
+ async function validateRenderPackInputAsync(input, ctx) {
4669
+ const { packId } = input;
4670
+ if (!await ctx.resolveContextPack(packId)) return [{
4671
+ field: "packId",
4672
+ message: `packId ${packId} does not resolve to a context pack you can read`
4673
+ }];
4674
+ return [];
4675
+ }
4326
4676
  //#endregion
4327
4677
  //#region ../../libs/tasks/src/task-types/run-eval.ts
4328
4678
  /**
@@ -4430,7 +4780,8 @@ var BUILT_IN_TASK_TYPES = {
4430
4780
  outputSchema: AssessBriefOutput,
4431
4781
  outputKind: "judgment",
4432
4782
  requiresReferences: true,
4433
- validateInput: validateJudgmentInput
4783
+ validateInput: validateJudgmentInput,
4784
+ validateInputAsync: validateAssessBriefInputAsync
4434
4785
  },
4435
4786
  [CURATE_PACK_TYPE]: {
4436
4787
  name: CURATE_PACK_TYPE,
@@ -4446,7 +4797,8 @@ var BUILT_IN_TASK_TYPES = {
4446
4797
  outputSchema: RenderPackOutput,
4447
4798
  outputKind: "artifact",
4448
4799
  requiresReferences: false,
4449
- validateOutput: requireVerificationWhenCriteriaPresent
4800
+ validateOutput: requireVerificationWhenCriteriaPresent,
4801
+ validateInputAsync: validateRenderPackInputAsync
4450
4802
  },
4451
4803
  [JUDGE_PACK_TYPE]: {
4452
4804
  name: JUDGE_PACK_TYPE,
@@ -4455,7 +4807,8 @@ var BUILT_IN_TASK_TYPES = {
4455
4807
  outputKind: "judgment",
4456
4808
  requiresReferences: true,
4457
4809
  validateInput: validateJudgmentInput,
4458
- validateOutput: validateJudgePackOutput
4810
+ validateOutput: validateJudgePackOutput,
4811
+ validateInputAsync: validateJudgePackInputAsync
4459
4812
  },
4460
4813
  [RUN_EVAL_TYPE]: {
4461
4814
  name: RUN_EVAL_TYPE,
@@ -4464,6 +4817,18 @@ var BUILT_IN_TASK_TYPES = {
4464
4817
  outputKind: "artifact",
4465
4818
  requiresReferences: false,
4466
4819
  validateOutput: validateRunEvalOutput
4820
+ },
4821
+ [JUDGE_EVAL_VARIANT_TYPE]: {
4822
+ name: JUDGE_EVAL_VARIANT_TYPE,
4823
+ inputSchema: JudgeEvalVariantInput,
4824
+ outputSchema: JudgeEvalVariantOutput,
4825
+ outputKind: "judgment",
4826
+ requiresReferences: false,
4827
+ validateInput: validateJudgeEvalVariantInput,
4828
+ validateOutput: validateJudgeEvalVariantOutput,
4829
+ validateInputAsync: validateJudgeEvalVariantInputAsync,
4830
+ onCreate: onCreateJudgeEvalVariant,
4831
+ usesSubagents: true
4467
4832
  }
4468
4833
  };
4469
4834
  //#endregion
@@ -5362,6 +5727,15 @@ function validateTaskOutput(taskType, output, input) {
5362
5727
  function getTaskOutputSchema(taskType) {
5363
5728
  return getTaskTypeEntry(taskType)?.outputSchema ?? null;
5364
5729
  }
5730
+ /**
5731
+ * Whether sessions running this task type should have the generic
5732
+ * `subagent` custom tool registered. Returns `false` for unknown task
5733
+ * types and for task types that didn't opt in. See `TaskTypeEntry`
5734
+ * for the design rationale.
5735
+ */
5736
+ function taskTypeUsesSubagents(taskType) {
5737
+ return getTaskTypeEntry(taskType)?.usesSubagents === true;
5738
+ }
5365
5739
  //#endregion
5366
5740
  //#region ../../libs/tasks/src/wire.ts
5367
5741
  /**
@@ -5728,6 +6102,78 @@ function isHelpFlag(args) {
5728
6102
  return args.includes("--help") || args.includes("-h");
5729
6103
  }
5730
6104
  //#endregion
6105
+ //#region ../../libs/agent-runtime/src/subagent-output-contracts.ts
6106
+ var REGISTRY = /* @__PURE__ */ new Map();
6107
+ /**
6108
+ * Register a subagent output contract. Idempotent: re-registering the
6109
+ * same name with a different schema throws — contracts are meant to
6110
+ * be stable. Re-registering with the identical contract object (same
6111
+ * reference) is a no-op for HMR and test convenience.
6112
+ *
6113
+ * Typically called at module-init time alongside task-type
6114
+ * registration. See task-types/index.ts in @moltnet/tasks for the
6115
+ * conventional pattern.
6116
+ */
6117
+ function registerSubagentOutputContract(contract) {
6118
+ if (!contract.name || contract.name.trim().length === 0) throw new Error("subagent output contract name is required");
6119
+ if (!/^[a-z][a-z0-9_]*$/.test(contract.name)) throw new Error(`subagent output contract name '${contract.name}' must be lower_snake_case (starts with a letter, then [a-z0-9_]+)`);
6120
+ const existing = REGISTRY.get(contract.name);
6121
+ if (existing && existing !== contract) {
6122
+ if (existing.parametersSchema !== contract.parametersSchema) throw new Error(`subagent output contract '${contract.name}' is already registered with a different schema; refusing to override`);
6123
+ }
6124
+ REGISTRY.set(contract.name, contract);
6125
+ }
6126
+ /**
6127
+ * Resolve a subagent output contract by name. Returns `null` for
6128
+ * unknown names — callers (the subagent custom tool) decide whether
6129
+ * that's a tool error the parent LLM can recover from or a hard fail.
6130
+ */
6131
+ function getSubagentOutputContract(name) {
6132
+ return REGISTRY.get(name) ?? null;
6133
+ }
6134
+ /**
6135
+ * List all registered contracts. Useful for diagnostics and for the
6136
+ * subagent tool's parameter description so a parent LLM can see what
6137
+ * contracts are available without enumerating them in its prompt.
6138
+ */
6139
+ function listSubagentOutputContracts() {
6140
+ return [...REGISTRY.values()];
6141
+ }
6142
+ //#endregion
6143
+ //#region ../../libs/agent-runtime/src/built-in-contract-registrations.ts
6144
+ /**
6145
+ * Built-in subagent output contracts (#1087, #943).
6146
+ *
6147
+ * Why this is an exported function and not a module-init side
6148
+ * effect:
6149
+ *
6150
+ * - The registry is process-global. Module-init registration
6151
+ * fires exactly once per Node process (ESM modules are cached
6152
+ * by URL). Tests that call `__resetSubagentOutputContractsForTests()`
6153
+ * to start from an empty registry have no way to repopulate
6154
+ * the built-ins without re-evaluating the module — which the
6155
+ * cache prevents. PR #1101 review M4.
6156
+ * - An explicit `registerBuiltInSubagentContracts()` lets the
6157
+ * package index call it once at module load AND lets test
6158
+ * setup hooks call it again after `__reset...`.
6159
+ * - `registerSubagentOutputContract` is itself idempotent for
6160
+ * identical re-registrations, so calling this function twice
6161
+ * in the same process is safe.
6162
+ *
6163
+ * Adding a new built-in: extend the body of this function. Do not
6164
+ * call `registerSubagentOutputContract` from anywhere else in the
6165
+ * package — keeping all built-ins in one function makes the set
6166
+ * auditable.
6167
+ */
6168
+ function registerBuiltInSubagentContracts() {
6169
+ registerSubagentOutputContract({
6170
+ name: "judge_eval_variant_result",
6171
+ description: "Per-variant grading result produced by a subagent of judge_eval_variant: scores against the shared rubric, composite, and a 1-3 sentence verdict for a single variant.",
6172
+ parametersSchema: JudgeEvalVariantResult
6173
+ });
6174
+ }
6175
+ registerBuiltInSubagentContracts();
6176
+ //#endregion
5731
6177
  //#region ../../libs/agent-runtime/src/context-bindings.ts
5732
6178
  var PROMPT_SEPARATOR = "\n\n---\n\n";
5733
6179
  /**
@@ -6255,6 +6701,109 @@ function buildFulfillBriefUserPrompt(input, ctx) {
6255
6701
  ].filter(Boolean).join("\n");
6256
6702
  }
6257
6703
  //#endregion
6704
+ //#region ../../libs/agent-runtime/src/prompts/judge-eval-variant.ts
6705
+ /**
6706
+ * Build the first user-message prompt for a `judge_eval_variant` task
6707
+ * (#943 Slice 2).
6708
+ *
6709
+ * The parent agent's job is **fan-out-and-collect**: for each
6710
+ * `runTaskIds[i]`, spawn an isolated subagent via the `subagent` custom
6711
+ * tool (#1087), have it grade that variant against the shared rubric,
6712
+ * and collect each subagent's structured `judge_eval_variant_result`
6713
+ * payload. The parent does NOT grade itself; it composes the per-
6714
+ * variant results into the final `judge_eval_variant` output (results
6715
+ * array + optional deltas + verdicts).
6716
+ *
6717
+ * Isolation is the point: each variant gets a fresh subagent session
6718
+ * with no carryover context from sibling variants, so per-variant
6719
+ * grading is independent. Cost is bounded by `maxItems: 10` on
6720
+ * runTaskIds.
6721
+ */
6722
+ function buildJudgeEvalVariantUserPrompt(input, ctx) {
6723
+ const { runTaskIds, successCriteria } = input;
6724
+ const rubric = successCriteria.rubric;
6725
+ if (!rubric) throw new Error("judge_eval_variant requires successCriteria.rubric — none present");
6726
+ const escapeCell = (s) => s.replace(/\\/g, "\\\\").replace(/\|/g, "\\|").replace(/\r?\n/g, " ");
6727
+ const criteriaTable = rubric.criteria.map((c) => `| \`${c.id}\` | ${c.weight.toFixed(3)} | ${c.scoring} | ${escapeCell(c.description)} |`).join("\n");
6728
+ const targetsBlock = runTaskIds.map((id, i) => `${i + 1}. \`${id}\``).join("\n");
6729
+ const finalOutputBlock = buildFinalOutputBlock({
6730
+ taskType: "judge_eval_variant",
6731
+ outputSchemaName: "JudgeEvalVariantOutput",
6732
+ shapeSketch: [
6733
+ "{",
6734
+ " \"results\": [",
6735
+ " {",
6736
+ " \"runTaskId\": \"<runTaskIds[i]>\",",
6737
+ " \"variantLabel\": \"<from variant input>\",",
6738
+ " \"scores\": [ { \"criterionId\": \"...\", \"score\": 0..1, \"rationale\": \"...\", \"assertions\": [...]? } ],",
6739
+ " \"composite\": <Σ(weight × score), 0..1>,",
6740
+ " \"verdict\": \"<1-3 sentences>\"",
6741
+ " },",
6742
+ " ...one entry per runTaskIds[i], same order",
6743
+ " ],",
6744
+ " \"deltas\": { \"<labelA> - <labelB>\": <composite(A) - composite(B)> }, // optional",
6745
+ " \"judgeModel\": \"<id>\", // optional",
6746
+ " \"traceparent\": \"<from claim>\"",
6747
+ "}"
6748
+ ].join("\n")
6749
+ });
6750
+ return [
6751
+ "# Judge Eval Variants\n",
6752
+ `You are grading ${runTaskIds.length} variants of a single run_eval scenario`,
6753
+ "against ONE shared rubric. Your job is fan-out-and-collect — you do not",
6754
+ "grade yourself.",
6755
+ "",
6756
+ `Task id: \`${ctx.taskId}\``,
6757
+ `Diary: \`${ctx.diaryId}\``,
6758
+ "",
6759
+ "### Targets (variants to grade)",
6760
+ "",
6761
+ targetsBlock,
6762
+ "",
6763
+ "Each target is a completed `run_eval` task in the same correlation group.",
6764
+ "Read its accepted attempt via `moltnet_get_task` / `moltnet_list_task_attempts`",
6765
+ "to see the producer's output before grading.",
6766
+ "",
6767
+ "### Rubric",
6768
+ "",
6769
+ rubric.preamble ? `${rubric.preamble}\n` : "",
6770
+ "| Criterion | Weight | Scoring | Description |",
6771
+ "| --- | --- | --- | --- |",
6772
+ criteriaTable,
6773
+ "",
6774
+ "### How to grade",
6775
+ "",
6776
+ "For EACH `runTaskIds[i]`:",
6777
+ "",
6778
+ "1. Call the `subagent` custom tool with:",
6779
+ " - `task`: a brief instructing the subagent to grade ONLY that variant",
6780
+ " against the rubric above; include the target task id and the rubric",
6781
+ " verbatim. The subagent has the same MoltNet tools and can fetch the",
6782
+ " accepted attempt output independently.",
6783
+ " - `output_schema`: `\"judge_eval_variant_result\"`",
6784
+ "2. Receive the subagent's structured `judge_eval_variant_result` payload.",
6785
+ "3. Append it to your `results[]` array, **in the same order as input.runTaskIds**.",
6786
+ "",
6787
+ "Do NOT score any variant in your own session. The whole point of the",
6788
+ "subagent fan-out is per-variant context isolation — grading two variants",
6789
+ "back-to-back in one session lets the second be biased by the first.",
6790
+ "",
6791
+ "### Composite arithmetic",
6792
+ "",
6793
+ "Each `composite` MUST equal `Σ(criterion.weight × score)` over the rubric",
6794
+ "criteria. Drift > 0.001 is rejected. Subagents are instructed to compute it",
6795
+ "themselves; double-check before assembling the final output.",
6796
+ "",
6797
+ "### Deltas (optional)",
6798
+ "",
6799
+ "If useful, populate `deltas` with pairwise composite differences keyed by",
6800
+ "`\"<variantLabel-A> - <variantLabel-B>\"` (single space-hyphen-space). Both",
6801
+ "labels must appear in `results`. Omit `deltas` entirely if not used.",
6802
+ "",
6803
+ finalOutputBlock
6804
+ ].filter((s) => s !== "").join("\n");
6805
+ }
6806
+ //#endregion
6258
6807
  //#region ../../libs/agent-runtime/src/prompts/judge-pack.ts
6259
6808
  function buildJudgePackUserPrompt(input, ctx) {
6260
6809
  const { renderedPackId, sourcePackId, successCriteria } = input;
@@ -6561,6 +7110,15 @@ function buildTaskUserPrompt(task, ctx) {
6561
7110
  diaryId: ctx.diaryId,
6562
7111
  taskId: ctx.taskId
6563
7112
  });
7113
+ case JUDGE_EVAL_VARIANT_TYPE:
7114
+ if (!Check(JudgeEvalVariantInput, task.input)) {
7115
+ const errors = [...Errors(JudgeEvalVariantInput, task.input)];
7116
+ throw new Error(`judge_eval_variant input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
7117
+ }
7118
+ return buildJudgeEvalVariantUserPrompt(task.input, {
7119
+ diaryId: ctx.diaryId,
7120
+ taskId: ctx.taskId
7121
+ });
6564
7122
  case RUN_EVAL_TYPE:
6565
7123
  if (!Check(RunEvalInput, task.input)) {
6566
7124
  const errors = [...Errors(RunEvalInput, task.input)];
@@ -9442,11 +10000,12 @@ function createCryptoNamespace(context, signingRequests) {
9442
10000
  function createDiariesNamespace(context) {
9443
10001
  const { client, auth } = context;
9444
10002
  return {
9445
- async list(query) {
10003
+ async list(query, headers) {
9446
10004
  return unwrapResult(await listDiaries({
9447
10005
  client,
9448
10006
  auth,
9449
- query
10007
+ query,
10008
+ headers
9450
10009
  }));
9451
10010
  },
9452
10011
  async create(body, headers) {
@@ -14288,6 +14847,27 @@ var BASE_ALLOWED_HOSTS = [
14288
14847
  "*.googlesource.com"
14289
14848
  ];
14290
14849
  /**
14850
+ * Run a shell command in the guest and throw if it fails. Mirror of
14851
+ * `run()` in `snapshot.ts` for the resume-side hook chain — every
14852
+ * setup step is essential to a healthy session, so a silent non-zero
14853
+ * exit (e.g. a mount that fails into the FUSE write path, or a
14854
+ * consumer-provided resume command that fails to install pnpm) must
14855
+ * surface immediately rather than fall through to cryptic agent
14856
+ * errors later.
14857
+ */
14858
+ async function vmRun(vm, label, command) {
14859
+ const wrapped = `set -eu\nset -o pipefail\n${command}`;
14860
+ const r = await vm.exec([
14861
+ "sh",
14862
+ "-c",
14863
+ wrapped
14864
+ ]);
14865
+ if (r.exitCode !== 0) {
14866
+ const tail = [r.stderr, r.stdout].filter(Boolean).join("\n").slice(-800);
14867
+ throw new Error(`resume step "${label}" failed (exit ${r.exitCode}):\n${tail}`);
14868
+ }
14869
+ }
14870
+ /**
14291
14871
  * Resume a VM from a checkpoint, inject credentials, configure egress +
14292
14872
  * TLS. Returns the managed VM handle.
14293
14873
  */
@@ -14347,8 +14927,9 @@ async function resumeVm(config) {
14347
14927
  update-ca-certificates 2>/dev/null
14348
14928
  cat /etc/gondolin/mitm/ca.crt >> /etc/ssl/certs/ca-certificates.crt
14349
14929
  '`);
14350
- await vm.exec(`sh -c 'echo "nameserver 8.8.8.8
14351
- nameserver 1.1.1.1" > /etc/resolv.conf'`);
14930
+ await vmRun(vm, "DNS resolvers", `printf 'nameserver 8.8.8.8\\nnameserver 1.1.1.1\\n' > /etc/resolv.conf`);
14931
+ await vmRun(vm, "git safe.directory", `git config --system --add safe.directory '*'`);
14932
+ for (const [i, cmd] of (config.sandboxConfig?.resumeCommands ?? []).entries()) await vmRun(vm, `resumeCommands[${i}]`, cmd);
14352
14933
  const vmSshDir = `${vmAgentDir}/ssh`;
14353
14934
  await vm.exec(`mkdir -p ${vmAgentDir}/ssh /home/agent/.pi/agent`);
14354
14935
  if (creds.piAuthJson !== null) await vm.fs.writeFile("/home/agent/.pi/agent/auth.json", creds.piAuthJson, { mode: 384 });
@@ -14691,6 +15272,39 @@ function extractUsage(message) {
14691
15272
  };
14692
15273
  }
14693
15274
  //#endregion
15275
+ //#region ../../libs/pi-extension/src/runtime/agent-session-factory.ts
15276
+ var NO_SKILLS = () => ({
15277
+ skills: [],
15278
+ diagnostics: []
15279
+ });
15280
+ /**
15281
+ * Construct an in-memory `AgentSession`. The caller is responsible for
15282
+ * eventually invoking `session.prompt(...)` and for tearing down — the
15283
+ * helper does no lifecycle management beyond construction.
15284
+ */
15285
+ async function buildAgentSession(args) {
15286
+ const piOtelExtension = createPiOtelExtension({
15287
+ agentName: args.agentName,
15288
+ spanAttributes: args.otelSpanAttrs
15289
+ });
15290
+ const resourceLoader = new DefaultResourceLoader({
15291
+ cwd: args.mountPath,
15292
+ agentDir: args.piAuthDir,
15293
+ extensionFactories: [piOtelExtension],
15294
+ appendSystemPrompt: args.appendSystemPrompt,
15295
+ skillsOverride: args.skillsOverride ?? NO_SKILLS
15296
+ });
15297
+ await resourceLoader.reload();
15298
+ return (await createAgentSession({
15299
+ agentDir: args.piAuthDir,
15300
+ cwd: args.mountPath,
15301
+ model: args.modelHandle,
15302
+ customTools: args.customTools,
15303
+ sessionManager: SessionManager.inMemory(),
15304
+ resourceLoader
15305
+ })).session;
15306
+ }
15307
+ //#endregion
14694
15308
  //#region ../../libs/pi-extension/src/runtime/inject-task-context.ts
14695
15309
  /**
14696
15310
  * Slice 1.5 of #943 — wire the agent-runtime resolver into the
@@ -14884,6 +15498,190 @@ function buildRuntimeInstructor(ctx) {
14884
15498
  ].join("\n");
14885
15499
  }
14886
15500
  //#endregion
15501
+ //#region ../../libs/pi-extension/src/runtime/subagent-tool.ts
15502
+ var SUBAGENT_SUBMIT_TOOL_NAME = "submit_subagent_output";
15503
+ /**
15504
+ * Parameters shape the parent LLM sees when calling the subagent tool.
15505
+ *
15506
+ * - `task` — natural-language instructions for the subagent.
15507
+ * The parent authors this per call. Must be
15508
+ * non-empty.
15509
+ * - `output_schema` — name of a registered SubagentOutputContract.
15510
+ * Resolved at call time; unknown names error.
15511
+ */
15512
+ var SubagentToolParameters = Type$2.Object({
15513
+ task: Type$2.String({
15514
+ minLength: 1,
15515
+ description: "Natural-language instructions for the subagent. The subagent starts with a fresh conversation and a narrowed system prompt; this is the only context it has from you."
15516
+ }),
15517
+ output_schema: Type$2.String({
15518
+ minLength: 1,
15519
+ description: "Name of a registered subagent output contract. The subagent must submit a structured payload via `submit_subagent_output` matching this contract."
15520
+ })
15521
+ }, { additionalProperties: false });
15522
+ var DEFAULT_SUBAGENT_TIMEOUT_MS = 300 * 1e3;
15523
+ /**
15524
+ * Build the subagent custom tool for a parent session. The handle
15525
+ * exposes the call counter so executors can emit summary telemetry
15526
+ * when the parent terminates.
15527
+ */
15528
+ function createSubagentTool(args) {
15529
+ const buildSession = args.buildAgentSession ?? buildAgentSession;
15530
+ let callCount = 0;
15531
+ return {
15532
+ tool: defineTool({
15533
+ name: "subagent",
15534
+ label: "Delegate to subagent",
15535
+ description: subagentToolDescription(),
15536
+ parameters: SubagentToolParameters,
15537
+ async execute(_id, params) {
15538
+ if (!Check(SubagentToolParameters, params)) return toolError(`subagent: invalid parameters: ${JSON.stringify([...Errors(SubagentToolParameters, params)].slice(0, 3))}`);
15539
+ const { task, output_schema } = params;
15540
+ const contract = getSubagentOutputContract(output_schema);
15541
+ if (!contract) return toolError(`subagent: unknown output_schema "${output_schema}". Registered contracts: [${listSubagentOutputContracts().map((c) => c.name).join(", ")}]`);
15542
+ callCount += 1;
15543
+ const callIndex = callCount;
15544
+ let captured = null;
15545
+ const submitTool = defineTool({
15546
+ name: SUBAGENT_SUBMIT_TOOL_NAME,
15547
+ label: `Submit ${output_schema}`,
15548
+ description: `Submit your structured output for this subagent task. Call exactly once when done. Args MUST match the ${output_schema} contract; mismatches return a tool error you can recover from in the same session.`,
15549
+ parameters: contract.parametersSchema,
15550
+ async execute(_innerId, innerParams) {
15551
+ if (!Check(contract.parametersSchema, innerParams)) return toolError(`submit_subagent_output: schema validation failed: ${[...Errors(contract.parametersSchema, innerParams)].slice(0, 3).map((e) => `${e.path}: ${e.message}`).join("; ")}. Re-call with a corrected payload.`);
15552
+ captured = innerParams;
15553
+ return {
15554
+ content: [{
15555
+ type: "text",
15556
+ text: "Output captured. Subagent session will terminate; no further action needed."
15557
+ }],
15558
+ details: { captured: true },
15559
+ terminate: true
15560
+ };
15561
+ }
15562
+ });
15563
+ const subagentInstructor = buildSubagentInstructor({
15564
+ contractName: output_schema,
15565
+ contractDescription: contract.description,
15566
+ parentTaskId: args.parentTaskId,
15567
+ callIndex
15568
+ });
15569
+ const session = await buildSession({
15570
+ mountPath: args.mountPath,
15571
+ piAuthDir: args.piAuthDir,
15572
+ modelHandle: args.modelHandle,
15573
+ agentName: args.agentName,
15574
+ customTools: [...args.inheritedCustomTools, submitTool],
15575
+ appendSystemPrompt: [args.parentRuntimeInstructor, subagentInstructor],
15576
+ skillsOverride: () => ({
15577
+ skills: [],
15578
+ diagnostics: []
15579
+ }),
15580
+ otelSpanAttrs: {
15581
+ "moltnet.task.id": args.parentTaskId,
15582
+ "moltnet.task.type": args.parentTaskType,
15583
+ "moltnet.task.attempt": args.parentAttemptN,
15584
+ "moltnet.subagent.contract": output_schema,
15585
+ "moltnet.subagent.index": callIndex
15586
+ }
15587
+ });
15588
+ let abortReason = null;
15589
+ let abortInvoked = false;
15590
+ const fireAbort = (reason) => {
15591
+ if (abortInvoked) return;
15592
+ abortInvoked = true;
15593
+ abortReason = reason;
15594
+ session.abort().catch((err) => {
15595
+ const message = err instanceof Error ? err.message : String(err);
15596
+ process.stderr.write(`[subagent] inner session.abort() failed: ${message}\n`);
15597
+ });
15598
+ };
15599
+ const cancelListener = args.parentCancelSignal ? (() => {
15600
+ const signal = args.parentCancelSignal;
15601
+ const listener = () => fireAbort("parent_cancelled");
15602
+ if (signal.aborted) listener();
15603
+ else signal.addEventListener("abort", listener, { once: true });
15604
+ return () => signal.removeEventListener("abort", listener);
15605
+ })() : null;
15606
+ const timeoutMs = args.timeoutMs === void 0 || args.timeoutMs < 0 ? DEFAULT_SUBAGENT_TIMEOUT_MS : args.timeoutMs;
15607
+ const timeoutHandle = timeoutMs > 0 ? setTimeout(() => fireAbort("subagent_timed_out"), timeoutMs) : null;
15608
+ try {
15609
+ await session.prompt(task);
15610
+ } catch (err) {
15611
+ return toolError(`subagent: inner session.prompt() threw: ${err instanceof Error ? err.message : String(err)}`);
15612
+ } finally {
15613
+ if (timeoutHandle) clearTimeout(timeoutHandle);
15614
+ if (cancelListener) cancelListener();
15615
+ }
15616
+ if (abortReason !== null) return toolError(`subagent: ${abortReason === "subagent_timed_out" ? `subagent timed out after ${timeoutMs}ms` : "parent task was cancelled"}. The parent should fail this task or retry with a clearer scope.`);
15617
+ if (captured === null) return toolError(`subagent: inner session ended without calling ${SUBAGENT_SUBMIT_TOOL_NAME}. The parent should retry with clearer instructions or fail the task.`);
15618
+ return {
15619
+ content: [{
15620
+ type: "text",
15621
+ text: JSON.stringify(captured)
15622
+ }],
15623
+ details: {
15624
+ captured: true,
15625
+ contract: output_schema,
15626
+ callIndex
15627
+ }
15628
+ };
15629
+ }
15630
+ }),
15631
+ getCallCount: () => callCount
15632
+ };
15633
+ }
15634
+ function subagentToolDescription() {
15635
+ return [
15636
+ "Delegate a sub-task to a fresh subagent session with isolated context.",
15637
+ "",
15638
+ "The subagent starts with no conversation history and only the `task` ",
15639
+ "string you provide as its instructions. It runs in the same VM with ",
15640
+ "the same tools you have (Gondolin-routed Read/Write/Edit/Bash, ",
15641
+ "moltnet_* tools), and is expected to call ",
15642
+ `\`${SUBAGENT_SUBMIT_TOOL_NAME}\` with a payload matching the named `,
15643
+ "contract before its session ends.",
15644
+ "",
15645
+ "On success, the tool result is the JSON-stringified subagent payload.",
15646
+ "On failure (unknown contract, validation error, subagent did not ",
15647
+ "submit) the tool returns isError:true with a recoverable message."
15648
+ ].join("\n");
15649
+ }
15650
+ function buildSubagentInstructor(args) {
15651
+ return [
15652
+ "# You are a subagent",
15653
+ "",
15654
+ `Parent task: \`${args.parentTaskId}\` (subagent call #${args.callIndex}).`,
15655
+ "",
15656
+ `Your assigned output contract is \`${args.contractName}\`:`,
15657
+ `${args.contractDescription}`,
15658
+ "",
15659
+ "Rules for this session:",
15660
+ "",
15661
+ `- You MUST call \`${SUBAGENT_SUBMIT_TOOL_NAME}\` exactly once with a `,
15662
+ " payload matching the contract above. Your session terminates on ",
15663
+ " the valid call.",
15664
+ "- The parent's message above is your task. Do not invent additional ",
15665
+ " steps the parent did not request.",
15666
+ "- All MoltNet runtime invariants from the parent runtime instructor ",
15667
+ " apply (diary discipline, gh-auth pattern, etc.) IF you take any ",
15668
+ " action that would trigger them. Most subagents do not commit code ",
15669
+ " or open PRs — only do so if your task message explicitly requires it.",
15670
+ "- You do NOT have access to the `subagent` tool. Do not attempt nested ",
15671
+ " delegation; do the work yourself."
15672
+ ].join("\n");
15673
+ }
15674
+ function toolError(text) {
15675
+ return {
15676
+ content: [{
15677
+ type: "text",
15678
+ text
15679
+ }],
15680
+ details: { captured: false },
15681
+ isError: true
15682
+ };
15683
+ }
15684
+ //#endregion
14887
15685
  //#region ../../libs/pi-extension/src/runtime/task-output.ts
14888
15686
  var METER_NAME = "@themoltnet/pi-extension/task-output";
14889
15687
  var parseResultCounter = null;
@@ -15195,6 +15993,7 @@ async function executePiTask(claimedTask, reporter, opts) {
15195
15993
  const taskTeamId = task.teamId ?? "";
15196
15994
  let reporterOpen = false;
15197
15995
  let session = null;
15996
+ let subagentHandle = null;
15198
15997
  const finalUsage = emptyUsage(opts.provider, opts.model);
15199
15998
  let cancelListener = null;
15200
15999
  const makeFailedOutput = (code, message, usage = finalUsage) => ({
@@ -15312,47 +16111,55 @@ async function executePiTask(claimedTask, reporter, opts) {
15312
16111
  });
15313
16112
  const piAuthDir = process.env.PI_CODING_AGENT_DIR ?? join(homedir(), ".pi", "agent");
15314
16113
  const modelHandle = getModel(opts.provider, opts.model);
15315
- const piOtelExtension = createPiOtelExtension({
15316
- agentName: opts.agentName,
15317
- spanAttributes: {
15318
- "moltnet.task.id": task.id,
15319
- "moltnet.task.attempt": attemptN,
15320
- "moltnet.task.type": task.taskType
15321
- }
15322
- });
15323
- const appendSystemPrompt = [buildRuntimeInstructor({
16114
+ const runtimeInstructor = buildRuntimeInstructor({
15324
16115
  taskId: task.id,
15325
16116
  taskType: task.taskType,
15326
16117
  attemptN,
15327
16118
  diaryId,
15328
16119
  agentName: opts.agentName,
15329
16120
  correlationId: task.correlationId ?? null
15330
- })];
16121
+ });
16122
+ const appendSystemPrompt = [runtimeInstructor];
15331
16123
  if (injectedContext.systemPromptPrefix) appendSystemPrompt.push(injectedContext.systemPromptPrefix);
15332
16124
  const injectedSkills = injectedContext.skills;
15333
- const resourceLoader = new DefaultResourceLoader({
15334
- cwd: mountPath,
15335
- agentDir: piAuthDir,
15336
- extensionFactories: [piOtelExtension],
16125
+ const parentSubagentTools = [];
16126
+ if (taskTypeUsesSubagents(task.taskType)) {
16127
+ subagentHandle = createSubagentTool({
16128
+ mountPath,
16129
+ piAuthDir,
16130
+ modelHandle,
16131
+ agentName: opts.agentName,
16132
+ inheritedCustomTools: [...gondolinCustomTools, ...moltnetTools],
16133
+ parentRuntimeInstructor: runtimeInstructor,
16134
+ parentTaskId: task.id,
16135
+ parentTaskType: task.taskType,
16136
+ parentAttemptN: attemptN,
16137
+ parentCancelSignal: reporter.cancelSignal
16138
+ });
16139
+ parentSubagentTools.push(subagentHandle.tool);
16140
+ }
16141
+ session = await buildAgentSession({
16142
+ mountPath,
16143
+ piAuthDir,
16144
+ modelHandle,
16145
+ agentName: opts.agentName,
16146
+ customTools: [
16147
+ ...gondolinCustomTools,
16148
+ ...moltnetTools,
16149
+ ...submitTools,
16150
+ ...parentSubagentTools
16151
+ ],
15337
16152
  appendSystemPrompt,
15338
16153
  skillsOverride: () => ({
15339
16154
  skills: injectedSkills,
15340
16155
  diagnostics: []
15341
- })
16156
+ }),
16157
+ otelSpanAttrs: {
16158
+ "moltnet.task.id": task.id,
16159
+ "moltnet.task.attempt": attemptN,
16160
+ "moltnet.task.type": task.taskType
16161
+ }
15342
16162
  });
15343
- await resourceLoader.reload();
15344
- session = (await createAgentSession({
15345
- agentDir: piAuthDir,
15346
- cwd: mountPath,
15347
- model: modelHandle,
15348
- customTools: [
15349
- ...gondolinCustomTools,
15350
- ...moltnetTools,
15351
- ...submitTools
15352
- ],
15353
- sessionManager: SessionManager.inMemory(),
15354
- resourceLoader
15355
- })).session;
15356
16163
  } catch (err) {
15357
16164
  const message = err instanceof Error ? err.message : String(err);
15358
16165
  await emit("error", {
@@ -15423,6 +16230,10 @@ async function executePiTask(claimedTask, reporter, opts) {
15423
16230
  phase: "session_prompt"
15424
16231
  });
15425
16232
  }
16233
+ if (subagentHandle && subagentHandle.getCallCount() > 0) await emit("info", {
16234
+ event: "subagent_summary",
16235
+ callCount: subagentHandle.getCallCount()
16236
+ });
15426
16237
  await Promise.all(recordingPromise);
15427
16238
  const cancelled = reporter.cancelSignal.aborted;
15428
16239
  let parsedOutput = null;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@themoltnet/agent-daemon",
3
- "version": "0.5.0",
3
+ "version": "0.5.2",
4
4
  "license": "AGPL-3.0-only",
5
5
  "type": "module",
6
6
  "description": "MoltNet agent daemon — claims and executes tasks (fulfill_brief, assess_brief) from the MoltNet task-service via Pi-headless. CLI: moltnet-agent.",
@@ -33,19 +33,19 @@
33
33
  "@opentelemetry/semantic-conventions": "^1.39.0",
34
34
  "pino": "^10.3.1",
35
35
  "pino-pretty": "^13.1.3",
36
- "@themoltnet/agent-runtime": "0.12.0",
37
- "@themoltnet/pi-extension": "0.14.0",
38
- "@themoltnet/sdk": "0.100.0"
36
+ "@themoltnet/pi-extension": "0.15.1",
37
+ "@themoltnet/agent-runtime": "0.14.0",
38
+ "@themoltnet/sdk": "0.101.0"
39
39
  },
40
40
  "devDependencies": {
41
41
  "tsx": "^4.7.0",
42
42
  "typescript": "^5.3.3",
43
43
  "vite": "^8.0.0",
44
44
  "vitest": "^3.0.0",
45
- "@moltnet/bootstrap": "0.1.0",
46
45
  "@moltnet/crypto-service": "0.1.0",
47
- "@moltnet/tasks": "0.1.0",
48
- "@moltnet/database": "0.1.0"
46
+ "@moltnet/database": "0.1.0",
47
+ "@moltnet/bootstrap": "0.1.0",
48
+ "@moltnet/tasks": "0.1.0"
49
49
  },
50
50
  "nx": {
51
51
  "tags": [