@themoltnet/agent-daemon 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/README.md +7 -0
  2. package/dist/main.js +774 -442
  3. package/package.json +4 -4
package/dist/main.js CHANGED
@@ -1,12 +1,12 @@
1
1
  #!/usr/bin/env node
2
2
  import crypto, { createHash } from "crypto";
3
+ import path, { dirname, isAbsolute, join, relative, resolve } from "node:path";
3
4
  import { parseArgs, parseEnv, promisify } from "node:util";
4
5
  import { existsSync, mkdirSync, readFileSync, readdirSync, rmSync, statSync } from "node:fs";
5
6
  import { ROOT_CONTEXT, SpanStatusCode, context, metrics, propagation, trace } from "@opentelemetry/api";
6
7
  import { pino, transport } from "pino";
7
8
  import { readFile } from "node:fs/promises";
8
9
  import { createHash as createHash$1 } from "node:crypto";
9
- import path, { dirname, isAbsolute, join, relative, resolve } from "node:path";
10
10
  import { homedir } from "node:os";
11
11
  import { execFile, execFileSync } from "node:child_process";
12
12
  import { DefaultResourceLoader, SessionManager, createAgentSession, createBashToolDefinition, createEditToolDefinition, createReadToolDefinition, createSyntheticSourceInfo, createWriteToolDefinition, defineTool, parseFrontmatter } from "@earendil-works/pi-coding-agent";
@@ -2896,6 +2896,7 @@ if (!Has$1("date-time")) Set$1("date-time", (v) => !Number.isNaN(Date.parse(v)))
2896
2896
  */
2897
2897
  var ContextBinding = Type$2.Union([
2898
2898
  Type$2.Literal("skill"),
2899
+ Type$2.Literal("context_inline"),
2899
2900
  Type$2.Literal("prompt_prefix"),
2900
2901
  Type$2.Literal("user_inline")
2901
2902
  ], { $id: "ContextBinding" });
@@ -2912,9 +2913,14 @@ var ContextBinding = Type$2.Union([
2912
2913
  * name under the runtime's skill discovery path. Must be
2913
2914
  * kebab-case-safe (alphanumeric + dashes/underscores).
2914
2915
  * - `binding` — how the bytes are delivered to the LLM (see above).
2915
- * - `content` — the actual bytes (UTF-8 text). Capped at 32 KiB per
2916
+ * - `content` — the actual bytes (UTF-8 text). Capped at 64 KiB per
2916
2917
  * entry; total per-task context bytes are bounded by the
2917
2918
  * soft `maxItems` cap and per-binding daemon limits.
2919
+ * Raised from 32 KiB in 2026-05 — protocol-heavy operator
2920
+ * skills (e.g. `.claude/skills/legreffier/SKILL.md`) ship
2921
+ * at ~35 KiB inline, and the original cap was sized for
2922
+ * short example skills, not the kind of skill the eval
2923
+ * substrate is dogfooded on (#943, #823).
2918
2924
  */
2919
2925
  var ContextRef = Type$2.Object({
2920
2926
  slug: Type$2.String({
@@ -2925,7 +2931,7 @@ var ContextRef = Type$2.Object({
2925
2931
  binding: ContextBinding,
2926
2932
  content: Type$2.String({
2927
2933
  minLength: 1,
2928
- maxLength: 32768
2934
+ maxLength: 65536
2929
2935
  })
2930
2936
  }, {
2931
2937
  $id: "ContextRef",
@@ -4330,61 +4336,33 @@ async function validateJudgePackInputAsync(input, ctx) {
4330
4336
  return errors;
4331
4337
  }
4332
4338
  //#endregion
4333
- //#region ../../libs/tasks/src/task-types/judge-eval-variant.ts
4339
+ //#region ../../libs/tasks/src/task-types/judge-eval-attempt.ts
4334
4340
  /**
4335
- * `judge_eval_variant` — score N variants of a `run_eval` scenario
4336
- * against a single rubric, in one pass, with per-variant subagent
4337
- * isolation.
4341
+ * `judge_eval_attempt` — score one completed `run_eval` attempt against a
4342
+ * hidden judge rubric.
4338
4343
  *
4339
4344
  * output_kind: judgment
4340
- * criteria: required (`successCriteria.rubric` — same envelope shape as
4341
- * `judge_pack` / `assess_brief`)
4342
- * references: not required at the input layer — `runTaskIds` already
4343
- * pin the targets being graded.
4344
- *
4345
- * Slice 2 of #943. The parent task carries the rubric and the list of
4346
- * variant `run_eval` task ids. The pi executor registers the generic
4347
- * `subagent` custom tool (#1087), and the parent LLM calls
4348
- * `subagent({ task, output_schema: 'judge_eval_variant_result' })` once
4349
- * per variant — each child session has fresh context, fetches the
4350
- * variant's accepted attempt output via `moltnet_get_task` /
4351
- * `moltnet_list_task_attempts`, and grades against the rubric.
4345
+ * criteria: required (`successCriteria.rubric`)
4346
+ * references: not required at the input layer — `targetTaskId` +
4347
+ * `targetAttemptN` pin the producer attempt being judged.
4352
4348
  *
4353
- * Reuses `JudgePackScore` from `judge_pack` for per-criterion scoring
4354
- * (Lane 1 binary via `llm_checklist`, Lane 2 graded via `llm_score`,
4355
- * deterministic_*) — the score shape is the same across judgment
4356
- * tasks; only the wrapping (per-variant grouping + deltas) differs.
4357
- *
4358
- * Cross-task input invariants — "all targets share the same
4359
- * correlation_id, all are `run_eval`, all are completed with an
4360
- * accepted attempt, all share byte-identical `input.successCriteria`"
4361
- * — REQUIRE async DB lookups and live in `validateInputAsync` below,
4362
- * which the task service runs at create time (#1096 wiring). The
4363
- * TypeBox layer here only enforces shape: UUID format,
4364
- * minItems/maxItems, rubric presence + weight invariant.
4365
- */
4366
- var JUDGE_EVAL_VARIANT_TYPE = "judge_eval_variant";
4367
- var JudgeEvalVariantInput = Type$2.Object({
4368
- runTaskIds: Type$2.Array(Type$2.String({ format: "uuid" }), {
4369
- minItems: 2,
4370
- maxItems: 10
4371
- }),
4349
+ * This replaces the earlier parent/subagent `judge_eval_variant` design.
4350
+ * The unit of judgment is one producer attempt. Cross-variant deltas can be
4351
+ * computed later at read time from stored scores, rather than materialized as
4352
+ * their own task output.
4353
+ */
4354
+ var JUDGE_EVAL_ATTEMPT_TYPE = "judge_eval_attempt";
4355
+ var JudgeEvalAttemptInput = Type$2.Object({
4356
+ targetTaskId: Type$2.String({ format: "uuid" }),
4357
+ targetAttemptN: Type$2.Integer({ minimum: 1 }),
4372
4358
  successCriteria: SuccessCriteria
4373
4359
  }, {
4374
- $id: "JudgeEvalVariantInput",
4360
+ $id: "JudgeEvalAttemptInput",
4375
4361
  additionalProperties: false
4376
4362
  });
4377
- /**
4378
- * Per-variant grading. `scores[]` shape is identical to `JudgePackScore`
4379
- * (mode-aware: binary via `llm_checklist`, graded via `llm_score`,
4380
- * deterministic_*). Reuse the type rather than re-declare.
4381
- *
4382
- * This is also the **subagent output contract** — the parent's
4383
- * `subagent` tool resolves the contract name `judge_eval_variant_result`
4384
- * to this schema. See `agent-runtime`'s subagent contract registry.
4385
- */
4386
- var JudgeEvalVariantResult = Type$2.Object({
4387
- runTaskId: Type$2.String({ format: "uuid" }),
4363
+ var JudgeEvalAttemptOutput = Type$2.Object({
4364
+ targetTaskId: Type$2.String({ format: "uuid" }),
4365
+ targetAttemptN: Type$2.Integer({ minimum: 1 }),
4388
4366
  variantLabel: Type$2.String({
4389
4367
  minLength: 1,
4390
4368
  maxLength: 64,
@@ -4395,216 +4373,126 @@ var JudgeEvalVariantResult = Type$2.Object({
4395
4373
  minimum: 0,
4396
4374
  maximum: 1
4397
4375
  }),
4398
- verdict: Type$2.String({ minLength: 1 })
4399
- }, {
4400
- $id: "JudgeEvalVariantResult",
4401
- additionalProperties: false
4402
- });
4403
- var JudgeEvalVariantOutput = Type$2.Object({
4404
- results: Type$2.Array(JudgeEvalVariantResult, { minItems: 2 }),
4405
- deltas: Type$2.Optional(Type$2.Record(Type$2.String(), Type$2.Number({
4406
- minimum: -1,
4407
- maximum: 1
4408
- }))),
4376
+ verdict: Type$2.String({ minLength: 1 }),
4409
4377
  judgeModel: Type$2.Optional(Type$2.String({ minLength: 1 })),
4410
4378
  traceparent: Type$2.String({ minLength: 1 })
4411
4379
  }, {
4412
- $id: "JudgeEvalVariantOutput",
4380
+ $id: "JudgeEvalAttemptOutput",
4413
4381
  additionalProperties: false
4414
4382
  });
4415
- /**
4416
- * Synchronous input invariants beyond TypeBox shape: rubric must be
4417
- * present (already required by the schema, but the rubric body has
4418
- * its own per-criterion weight invariant) and the rubric's weights
4419
- * must sum to 1.
4420
- *
4421
- * Cross-task invariants (all targets are `run_eval`, all completed,
4422
- * share `correlation_id`, byte-identical `input.successCriteria`)
4423
- * are NOT checked here — they require async DB lookups against
4424
- * `runTaskIds` and live in `validateJudgeEvalVariantInputAsync`
4425
- * below, invoked by the task service at create time (#1096).
4426
- */
4427
- function validateJudgeEvalVariantInput(input) {
4383
+ function validateJudgeEvalAttemptInput(input) {
4428
4384
  const sc = input.successCriteria;
4429
- if (!sc) return "successCriteria is required for judge_eval_variant";
4430
- if (!sc.rubric) return "successCriteria.rubric is required for judge_eval_variant";
4385
+ if (!sc) return "successCriteria is required for judge_eval_attempt";
4386
+ if (!sc.rubric) return "successCriteria.rubric is required for judge_eval_attempt";
4431
4387
  return validateRubricWeights(sc.rubric);
4432
4388
  }
4433
- /**
4434
- * Output cross-field invariants the schema cannot express:
4435
- *
4436
- * 1. `results.length === input.runTaskIds.length` — every variant
4437
- * the imposer asked for must be graded. Partial grading
4438
- * invalidates cross-variant comparison; fail the whole task
4439
- * rather than silently report a subset.
4440
- *
4441
- * 2. `results[i].runTaskId === input.runTaskIds[i]` — order is
4442
- * load-bearing for downstream consumers (e.g. deltas keyed by
4443
- * adjacent pairs). Mismatch is an LLM bug; reject loudly.
4444
- *
4445
- * 3. Each `result.scores` follows the same `llm_checklist` rule
4446
- * `judge_pack` enforces (#999): if a score has an `assertions`
4447
- * array, the numeric score MUST be `1` iff every assertion
4448
- * passes. Inconsistent payloads pollute attestations.
4449
- *
4450
- * 4. Each `result.composite` MUST equal the rubric-weighted sum
4451
- * `Σ(weight_j × scores[j].score)`. The parent (and any subagent
4452
- * it delegated to) is supposed to compute this; surfacing a
4453
- * drift here catches LLMs that hand-wave the arithmetic.
4454
- *
4455
- * 5. Optional `deltas` keys MUST be of the form `"A - B"` where
4456
- * both `A` and `B` are variantLabels present in `results`.
4457
- * Values are not range-checked (any float in [-1, 1] is
4458
- * arithmetically possible).
4459
- */
4460
- function validateJudgeEvalVariantOutput(output, input) {
4389
+ function validateJudgeEvalAttemptOutput(output, input) {
4461
4390
  const out = output;
4462
4391
  const inp = input;
4463
4392
  if (inp) {
4464
- if (out.results.length !== inp.runTaskIds.length) return `results.length (${out.results.length}) does not match input.runTaskIds.length (${inp.runTaskIds.length}). Every variant must be graded; partial grading is rejected.`;
4465
- for (let i = 0; i < out.results.length; i++) if (out.results[i].runTaskId !== inp.runTaskIds[i]) return `results[${i}].runTaskId (${out.results[i].runTaskId}) does not match input.runTaskIds[${i}] (${inp.runTaskIds[i]}). Order must align with input for downstream delta computation.`;
4466
- }
4467
- for (let r = 0; r < out.results.length; r++) {
4468
- const result = out.results[r];
4469
- for (let s = 0; s < result.scores.length; s++) {
4470
- const sc = result.scores[s];
4471
- if (!sc.assertions) continue;
4472
- const allPassed = sc.assertions.every((a) => a.passed);
4473
- const expected = allPassed ? 1 : 0;
4474
- if (sc.score !== expected) return `results[${r}].scores[${s}] (criterionId="${sc.criterionId}"): assertions ${allPassed ? "all pass" : "have at least one fail"} but score=${sc.score}. Score must be derived: 1 iff every assertion passes, else 0 (#999 llm_checklist rule).`;
4475
- }
4393
+ if (out.targetTaskId !== inp.targetTaskId) return `output.targetTaskId (${out.targetTaskId}) does not match input.targetTaskId (${inp.targetTaskId})`;
4394
+ if (out.targetAttemptN !== inp.targetAttemptN) return `output.targetAttemptN (${out.targetAttemptN}) does not match input.targetAttemptN (${inp.targetAttemptN})`;
4395
+ }
4396
+ for (let s = 0; s < out.scores.length; s++) {
4397
+ const sc = out.scores[s];
4398
+ if (!sc.assertions) continue;
4399
+ const allPassed = sc.assertions.every((a) => a.passed);
4400
+ const expected = allPassed ? 1 : 0;
4401
+ if (sc.score !== expected) return `scores[${s}] (criterionId="${sc.criterionId}"): assertions ${allPassed ? "all pass" : "have at least one fail"} but score=${sc.score}. Score must be 1 iff every assertion passes, else 0.`;
4476
4402
  }
4477
4403
  if (inp?.successCriteria?.rubric) {
4478
4404
  const criteria = inp.successCriteria.rubric.criteria;
4479
4405
  const weightById = new Map(criteria.map((c) => [c.id, c.weight]));
4480
- for (let r = 0; r < out.results.length; r++) {
4481
- const result = out.results[r];
4482
- let sum = 0;
4483
- for (const sc of result.scores) {
4484
- const w = weightById.get(sc.criterionId);
4485
- if (w === void 0) return `results[${r}].scores: criterionId "${sc.criterionId}" is not in the input rubric (known: ${Array.from(weightById.keys()).join(", ")}). Score every rubric criterion exactly once; do not invent new ids.`;
4486
- sum += w * sc.score;
4487
- }
4488
- if (Math.abs(sum - result.composite) > .001) return `results[${r}].composite (${result.composite}) does not match Σ(weight × score) (${sum.toFixed(6)}). Composite must be the rubric-weighted sum of per-criterion scores (drift > 0.001).`;
4489
- }
4490
- }
4491
- if (out.deltas) {
4492
- const labels = new Set(out.results.map((r) => r.variantLabel));
4493
- for (const key of Object.keys(out.deltas)) {
4494
- const m = /^(.+?) - (.+)$/.exec(key);
4495
- if (!m) return `deltas key "${key}" is not of the form "<variantLabel-A> - <variantLabel-B>". Use a single space-hyphen-space separator between labels.`;
4496
- const [, a, b] = m;
4497
- if (!labels.has(a) || !labels.has(b)) return `deltas key "${key}" references variantLabel(s) not present in results: ${!labels.has(a) ? `"${a}" missing` : ""}${!labels.has(a) && !labels.has(b) ? ", " : ""}${!labels.has(b) ? `"${b}" missing` : ""}`;
4406
+ let sum = 0;
4407
+ for (const sc of out.scores) {
4408
+ const w = weightById.get(sc.criterionId);
4409
+ if (w === void 0) return `scores references unknown criterionId "${sc.criterionId}"`;
4410
+ sum += w * sc.score;
4498
4411
  }
4412
+ const rounded = Math.round(sum * 1e3) / 1e3;
4413
+ if (Math.abs(rounded - out.composite) > .001) return `composite (${out.composite}) does not match weighted rubric sum (${rounded})`;
4499
4414
  }
4500
4415
  return null;
4501
4416
  }
4502
- /**
4503
- * Local stable-stringify for cross-variant `successCriteria` byte-
4504
- * equality. Recursively sorts object keys; arrays preserve order
4505
- * (intentional — rubric criteria order is semantically meaningful).
4506
- * Mirrors the canonical-JSON shape `crypto-service` uses for CIDs,
4507
- * without taking on a crypto-service dep just for this comparison.
4508
- */
4509
- function stableStringify(value) {
4510
- if (value === null || typeof value !== "object") return JSON.stringify(value);
4511
- if (Array.isArray(value)) return "[" + value.map(stableStringify).join(",") + "]";
4512
- const obj = value;
4513
- return "{" + Object.keys(obj).sort().map((k) => JSON.stringify(k) + ":" + stableStringify(obj[k])).join(",") + "}";
4514
- }
4515
- /**
4516
- * Async preflight for `judge_eval_variant` (#1096 + #943):
4517
- *
4518
- * 1. Every `runTaskIds[i]` resolves to a task the caller can read.
4519
- * 2. Every resolved task is `taskType === 'run_eval'`.
4520
- * 3. Every resolved task is `status === 'completed'` with a
4521
- * non-null `acceptedAttemptN` — grading an unaccepted attempt
4522
- * races with re-attempts and pollutes the judge attestation.
4523
- * 4. Every resolved task shares a non-null `correlationId`, and all
4524
- * `correlationId`s are equal. Without this an imposer could
4525
- * fabricate a "variant set" by stapling unrelated runs together.
4526
- * 5. The shared `correlationId` is NOT already sealed. A previous
4527
- * judge_eval_variant against the same group is final; produce a
4528
- * fresh correlation_id for a new judging round rather than
4529
- * adding contradictory verdicts to a sealed group.
4530
- * 6. Every variant's `input.successCriteria` is byte-identical (via
4531
- * stable-stringify). Different rubrics across "variants" makes
4532
- * the comparison meaningless.
4533
- */
4534
- async function validateJudgeEvalVariantInputAsync(input, ctx) {
4535
- const { runTaskIds } = input;
4417
+ async function validateJudgeEvalAttemptInputAsync(input, ctx) {
4418
+ const inp = input;
4536
4419
  const errors = [];
4537
- const resolved = await Promise.all(runTaskIds.map((id) => ctx.resolveTask(id)));
4538
- let missingTargets = false;
4539
- const presentTargets = [];
4540
- for (let i = 0; i < runTaskIds.length; i++) {
4541
- const t = resolved[i];
4542
- if (!t) {
4543
- missingTargets = true;
4544
- errors.push({
4545
- field: `runTaskIds[${i}]`,
4546
- message: `runTaskIds[${i}]=${runTaskIds[i]} does not resolve to a task you can read`
4547
- });
4548
- continue;
4549
- }
4550
- presentTargets.push(t);
4551
- if (t.taskType !== "run_eval") errors.push({
4552
- field: `runTaskIds[${i}]`,
4553
- message: `runTaskIds[${i}]=${runTaskIds[i]} is a ${t.taskType}, not a run_eval`
4554
- });
4555
- if (t.status !== "completed" || t.acceptedAttemptN === null) errors.push({
4556
- field: `runTaskIds[${i}]`,
4557
- message: `runTaskIds[${i}]=${runTaskIds[i]} is not completed with an accepted attempt (status=${t.status}, acceptedAttemptN=${t.acceptedAttemptN})`
4558
- });
4559
- }
4560
- if (missingTargets || presentTargets.length === 0) return errors;
4561
- const correlationIds = new Set(presentTargets.map((t) => t.correlationId ?? "__null__"));
4562
- if (correlationIds.has("__null__")) errors.push({
4563
- field: "runTaskIds",
4564
- message: "one or more run_eval targets have no correlation_id; cannot group as variants"
4420
+ const target = await ctx.resolveTask(inp.targetTaskId);
4421
+ if (!target) return [{
4422
+ field: "targetTaskId",
4423
+ message: `targetTaskId=${inp.targetTaskId} does not resolve to a task you can read`
4424
+ }];
4425
+ if (target.taskType !== "run_eval") errors.push({
4426
+ field: "targetTaskId",
4427
+ message: `targetTaskId=${inp.targetTaskId} is a ${target.taskType}, not a run_eval`
4565
4428
  });
4566
- if (correlationIds.size > 1) errors.push({
4567
- field: "runTaskIds",
4568
- message: `run_eval targets span multiple correlation_ids (${Array.from(correlationIds).join(", ")}); variants must share one`
4429
+ if (target.status !== "completed" || target.acceptedAttemptN === null) errors.push({
4430
+ field: "targetTaskId",
4431
+ message: `targetTaskId=${inp.targetTaskId} is not completed with an accepted attempt (status=${target.status}, acceptedAttemptN=${target.acceptedAttemptN})`
4569
4432
  });
4570
- if (errors.length > 0) return errors;
4571
- const correlationId = presentTargets[0].correlationId;
4572
- if (!correlationId) return errors;
4573
- const seal = await ctx.findCorrelationSeal(correlationId);
4574
- if (seal) errors.push({
4575
- field: "runTaskIds",
4576
- message: `correlation_id ${correlationId} is already sealed by ${seal.sealedByTaskType}/${seal.sealedByTaskId} at ${seal.sealedAt}; use a fresh correlation_id for a new judging round`
4433
+ else if (target.acceptedAttemptN !== inp.targetAttemptN) errors.push({
4434
+ field: "targetAttemptN",
4435
+ message: `targetAttemptN=${inp.targetAttemptN} does not match the producer's acceptedAttemptN=${target.acceptedAttemptN}`
4436
+ });
4437
+ if (!target.correlationId) errors.push({
4438
+ field: "targetTaskId",
4439
+ message: "target run_eval has no correlation_id; cannot enforce duplicate-judge protection"
4440
+ });
4441
+ if (errors.length > 0 || !target.correlationId) return errors;
4442
+ const rubric = inp.successCriteria.rubric;
4443
+ const duplicate = (await ctx.listTasksByCorrelation(target.correlationId)).find((task) => {
4444
+ if (task.taskType !== "judge_eval_attempt") return false;
4445
+ if (task.status === "failed" || task.status === "cancelled" || task.status === "expired") return false;
4446
+ const existing = task.input;
4447
+ const existingRubric = existing.successCriteria?.rubric;
4448
+ return existing.targetTaskId === inp.targetTaskId && existing.targetAttemptN === inp.targetAttemptN && existingRubric?.rubricId === rubric?.rubricId && existingRubric?.version === rubric?.version;
4449
+ });
4450
+ if (duplicate) errors.push({
4451
+ field: "targetTaskId",
4452
+ message: `judge task ${duplicate.id} already exists for (${inp.targetTaskId}, attempt ${inp.targetAttemptN}, rubric ${rubric?.rubricId}@${rubric?.version})`
4577
4453
  });
4578
- const first = stableStringify(presentTargets[0].input.successCriteria);
4579
- for (let i = 1; i < presentTargets.length; i++) if (stableStringify(presentTargets[i].input.successCriteria) !== first) {
4580
- errors.push({
4581
- field: `runTaskIds[${i}]`,
4582
- message: `runTaskIds[${i}] has a different input.successCriteria than runTaskIds[0]; all variants must share the rubric and gates`
4583
- });
4584
- break;
4585
- }
4586
4454
  return errors;
4587
4455
  }
4588
- /**
4589
- * Side effect emitted on successful `judge_eval_variant` create:
4590
- * seal the shared correlation_id atomically with the insert. The
4591
- * task service applies the seal in the same transaction; a
4592
- * concurrent second `judge_eval_variant` against the same group
4593
- * loses the race and is rejected with a clean conflict error.
4594
- *
4595
- * The seal applies to the SHARED correlation_id of the targets —
4596
- * NOT to the judge task's own correlationId (which is typically
4597
- * null or distinct). The task service derives the correlationId
4598
- * for the effect from the resolved targets, not from the judge
4599
- * task row.
4600
- */
4601
- async function onCreateJudgeEvalVariant(input, ctx) {
4602
- const { runTaskIds } = input;
4603
- const first = await ctx.resolveTask(runTaskIds[0]);
4604
- if (!first?.correlationId) return [];
4456
+ async function onCreateJudgeEvalAttempt(input, _ctx) {
4457
+ const judge = input;
4458
+ const rubric = judge.successCriteria.rubric;
4459
+ if (!rubric) return [];
4605
4460
  return [{
4606
- kind: "sealCorrelation",
4607
- correlationId: first.correlationId
4461
+ kind: "guardTaskUniqueness",
4462
+ taskType: JUDGE_EVAL_ATTEMPT_TYPE,
4463
+ lockKey: [
4464
+ JUDGE_EVAL_ATTEMPT_TYPE,
4465
+ judge.targetTaskId,
4466
+ String(judge.targetAttemptN),
4467
+ rubric.rubricId,
4468
+ rubric.version
4469
+ ].join(":"),
4470
+ inputMatches: [
4471
+ {
4472
+ path: ["targetTaskId"],
4473
+ value: judge.targetTaskId
4474
+ },
4475
+ {
4476
+ path: ["targetAttemptN"],
4477
+ value: judge.targetAttemptN
4478
+ },
4479
+ {
4480
+ path: [
4481
+ "successCriteria",
4482
+ "rubric",
4483
+ "rubricId"
4484
+ ],
4485
+ value: rubric.rubricId
4486
+ },
4487
+ {
4488
+ path: [
4489
+ "successCriteria",
4490
+ "rubric",
4491
+ "version"
4492
+ ],
4493
+ value: rubric.version
4494
+ }
4495
+ ]
4608
4496
  }];
4609
4497
  }
4610
4498
  //#endregion
@@ -4728,14 +4616,43 @@ async function validateRenderPackInputAsync(input, ctx) {
4728
4616
  //#region ../../libs/tasks/src/task-types/run-eval.ts
4729
4617
  /**
4730
4618
  * `run_eval` — execute a scenario prompt under a named variant for
4731
- * later cross-variant grading by `judge_eval_variant` (Slice 2).
4619
+ * later per-attempt grading by `judge_eval_attempt` tasks.
4732
4620
  *
4733
4621
  * output_kind: artifact
4734
- * criteria: optional (when set, output.verification is required —
4735
- * producer self-assessment; the judge is the binding evaluator)
4622
+ * criteria: optional producer-only checks (when set,
4623
+ * output.verification is required — the judge rubric remains hidden
4624
+ * on downstream `judge_eval_attempt` tasks)
4736
4625
  * references: not required (scenario lives entirely in input)
4737
4626
  */
4738
4627
  var RUN_EVAL_TYPE = "run_eval";
4628
+ var RunEvalMode = Type$2.Union([Type$2.Literal("vitro"), Type$2.Literal("vivo")], { $id: "RunEvalMode" });
4629
+ var RunEvalWorkspace = Type$2.Union([
4630
+ Type$2.Literal("none"),
4631
+ Type$2.Literal("shared_mount"),
4632
+ Type$2.Literal("dedicated_worktree")
4633
+ ], { $id: "RunEvalWorkspace" });
4634
+ var RunEvalExecution = Type$2.Object({
4635
+ mode: RunEvalMode,
4636
+ workspace: RunEvalWorkspace
4637
+ }, {
4638
+ $id: "RunEvalExecution",
4639
+ additionalProperties: false
4640
+ });
4641
+ /**
4642
+ * Producer-visible checks for `run_eval`. Deliberately forbids `rubric`
4643
+ * so the variant runner cannot see the downstream judge's answer key.
4644
+ * Keep the rest of the SuccessCriteria envelope available for generic
4645
+ * process / structure checks (`gates`, `assertions`, `sideEffects`).
4646
+ */
4647
+ var RunEvalSuccessCriteria = Type$2.Object({
4648
+ version: Type$2.Literal(1),
4649
+ gates: Type$2.Optional(SuccessCriteria.properties.gates),
4650
+ assertions: Type$2.Optional(SuccessCriteria.properties.assertions),
4651
+ sideEffects: Type$2.Optional(SuccessCriteria.properties.sideEffects)
4652
+ }, {
4653
+ $id: "RunEvalSuccessCriteria",
4654
+ additionalProperties: false
4655
+ });
4739
4656
  var RunEvalInput = Type$2.Object({
4740
4657
  scenario: Type$2.Object({
4741
4658
  prompt: Type$2.String({ minLength: 1 }),
@@ -4745,8 +4662,9 @@ var RunEvalInput = Type$2.Object({
4745
4662
  minLength: 1,
4746
4663
  maxLength: 64
4747
4664
  }),
4665
+ execution: RunEvalExecution,
4748
4666
  context: TaskContext,
4749
- successCriteria: Type$2.Optional(SuccessCriteria)
4667
+ successCriteria: Type$2.Optional(RunEvalSuccessCriteria)
4750
4668
  }, {
4751
4669
  $id: "RunEvalInput",
4752
4670
  additionalProperties: false
@@ -4774,8 +4692,8 @@ var RunEvalOutput = Type$2.Object({
4774
4692
  function validateRunEvalOutput(output, input) {
4775
4693
  const hasCriteria = input !== null && input !== void 0 && input.successCriteria !== void 0;
4776
4694
  const hasVerification = output !== null && output !== void 0 && output.verification !== void 0;
4777
- if (hasCriteria && !hasVerification) return "output.verification is required because input.successCriteria is set; the producer LLM must self-assess against the criteria";
4778
- if (!hasCriteria && hasVerification) return "output.verification was supplied but input.successCriteria is unset; omit verification when there are no criteria to assess against";
4695
+ if (hasCriteria && !hasVerification) return "output.verification is required because input.successCriteria is set; the producer LLM must self-assess against the producer checks";
4696
+ if (!hasCriteria && hasVerification) return "output.verification was supplied but input.successCriteria is unset; omit verification when there are no producer checks to assess against";
4779
4697
  return null;
4780
4698
  }
4781
4699
  //#endregion
@@ -4891,24 +4809,24 @@ var BUILT_IN_TASK_TYPES = {
4891
4809
  inputSchema: RunEvalInput,
4892
4810
  outputSchema: RunEvalOutput,
4893
4811
  outputKind: "artifact",
4894
- workspaceScope: "attempt",
4812
+ resumable: true,
4813
+ workspaceScope: "session",
4895
4814
  sessionScope: "custom",
4896
4815
  requiresReferences: false,
4897
4816
  validateOutput: validateRunEvalOutput
4898
4817
  },
4899
- [JUDGE_EVAL_VARIANT_TYPE]: {
4900
- name: JUDGE_EVAL_VARIANT_TYPE,
4901
- inputSchema: JudgeEvalVariantInput,
4902
- outputSchema: JudgeEvalVariantOutput,
4818
+ [JUDGE_EVAL_ATTEMPT_TYPE]: {
4819
+ name: JUDGE_EVAL_ATTEMPT_TYPE,
4820
+ inputSchema: JudgeEvalAttemptInput,
4821
+ outputSchema: JudgeEvalAttemptOutput,
4903
4822
  outputKind: "judgment",
4904
4823
  workspaceScope: "attempt",
4905
- sessionScope: "custom",
4824
+ sessionScope: "none",
4906
4825
  requiresReferences: false,
4907
- validateInput: validateJudgeEvalVariantInput,
4908
- validateOutput: validateJudgeEvalVariantOutput,
4909
- validateInputAsync: validateJudgeEvalVariantInputAsync,
4910
- onCreate: onCreateJudgeEvalVariant,
4911
- usesSubagents: true
4826
+ validateInput: validateJudgeEvalAttemptInput,
4827
+ validateOutput: validateJudgeEvalAttemptOutput,
4828
+ validateInputAsync: validateJudgeEvalAttemptInputAsync,
4829
+ onCreate: onCreateJudgeEvalAttempt
4912
4830
  }
4913
4831
  };
4914
4832
  //#endregion
@@ -6258,6 +6176,11 @@ var PROMPT_SEPARATOR = "\n\n---\n\n";
6258
6176
  * - `skill` → `deliver.skill({ slug, content })` once per ref.
6259
6177
  * Slug collisions on distinct contents are
6260
6178
  * refused loudly.
6179
+ * - `context_inline`→ persist raw bytes via `deliver.contextFile(...)`
6180
+ * and inject them into the prompt in an explicit,
6181
+ * named block. Intended for eval/context experiments
6182
+ * where the content must be in the model context
6183
+ * window, not merely discoverable as a skill.
6261
6184
  * - `prompt_prefix` → content appended to `systemPromptPrefix` with
6262
6185
  * the canonical `\n\n---\n\n` separator (in
6263
6186
  * declared order).
@@ -6290,6 +6213,13 @@ async function resolveTaskContext(args) {
6290
6213
  slug: ref.slug,
6291
6214
  content: ref.content
6292
6215
  });
6216
+ } else if (ref.binding === "context_inline") {
6217
+ await args.deliver.contextFile({
6218
+ slug: ref.slug,
6219
+ content: ref.content,
6220
+ suggestedFileName: `${ref.slug}.md`
6221
+ });
6222
+ promptParts.push(formatInlineContextBlock(ref.slug, ref.content));
6293
6223
  } else if (ref.binding === "prompt_prefix") promptParts.push(ref.content);
6294
6224
  else userParts.push(ref.content);
6295
6225
  injected.push(ref);
@@ -6300,6 +6230,23 @@ async function resolveTaskContext(args) {
6300
6230
  userInlineSuffix: userParts.join(PROMPT_SEPARATOR)
6301
6231
  };
6302
6232
  }
6233
+ function formatInlineContextBlock(slug, content) {
6234
+ return [
6235
+ "### Injected Task Context",
6236
+ "",
6237
+ `Context id: \`${slug}\``,
6238
+ "The following raw context was supplied by the task creator. Treat it",
6239
+ "as task-relevant background that may override generic coding instincts",
6240
+ "when it contains repo- or workflow-specific constraints.",
6241
+ "The same content is also materialized in the workspace as",
6242
+ "`/workspace/context-pack.md` and mirrored in `AGENTS.md` for",
6243
+ "repo-context discovery.",
6244
+ "",
6245
+ "<context>",
6246
+ content,
6247
+ "</context>"
6248
+ ].join("\n");
6249
+ }
6303
6250
  //#endregion
6304
6251
  //#region ../../libs/agent-runtime/src/output-tools.ts
6305
6252
  /**
@@ -6365,20 +6312,16 @@ function buildFinalOutputBlock(opts) {
6365
6312
  "## Final output (read this carefully)",
6366
6313
  "",
6367
6314
  `Your VERY LAST action in this conversation MUST report the structured`,
6368
- `output matching \`${outputSchemaName}\`. Two ways to do it, in order of`,
6369
- `preference:`,
6315
+ `output matching \`${outputSchemaName}\`.`,
6370
6316
  "",
6371
- `1. **Preferred — call \`${submitTool}\` exactly once** with the payload.`,
6372
- ` The runtime captures the validated arguments and ends the session.`,
6373
- ` If the tool is registered, prefer this path.`,
6374
- `2. **Fallback** — if the submit tool is unavailable, your very last`,
6375
- ` assistant message MUST be a single JSON object matching`,
6376
- ` \`${outputSchemaName}\`. No prose before or after. No code fences.`,
6377
- ` No "ok" or "done". The runtime parses the last balanced top-level`,
6378
- ` JSON object as the output.`,
6317
+ `Call \`${submitTool}\` exactly once with the payload.`,
6318
+ `The runtime captures the validated arguments and ends the session.`,
6319
+ `Do NOT emit the output as plain assistant text. Do NOT rely on a`,
6320
+ `JSON-in-message fallback. If you do not call \`${submitTool}\`, the`,
6321
+ `attempt fails even if the underlying work succeeded.`,
6379
6322
  "",
6380
- `Failing to report structured output as the very last action means the`,
6381
- `attempt is marked failed even if the underlying work succeeded.`,
6323
+ `Your final assistant text before that tool call may explain your work,`,
6324
+ `but the submit-tool call itself must be your VERY LAST action.`,
6382
6325
  "",
6383
6326
  `Output shape:`,
6384
6327
  "",
@@ -6516,21 +6459,30 @@ function buildAssessBriefUserPrompt(input, ctx) {
6516
6459
  }
6517
6460
  //#endregion
6518
6461
  //#region ../../libs/agent-runtime/src/prompts/self-verification.ts
6519
- function buildSelfVerificationBlock(taskId) {
6462
+ function buildSelfVerificationBlock(taskId, criteriaField = "successCriteria") {
6520
6463
  return [
6521
6464
  "## Self-verification",
6522
6465
  "",
6523
- `Call \`moltnet_get_task\` with task id \`${taskId}\` and read \`input.successCriteria\`.`,
6466
+ `If \`input.${criteriaField}\` is set on this task, your final output MUST`,
6467
+ "include a `verification` block. **The runtime/server rejects task",
6468
+ `submission without \`verification\` when \`${criteriaField}\` is present**`,
6469
+ "— the request fails validation and the attempt is discarded, even if the",
6470
+ "underlying work succeeded. Do not call the submit tool until you have",
6471
+ "computed the verification payload.",
6524
6472
  "",
6525
- "- If `input.successCriteria` is **absent**, omit `verification` from your",
6473
+ `Call \`moltnet_get_task\` with task id \`${taskId}\` and read \`input.${criteriaField}\`.`,
6474
+ "",
6475
+ `- If \`input.${criteriaField}\` is **absent**, omit \`verification\` from your`,
6526
6476
  " final output entirely.",
6527
- "- If `input.successCriteria` is **present**, you MUST include a",
6528
- " `verification` block in your final output. Evaluate every applicable",
6477
+ `- If \`input.${criteriaField}\` is **present**, evaluate every applicable`,
6529
6478
  " item — `gates`, `assertions`, `rubric` criteria, `sideEffects` — against",
6530
6479
  " your produced work and emit one result per id. Be honest: a `fail` with",
6531
6480
  " a one-line reason is more useful than a false `pass`. Use `skip` (with a",
6532
6481
  " `detail`) when you genuinely could not determine a result. Compute",
6533
6482
  " `passed = results.every(r => r.status !== 'fail')`.",
6483
+ "- `verification` MUST be a JSON object. Never send a string, markdown",
6484
+ " block, null, or an empty placeholder. The submit tool expects an object",
6485
+ " with `inputCid`, `results`, and `passed` fields.",
6534
6486
  "",
6535
6487
  "Verification shape:",
6536
6488
  "",
@@ -6544,6 +6496,23 @@ function buildSelfVerificationBlock(taskId) {
6544
6496
  " \"passed\": <boolean>",
6545
6497
  "}",
6546
6498
  "```",
6499
+ "",
6500
+ "Minimal valid example:",
6501
+ "",
6502
+ "```json",
6503
+ "{",
6504
+ " \"inputCid\": \"<task inputCid>\",",
6505
+ " \"results\": [",
6506
+ " {",
6507
+ " \"id\": \"<criterion id>\",",
6508
+ " \"kind\": \"rubric\",",
6509
+ " \"status\": \"pass\",",
6510
+ " \"detail\": \"one-line reason\"",
6511
+ " }",
6512
+ " ],",
6513
+ " \"passed\": true",
6514
+ "}",
6515
+ "```",
6547
6516
  ""
6548
6517
  ].join("\n");
6549
6518
  }
@@ -6794,69 +6763,62 @@ function buildFulfillBriefUserPrompt(input, ctx) {
6794
6763
  ].filter(Boolean).join("\n");
6795
6764
  }
6796
6765
  //#endregion
6797
- //#region ../../libs/agent-runtime/src/prompts/judge-eval-variant.ts
6798
- /**
6799
- * Build the first user-message prompt for a `judge_eval_variant` task
6800
- * (#943 Slice 2).
6801
- *
6802
- * The parent agent's job is **fan-out-and-collect**: for each
6803
- * `runTaskIds[i]`, spawn an isolated subagent via the `subagent` custom
6804
- * tool (#1087), have it grade that variant against the shared rubric,
6805
- * and collect each subagent's structured `judge_eval_variant_result`
6806
- * payload. The parent does NOT grade itself; it composes the per-
6807
- * variant results into the final `judge_eval_variant` output (results
6808
- * array + optional deltas + verdicts).
6809
- *
6810
- * Isolation is the point: each variant gets a fresh subagent session
6811
- * with no carryover context from sibling variants, so per-variant
6812
- * grading is independent. Cost is bounded by `maxItems: 10` on
6813
- * runTaskIds.
6814
- */
6815
- function buildJudgeEvalVariantUserPrompt(input, ctx) {
6816
- const { runTaskIds, successCriteria } = input;
6817
- const rubric = successCriteria.rubric;
6818
- if (!rubric) throw new Error("judge_eval_variant requires successCriteria.rubric — none present");
6766
+ //#region ../../libs/agent-runtime/src/prompts/judge-eval-attempt.ts
6767
+ function buildJudgeEvalAttemptUserPrompt(input, ctx) {
6768
+ const rubric = input.successCriteria.rubric;
6769
+ if (!rubric) throw new Error("judge_eval_attempt requires successCriteria.rubric — none present");
6819
6770
  const escapeCell = (s) => s.replace(/\\/g, "\\\\").replace(/\|/g, "\\|").replace(/\r?\n/g, " ");
6820
6771
  const criteriaTable = rubric.criteria.map((c) => `| \`${c.id}\` | ${c.weight.toFixed(3)} | ${c.scoring} | ${escapeCell(c.description)} |`).join("\n");
6821
- const targetsBlock = runTaskIds.map((id, i) => `${i + 1}. \`${id}\``).join("\n");
6822
6772
  const finalOutputBlock = buildFinalOutputBlock({
6823
- taskType: "judge_eval_variant",
6824
- outputSchemaName: "JudgeEvalVariantOutput",
6773
+ taskType: "judge_eval_attempt",
6774
+ outputSchemaName: "JudgeEvalAttemptOutput",
6825
6775
  shapeSketch: [
6826
6776
  "{",
6827
- " \"results\": [",
6828
- " {",
6829
- " \"runTaskId\": \"<runTaskIds[i]>\",",
6830
- " \"variantLabel\": \"<from variant input>\",",
6831
- " \"scores\": [ { \"criterionId\": \"...\", \"score\": 0..1, \"rationale\": \"...\", \"assertions\": [...]? } ],",
6832
- " \"composite\": <Σ(weight × score), 0..1>,",
6833
- " \"verdict\": \"<1-3 sentences>\"",
6834
- " },",
6835
- " ...one entry per runTaskIds[i], same order",
6836
- " ],",
6837
- " \"deltas\": { \"<labelA> - <labelB>\": <composite(A) - composite(B)> }, // optional",
6777
+ ` "targetTaskId": "${input.targetTaskId}",`,
6778
+ ` "targetAttemptN": ${input.targetAttemptN},`,
6779
+ " \"variantLabel\": \"<from producer input>\",",
6780
+ " \"scores\": [ { \"criterionId\": \"...\", \"score\": 0..1, \"rationale\": \"...\", \"assertions\": [...]? } ],",
6781
+ " \"composite\": <Σ(weight × score), 0..1>,",
6782
+ " \"verdict\": \"<1-3 sentences>\",",
6838
6783
  " \"judgeModel\": \"<id>\", // optional",
6839
6784
  " \"traceparent\": \"<from claim>\"",
6840
6785
  "}"
6841
6786
  ].join("\n")
6842
6787
  });
6788
+ const workspaceSection = ctx.workspace?.attached === true ? [
6789
+ "### Workspace",
6790
+ "",
6791
+ "Your current workspace is already attached to the producer attempt",
6792
+ "you are judging. Inspect files directly from the current workspace",
6793
+ "root instead of inventing synthetic `artifact_<taskId>` paths.",
6794
+ "If the accepted attempt output lists `artifacts[].path`, treat those",
6795
+ "paths as relative to the current workspace root unless the output",
6796
+ "explicitly says otherwise.",
6797
+ ctx.workspace.mode === "dedicated_worktree" ? `This attachment is a dedicated producer worktree${ctx.workspace.branch ? ` on branch \`${ctx.workspace.branch}\`` : ""}.` : ctx.workspace.mode === "scratch_mount" ? "This attachment is the producer scratch workspace mounted with shadow writes for safe inspection." : "This attachment is the producer shared workspace mounted with shadow writes for safe inspection.",
6798
+ ""
6799
+ ].join("\n") : "";
6843
6800
  return [
6844
- "# Judge Eval Variants\n",
6845
- `You are grading ${runTaskIds.length} variants of a single run_eval scenario`,
6846
- "against ONE shared rubric. Your job is fan-out-and-collect — you do not",
6847
- "grade yourself.",
6801
+ "# Judge Eval Attempt\n",
6802
+ "You are grading one accepted `run_eval` producer attempt against a hidden",
6803
+ "judge rubric. Do not delegate to subagents. Grade in this session only.",
6848
6804
  "",
6849
6805
  `Task id: \`${ctx.taskId}\``,
6850
6806
  `Diary: \`${ctx.diaryId}\``,
6807
+ `Producer task: \`${input.targetTaskId}\``,
6808
+ `Producer attempt: \`${input.targetAttemptN}\``,
6851
6809
  "",
6852
- "### Targets (variants to grade)",
6853
- "",
6854
- targetsBlock,
6810
+ "### Evidence gathering",
6855
6811
  "",
6856
- "Each target is a completed `run_eval` task in the same correlation group.",
6857
- "Read its accepted attempt via `moltnet_get_task` / `moltnet_list_task_attempts`",
6858
- "to see the producer's output before grading.",
6812
+ `1. Call \`moltnet_get_task\` with taskId=\`${input.targetTaskId}\`.`,
6813
+ `2. Call \`moltnet_list_task_attempts\` with taskId=\`${input.targetTaskId}\` and inspect the accepted attempt matching \`${input.targetAttemptN}\`.`,
6814
+ `3. Call \`moltnet_list_task_messages\` with taskId=\`${input.targetTaskId}\`, attemptN=\`${input.targetAttemptN}\` to inspect the producer's turn-by-turn behavior.`,
6815
+ "4. Use the accepted attempt output, attempt messages, and any accessible",
6816
+ " artifacts or workspace evidence available in your environment.",
6817
+ " Read artifact files from the mounted producer workspace when present;",
6818
+ " do not assume detached `artifact_<taskId>` directories exist.",
6819
+ "5. Score strictly against the rubric below.",
6859
6820
  "",
6821
+ workspaceSection,
6860
6822
  "### Rubric",
6861
6823
  "",
6862
6824
  rubric.preamble ? `${rubric.preamble}\n` : "",
@@ -6864,34 +6826,10 @@ function buildJudgeEvalVariantUserPrompt(input, ctx) {
6864
6826
  "| --- | --- | --- | --- |",
6865
6827
  criteriaTable,
6866
6828
  "",
6867
- "### How to grade",
6868
- "",
6869
- "For EACH `runTaskIds[i]`:",
6870
- "",
6871
- "1. Call the `subagent` custom tool with:",
6872
- " - `task`: a brief instructing the subagent to grade ONLY that variant",
6873
- " against the rubric above; include the target task id and the rubric",
6874
- " verbatim. The subagent has the same MoltNet tools and can fetch the",
6875
- " accepted attempt output independently.",
6876
- " - `output_schema`: `\"judge_eval_variant_result\"`",
6877
- "2. Receive the subagent's structured `judge_eval_variant_result` payload.",
6878
- "3. Append it to your `results[]` array, **in the same order as input.runTaskIds**.",
6879
- "",
6880
- "Do NOT score any variant in your own session. The whole point of the",
6881
- "subagent fan-out is per-variant context isolation — grading two variants",
6882
- "back-to-back in one session lets the second be biased by the first.",
6883
- "",
6884
6829
  "### Composite arithmetic",
6885
6830
  "",
6886
- "Each `composite` MUST equal `Σ(criterion.weight × score)` over the rubric",
6887
- "criteria. Drift > 0.001 is rejected. Subagents are instructed to compute it",
6888
- "themselves; double-check before assembling the final output.",
6889
- "",
6890
- "### Deltas (optional)",
6891
- "",
6892
- "If useful, populate `deltas` with pairwise composite differences keyed by",
6893
- "`\"<variantLabel-A> - <variantLabel-B>\"` (single space-hyphen-space). Both",
6894
- "labels must appear in `results`. Omit `deltas` entirely if not used.",
6831
+ "Your `composite` MUST equal `Σ(criterion.weight × score)` over the rubric",
6832
+ "criteria. Drift > 0.001 is rejected.",
6895
6833
  "",
6896
6834
  finalOutputBlock
6897
6835
  ].filter((s) => s !== "").join("\n");
@@ -7188,8 +7126,9 @@ function buildRenderPackUserPrompt(input, ctx) {
7188
7126
  * Build the first user-message prompt for a `run_eval` task.
7189
7127
  *
7190
7128
  * Free-form: no git workflow, no commit ceremony. The executor produces
7191
- * a textual response (and optional file artifacts) that a later
7192
- * `judge_eval_variant` task (Slice 2) grades against the rubric.
7129
+ * a textual response (and optional file artifacts) that later
7130
+ * `judge_eval_attempt` task(s) grade against their own hidden
7131
+ * rubric.
7193
7132
  *
7194
7133
  * Context delivery is handled by `resolveTaskContext` (see
7195
7134
  * libs/agent-runtime/src/context-bindings.ts) and runs BEFORE this
@@ -7199,7 +7138,9 @@ function buildRenderPackUserPrompt(input, ctx) {
7199
7138
  * builder does NOT inline `input.context[]` itself.
7200
7139
  */
7201
7140
  function buildRunEvalUserPrompt(input, ctx) {
7202
- const { scenario, variantLabel, successCriteria } = input;
7141
+ const { scenario, variantLabel, execution, successCriteria } = input;
7142
+ const hasContext = input.context.length > 0;
7143
+ const hasInlineContext = input.context.some((entry) => entry.binding === "context_inline");
7203
7144
  const inputFilesSection = scenario.inputFiles?.length ? [
7204
7145
  "### Input files",
7205
7146
  "",
@@ -7212,9 +7153,30 @@ function buildRunEvalUserPrompt(input, ctx) {
7212
7153
  "",
7213
7154
  `This task carries correlationId \`${ctx.correlationId}\`. It joins`,
7214
7155
  "this variant to its sibling `run_eval` tasks (other variants of the",
7215
- "same scenario) and to the eventual `judge_eval_variant` task that",
7216
- "will grade them together. You do not need to act on it directly —",
7217
- "it is recorded for cross-variant aggregation at query time.",
7156
+ "same scenario and to any later `judge_eval_attempt` tasks created",
7157
+ "against those variants. You do not need to act on it directly — it",
7158
+ "is recorded for cross-variant aggregation at query time.",
7159
+ ""
7160
+ ].join("\n") : "";
7161
+ const executionSection = [
7162
+ "### Execution mode",
7163
+ "",
7164
+ `Mode: \`${execution.mode}\``,
7165
+ `Workspace: \`${execution.workspace}\``,
7166
+ execution.workspace === "none" ? "You are running in a scratch workspace with no repository checkout mounted. Do not assume git history or repo files are present unless the scenario provided them explicitly." : execution.workspace === "shared_mount" ? "You are running against the daemon shared mount. Treat any repository mutations as affecting the mounted checkout directly." : "You are running in a dedicated disposable git worktree isolated from the daemon shared checkout.",
7167
+ ""
7168
+ ].join("\n");
7169
+ const contextDisciplineSection = hasContext ? [
7170
+ "### Injected context discipline",
7171
+ "",
7172
+ "This task includes extra injected context from the task creator.",
7173
+ "You MUST inspect and use that context BEFORE you write solution",
7174
+ "files or draft your final answer.",
7175
+ "Do not solve first and only review the context afterward.",
7176
+ hasInlineContext ? "For `context_inline`, your FIRST content-inspection step should be a `read` of `/workspace/context-pack.md` before your first `write` call. The same content is also mirrored in `/workspace/AGENTS.md` and may be referenced from `/workspace/.claude/CLAUDE.md`." : "If injected context was provided as a skill, inspect that task-injected context before solving.",
7177
+ hasInlineContext ? "If `/workspace/context-pack.md` exists and you skip reading it before writing solution files, you are not following the task instructions." : "Do not rely on memory alone when task-injected context is available; inspect it first.",
7178
+ "If the injected context contains repo- or workflow-specific rules,",
7179
+ "those rules override your generic instincts.",
7218
7180
  ""
7219
7181
  ].join("\n") : "";
7220
7182
  const finalOutputBlock = buildFinalOutputBlock({
@@ -7227,7 +7189,13 @@ function buildRunEvalUserPrompt(input, ctx) {
7227
7189
  " \"totalTokens\": <int>,",
7228
7190
  " \"durationMs\": <int>,",
7229
7191
  " \"traceparent\": \"<from claim>\",",
7230
- " \"verification\": <required iff input.successCriteria; see Self-verification>",
7192
+ " \"verification\": {",
7193
+ " \"inputCid\": \"<task inputCid>\",",
7194
+ " \"results\": [",
7195
+ " { \"id\": \"<criterion id>\", \"kind\": \"rubric\", \"status\": \"pass|fail|skip\", \"detail\": \"<optional one-liner>\" }",
7196
+ " ],",
7197
+ " \"passed\": <boolean>",
7198
+ " } // required iff input.successCriteria; must be an object, never a string",
7231
7199
  "}"
7232
7200
  ].join("\n")
7233
7201
  });
@@ -7235,6 +7203,8 @@ function buildRunEvalUserPrompt(input, ctx) {
7235
7203
  "# Run Eval Agent\n",
7236
7204
  `You are running an evaluation scenario as variant \`${variantLabel}\`.\nTask id: \`${ctx.taskId}\`\n`,
7237
7205
  correlationSection,
7206
+ executionSection,
7207
+ contextDisciplineSection,
7238
7208
  `### Scenario\n\n${scenario.prompt}\n`,
7239
7209
  inputFilesSection,
7240
7210
  verificationSection,
@@ -7306,6 +7276,16 @@ function buildTaskUserPrompt(task, ctx) {
7306
7276
  diaryId: ctx.diaryId,
7307
7277
  taskId: ctx.taskId
7308
7278
  });
7279
+ case JUDGE_EVAL_ATTEMPT_TYPE:
7280
+ if (!Check(JudgeEvalAttemptInput, task.input)) {
7281
+ const errors = [...Errors(JudgeEvalAttemptInput, task.input)];
7282
+ throw new Error(`judge_eval_attempt input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
7283
+ }
7284
+ return buildJudgeEvalAttemptUserPrompt(task.input, {
7285
+ diaryId: ctx.diaryId,
7286
+ taskId: ctx.taskId,
7287
+ workspace: ctx.workspace
7288
+ });
7309
7289
  case PR_REVIEW_TYPE:
7310
7290
  if (!Check(PrReviewInput, task.input)) {
7311
7291
  const errors = [...Errors(PrReviewInput, task.input)];
@@ -7316,15 +7296,6 @@ function buildTaskUserPrompt(task, ctx) {
7316
7296
  taskId: ctx.taskId,
7317
7297
  workspace: ctx.workspace
7318
7298
  });
7319
- case JUDGE_EVAL_VARIANT_TYPE:
7320
- if (!Check(JudgeEvalVariantInput, task.input)) {
7321
- const errors = [...Errors(JudgeEvalVariantInput, task.input)];
7322
- throw new Error(`judge_eval_variant input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
7323
- }
7324
- return buildJudgeEvalVariantUserPrompt(task.input, {
7325
- diaryId: ctx.diaryId,
7326
- taskId: ctx.taskId
7327
- });
7328
7299
  case RUN_EVAL_TYPE:
7329
7300
  if (!Check(RunEvalInput, task.input)) {
7330
7301
  const errors = [...Errors(RunEvalInput, task.input)];
@@ -10077,12 +10048,20 @@ var MoltNetError = class extends Error {
10077
10048
  code;
10078
10049
  statusCode;
10079
10050
  detail;
10051
+ /**
10052
+ * Populated when the server returned a `VALIDATION_FAILED` problem
10053
+ * (status 400) with field-level errors. Empty / undefined for every
10054
+ * other problem kind. Imposer scripts surface these to operators so
10055
+ * they don't have to re-run with curl to see what was rejected.
10056
+ */
10057
+ validationErrors;
10080
10058
  constructor(message, options) {
10081
10059
  super(message);
10082
10060
  this.name = "MoltNetError";
10083
10061
  this.code = options.code;
10084
10062
  this.statusCode = options.statusCode;
10085
10063
  this.detail = options.detail;
10064
+ this.validationErrors = options.validationErrors;
10086
10065
  }
10087
10066
  };
10088
10067
  var NetworkError = class extends MoltNetError {
@@ -10106,10 +10085,14 @@ var AuthenticationError = class extends MoltNetError {
10106
10085
  };
10107
10086
  function problemToError(problem, statusCode) {
10108
10087
  const title = problem.title ?? "Request failed";
10109
- return new MoltNetError(problem.detail ? `${title}: ${problem.detail}` : title, {
10088
+ const message = problem.detail ? `${title}: ${problem.detail}` : title;
10089
+ const rawErrors = problem.errors;
10090
+ const validationErrors = Array.isArray(rawErrors) ? rawErrors.filter((e) => typeof e === "object" && e !== null && typeof e.field === "string" && typeof e.message === "string") : void 0;
10091
+ return new MoltNetError(message, {
10110
10092
  code: problem.type ?? problem.code ?? "UNKNOWN",
10111
10093
  statusCode,
10112
- detail: problem.detail
10094
+ detail: problem.detail,
10095
+ validationErrors
10113
10096
  });
10114
10097
  }
10115
10098
  //#endregion
@@ -14034,31 +14017,6 @@ function abortableSleep(ms, signal) {
14034
14017
  });
14035
14018
  }
14036
14019
  //#endregion
14037
- //#region ../../libs/agent-runtime/src/subagent-output-contracts.ts
14038
- /**
14039
- * Construct an immutable contract registry from a static list.
14040
- *
14041
- * The resulting registry is safe to share across sessions and
14042
- * invocations — no mutation is possible after construction.
14043
- */
14044
- function createSubagentContractRegistry(contracts) {
14045
- const lookup = /* @__PURE__ */ new Map();
14046
- for (const c of contracts) {
14047
- if (!c.name || c.name.trim().length === 0) throw new Error("subagent output contract name is required");
14048
- if (!/^[a-z][a-z0-9_]*$/.test(c.name)) throw new Error(`subagent output contract name '${c.name}' must be lower_snake_case (starts with a letter, then [a-z0-9_]+)`);
14049
- if (lookup.has(c.name)) throw new Error(`duplicate subagent output contract name '${c.name}' in constructor args`);
14050
- lookup.set(c.name, c);
14051
- }
14052
- return {
14053
- get(name) {
14054
- return lookup.get(name) ?? null;
14055
- },
14056
- list() {
14057
- return [...lookup.values()];
14058
- }
14059
- };
14060
- }
14061
- //#endregion
14062
14020
  //#region ../../libs/pi-extension/src/moltnet/render-phase6.ts
14063
14021
  function slugToTitle(value) {
14064
14022
  return value.split(/[:/_-]+/).filter(Boolean).map((part) => part[0]?.toUpperCase() + part.slice(1)).join(" ");
@@ -14669,6 +14627,41 @@ function createMoltNetTools(config) {
14669
14627
  };
14670
14628
  }
14671
14629
  });
14630
+ const listTaskMessages = defineTool({
14631
+ name: "moltnet_list_task_messages",
14632
+ label: "List MoltNet Task Attempt Messages",
14633
+ description: "List messages for a specific task attempt. Use this when you need the turn-by-turn execution record behind an accepted attempt — tool calls, text deltas, and error/info events that do not appear in the attempt output alone.",
14634
+ parameters: Type.Object({
14635
+ taskId: Type.String({ description: "Task ID (UUID)." }),
14636
+ attemptN: Type.Integer({
14637
+ minimum: 1,
14638
+ description: "Attempt number to inspect."
14639
+ }),
14640
+ afterSeq: Type.Optional(Type.Integer({
14641
+ minimum: 0,
14642
+ description: "Optional cursor: only return messages with seq > afterSeq."
14643
+ })),
14644
+ limit: Type.Optional(Type.Integer({
14645
+ minimum: 1,
14646
+ maximum: 500,
14647
+ description: "Optional maximum messages to return. Defaults to the API value."
14648
+ }))
14649
+ }),
14650
+ async execute(_id, params) {
14651
+ const { agent } = ensureConnected(config);
14652
+ const messages = await agent.tasks.listMessages(params.taskId, params.attemptN, {
14653
+ afterSeq: params.afterSeq,
14654
+ limit: params.limit
14655
+ });
14656
+ return {
14657
+ content: [{
14658
+ type: "text",
14659
+ text: JSON.stringify(messages, null, 2)
14660
+ }],
14661
+ details: {}
14662
+ };
14663
+ }
14664
+ });
14672
14665
  const reviewSessionErrors = defineTool({
14673
14666
  name: "moltnet_review_session_errors",
14674
14667
  label: "Review Session Tool Errors",
@@ -14717,6 +14710,7 @@ function createMoltNetTools(config) {
14717
14710
  createEntry,
14718
14711
  getTask,
14719
14712
  listTaskAttempts,
14713
+ listTaskMessages,
14720
14714
  reviewSessionErrors,
14721
14715
  defineTool({
14722
14716
  name: "moltnet_host_exec",
@@ -15015,6 +15009,12 @@ var GUEST_WORKSPACE$1 = "/workspace";
15015
15009
  * investigation and the alternatives we rejected.
15016
15010
  */
15017
15011
  var GUEST_TASK_SKILLS_MOUNT = "/moltnet-task-skills";
15012
+ function shouldRunResumeCommand(entry, ctx) {
15013
+ if (typeof entry === "string") return true;
15014
+ const workspaceModes = entry.when?.workspaceMode;
15015
+ if (workspaceModes && !workspaceModes.includes(ctx.workspaceMode)) return false;
15016
+ return true;
15017
+ }
15018
15018
  /**
15019
15019
  * Resolve the main worktree root (where .moltnet/ lives — it's untracked,
15020
15020
  * only exists in the main worktree, not in git worktrees).
@@ -15160,6 +15160,7 @@ async function resumeVm(config) {
15160
15160
  ...envOverrides
15161
15161
  };
15162
15162
  const resources = config.sandboxConfig?.resources;
15163
+ const workspaceMode = config.workspaceMode ?? "shared_mount";
15163
15164
  const vm = await VmCheckpoint.load(config.checkpointPath).resume({
15164
15165
  httpHooks,
15165
15166
  env: vmEnv,
@@ -15178,7 +15179,32 @@ async function resumeVm(config) {
15178
15179
  '`);
15179
15180
  await vmRun(vm, "DNS resolvers", `printf 'nameserver 8.8.8.8\\nnameserver 1.1.1.1\\n' > /etc/resolv.conf`);
15180
15181
  await vmRun(vm, "git safe.directory", `git config --system --add safe.directory '*'`);
15181
- for (const [i, cmd] of (config.sandboxConfig?.resumeCommands ?? []).entries()) await vmRun(vm, `resumeCommands[${i}]`, cmd);
15182
+ for (const [i, entry] of (config.sandboxConfig?.resumeCommands ?? []).entries()) {
15183
+ if (!shouldRunResumeCommand(entry, { workspaceMode })) continue;
15184
+ const { run, retries, backoffMs } = typeof entry === "string" ? {
15185
+ run: entry,
15186
+ retries: 0,
15187
+ backoffMs: 2e3
15188
+ } : {
15189
+ run: entry.run,
15190
+ retries: entry.retries ?? 0,
15191
+ backoffMs: entry.retryBackoffMs ?? 2e3
15192
+ };
15193
+ const label = `resumeCommands[${i}]`;
15194
+ let lastErr;
15195
+ for (let attempt = 0; attempt <= retries; attempt++) try {
15196
+ await vmRun(vm, label, run);
15197
+ lastErr = void 0;
15198
+ break;
15199
+ } catch (err) {
15200
+ lastErr = err;
15201
+ if (attempt === retries) break;
15202
+ await new Promise((resolve) => {
15203
+ setTimeout(resolve, (attempt + 1) * backoffMs);
15204
+ });
15205
+ }
15206
+ if (lastErr) throw lastErr instanceof Error ? lastErr : new Error(String(lastErr));
15207
+ }
15182
15208
  const vmSshDir = `${vmAgentDir}/ssh`;
15183
15209
  await vm.exec(`mkdir -p ${vmAgentDir}/ssh /home/agent/.pi/agent`);
15184
15210
  if (creds.piAuthJson !== null) await vm.fs.writeFile("/home/agent/.pi/agent/auth.json", creds.piAuthJson, { mode: 384 });
@@ -15557,7 +15583,8 @@ async function buildAgentSession(args) {
15557
15583
  await resourceLoader.reload();
15558
15584
  const sessionManager = args.sessionPersistence ? await resolvePersistentSessionManager({
15559
15585
  cwd: args.cwdPath,
15560
- sessionDir: args.sessionPersistence.sessionDir
15586
+ sessionDir: args.sessionPersistence.sessionDir,
15587
+ forkFromSessionPath: args.sessionPersistence.forkFromSessionPath
15561
15588
  }) : SessionManager.inMemory(args.cwdPath);
15562
15589
  return (await createAgentSession({
15563
15590
  agentDir: args.piAuthDir,
@@ -15569,6 +15596,7 @@ async function buildAgentSession(args) {
15569
15596
  })).session;
15570
15597
  }
15571
15598
  async function resolvePersistentSessionManager(args) {
15599
+ if (args.forkFromSessionPath) return SessionManager.forkFrom(args.forkFromSessionPath, args.cwd, args.sessionDir);
15572
15600
  await SessionManager.list(args.cwd, args.sessionDir);
15573
15601
  return SessionManager.continueRecent(args.cwd, args.sessionDir);
15574
15602
  }
@@ -15609,6 +15637,11 @@ async function resolvePersistentSessionManager(args) {
15609
15637
  * paths under this mount via `toGuestPath` in `tool-operations.ts`.
15610
15638
  */
15611
15639
  var SKILL_ROOT_IN_VM = GUEST_TASK_SKILLS_MOUNT;
15640
+ var INLINE_CONTEXT_ROOT_IN_VM = "/workspace/.moltnet/context";
15641
+ var WORKSPACE_CONTEXT_PACK = "/workspace/context-pack.md";
15642
+ var WORKSPACE_AGENTS_MD = "/workspace/AGENTS.md";
15643
+ var WORKSPACE_CLAUDE_DIR = "/workspace/.claude";
15644
+ var WORKSPACE_CLAUDE_MD = "/workspace/.claude/CLAUDE.md";
15612
15645
  /** Bounds borrowed from pi's skill validation; conservative caps so a
15613
15646
  * malformed SKILL.md doesn't bloat the system prompt. */
15614
15647
  var MAX_SKILL_NAME = 64;
@@ -15619,21 +15652,40 @@ var MAX_SKILL_DESCRIPTION = 1024;
15619
15652
  */
15620
15653
  async function injectTaskContext(args) {
15621
15654
  const skills = [];
15655
+ const inlineContexts = [];
15622
15656
  const resolved = await resolveTaskContext({
15623
15657
  context: args.context,
15624
- deliver: { skill: async ({ slug, content }) => {
15625
- const dir = `${SKILL_ROOT_IN_VM}/${slug}`;
15626
- const filePath = `${dir}/SKILL.md`;
15627
- await args.fs.mkdir(dir, { recursive: true });
15628
- await args.fs.writeFile(filePath, content, { mode: 420 });
15629
- skills.push(buildSyntheticSkill({
15630
- slug,
15631
- content,
15632
- filePath,
15633
- dir
15634
- }));
15635
- } }
15658
+ deliver: {
15659
+ skill: async ({ slug, content }) => {
15660
+ const dir = `${SKILL_ROOT_IN_VM}/${slug}`;
15661
+ const filePath = `${dir}/SKILL.md`;
15662
+ await args.fs.mkdir(dir, { recursive: true });
15663
+ await args.fs.writeFile(filePath, content, { mode: 420 });
15664
+ skills.push(buildSyntheticSkill({
15665
+ slug,
15666
+ content,
15667
+ filePath,
15668
+ dir
15669
+ }));
15670
+ },
15671
+ contextFile: async ({ suggestedFileName, content }) => {
15672
+ await args.fs.mkdir(INLINE_CONTEXT_ROOT_IN_VM, { recursive: true });
15673
+ const filePath = `${INLINE_CONTEXT_ROOT_IN_VM}/${suggestedFileName}`;
15674
+ await args.fs.writeFile(filePath, content, { mode: 420 });
15675
+ inlineContexts.push({
15676
+ slug: suggestedFileName.replace(/\.md$/u, ""),
15677
+ content
15678
+ });
15679
+ }
15680
+ }
15636
15681
  });
15682
+ if (inlineContexts.length > 0) {
15683
+ const packContent = buildWorkspaceContextPack(inlineContexts);
15684
+ await args.fs.writeFile(WORKSPACE_CONTEXT_PACK, packContent, { mode: 420 });
15685
+ await args.fs.writeFile(WORKSPACE_AGENTS_MD, packContent, { mode: 420 });
15686
+ await args.fs.mkdir(WORKSPACE_CLAUDE_DIR, { recursive: true });
15687
+ await args.fs.writeFile(WORKSPACE_CLAUDE_MD, "@../context-pack.md\n", { mode: 420 });
15688
+ }
15637
15689
  return {
15638
15690
  injected: resolved.injected,
15639
15691
  skills,
@@ -15641,6 +15693,17 @@ async function injectTaskContext(args) {
15641
15693
  userInlineSuffix: resolved.userInlineSuffix
15642
15694
  };
15643
15695
  }
15696
+ function buildWorkspaceContextPack(contexts) {
15697
+ return [
15698
+ "# Context Pack",
15699
+ "",
15700
+ ...contexts.map(({ slug, content }) => [
15701
+ `## ${slug}`,
15702
+ "",
15703
+ content.trimEnd()
15704
+ ].join("\n"))
15705
+ ].join("\n\n").trimEnd() + "\n";
15706
+ }
15644
15707
  /**
15645
15708
  * Build a `Skill` object pi will faithfully render in
15646
15709
  * `<available_skills>`. We extract `name` and `description` from the
@@ -16004,7 +16067,7 @@ async function parseStructuredTaskOutput(assistantText, taskType, opts = {}) {
16004
16067
  }
16005
16068
  };
16006
16069
  }
16007
- const errors = validateTaskOutput(taskType, extracted);
16070
+ const errors = validateTaskOutput(taskType, extracted, opts.input);
16008
16071
  if (errors.length > 0) {
16009
16072
  const details = errors.slice(0, 3).map((error) => `${error.field}: ${error.message}`);
16010
16073
  const [firstError] = errors;
@@ -16118,7 +16181,7 @@ function createSubmitOutputTool(taskType, opts = {}) {
16118
16181
  description: contract.description,
16119
16182
  parameters: schema,
16120
16183
  async execute(_id, params) {
16121
- const errors = validateTaskOutput(taskType, params);
16184
+ const errors = validateTaskOutput(taskType, params, opts.input);
16122
16185
  if (errors.length > 0) {
16123
16186
  const detailMsg = errors.slice(0, 3).map((err) => `${err.field}: ${err.message}`).join("; ");
16124
16187
  const details = {
@@ -16187,6 +16250,39 @@ function resolveSubmitTools(taskType, opts = {}) {
16187
16250
  //#region ../../libs/pi-extension/src/runtime/task-workspace.ts
16188
16251
  function prepareTaskWorkspace(task, requestedMountPath, executionPlan) {
16189
16252
  const branch = executionPlan?.worktreeBranch ?? null;
16253
+ const workspaceMode = executionPlan?.workspaceMode ?? "shared_mount";
16254
+ const attachedWorkspace = executionPlan?.workspaceAttachment ?? null;
16255
+ if (attachedWorkspace) return {
16256
+ mountPath: attachedWorkspace.mountPath,
16257
+ cwdPath: attachedWorkspace.cwdPath,
16258
+ mode: workspaceMode,
16259
+ branch,
16260
+ cleanup: () => {}
16261
+ };
16262
+ if (workspaceMode === "scratch_mount") {
16263
+ const scratchDir = resolveTaskScratchPath(findMainWorktree(), executionPlan?.workspaceId ?? `task-${task.id}`);
16264
+ const keepWorkspace = executionPlan?.workspaceScope === "session" && executionPlan.sessionKey !== null;
16265
+ if (keepWorkspace) mkdirSync(scratchDir, { recursive: true });
16266
+ else {
16267
+ rmSync(scratchDir, {
16268
+ recursive: true,
16269
+ force: true
16270
+ });
16271
+ mkdirSync(scratchDir, { recursive: true });
16272
+ }
16273
+ return {
16274
+ mountPath: scratchDir,
16275
+ cwdPath: scratchDir,
16276
+ mode: "scratch_mount",
16277
+ branch: null,
16278
+ cleanup: keepWorkspace ? () => {} : () => {
16279
+ rmSync(scratchDir, {
16280
+ recursive: true,
16281
+ force: true
16282
+ });
16283
+ }
16284
+ };
16285
+ }
16190
16286
  if (!branch) return {
16191
16287
  mountPath: requestedMountPath,
16192
16288
  cwdPath: requestedMountPath,
@@ -16224,6 +16320,9 @@ function prepareTaskWorkspace(task, requestedMountPath, executionPlan) {
16224
16320
  function resolveTaskWorktreePath(mainRepo, workspaceId) {
16225
16321
  return join(mainRepo, ".worktrees", workspaceId);
16226
16322
  }
16323
+ function resolveTaskScratchPath(mainRepo, workspaceId) {
16324
+ return join(mainRepo, ".moltnet", "d", "task-workspaces", workspaceId);
16325
+ }
16227
16326
  function ensureReusableTaskWorktree(mainRepo, worktreeDir, branch) {
16228
16327
  if (isRegisteredWorktree$1(mainRepo, worktreeDir)) return;
16229
16328
  if (existsSync(worktreeDir)) throw new Error(`Expected reusable worktree ${worktreeDir} to be git-managed, but it exists outside git worktree metadata.`);
@@ -16460,12 +16559,14 @@ async function executePiTask(claimedTask, reporter, opts) {
16460
16559
  return makeFailedOutput("worktree_setup_failed", message);
16461
16560
  }
16462
16561
  try {
16562
+ const sandboxConfig = applyExecutionPlanSandboxOverrides(opts.sandboxConfig, executionPlan);
16463
16563
  managed = await resumeVm({
16464
16564
  checkpointPath,
16465
16565
  agentName: opts.agentName,
16466
16566
  mountPath,
16567
+ workspaceMode: workspace.mode,
16467
16568
  extraAllowedHosts: opts.extraAllowedHosts,
16468
- sandboxConfig: opts.sandboxConfig
16569
+ sandboxConfig
16469
16570
  });
16470
16571
  } catch (err) {
16471
16572
  const message = err instanceof Error ? err.message : String(err);
@@ -16494,7 +16595,8 @@ async function executePiTask(claimedTask, reporter, opts) {
16494
16595
  taskId: task.id,
16495
16596
  workspace: {
16496
16597
  mode: activeWorkspace.mode,
16497
- branch: activeWorkspace.branch
16598
+ branch: activeWorkspace.branch,
16599
+ attached: executionPlan?.workspaceAttachment !== void 0
16498
16600
  },
16499
16601
  extras: opts.promptExtras
16500
16602
  });
@@ -16536,7 +16638,10 @@ async function executePiTask(claimedTask, reporter, opts) {
16536
16638
  createEditToolDefinition(mountPath, { operations: createGondolinEditOps(managed.vm, mountPath) }),
16537
16639
  createBashToolDefinition(mountPath, { operations: createGondolinBashOps(managed.vm, mountPath) })
16538
16640
  ];
16539
- const { handle: submitToolHandle, tools: submitToolDefs } = resolveSubmitTools(task.taskType, { model: opts.model });
16641
+ const { handle: submitToolHandle, tools: submitToolDefs } = resolveSubmitTools(task.taskType, {
16642
+ model: opts.model,
16643
+ input: task.input
16644
+ });
16540
16645
  const submitTools = submitToolDefs;
16541
16646
  try {
16542
16647
  const moltnetAgent = await connect({ configDir: managed.agentDir });
@@ -16755,8 +16860,20 @@ async function executePiTask(claimedTask, reporter, opts) {
16755
16860
  phase: "output_validation"
16756
16861
  });
16757
16862
  }
16758
- else {
16759
- const parsed = await parseStructuredTaskOutput(assistantText, task.taskType, { model: opts.model });
16863
+ else if (submitToolHandle) {
16864
+ parseError = {
16865
+ code: "output_missing",
16866
+ message: "Agent did not submit output through the task submit tool. A valid submit tool call is required to complete this task type."
16867
+ };
16868
+ await emit("error", {
16869
+ message: parseError.message,
16870
+ phase: "output_validation"
16871
+ });
16872
+ } else {
16873
+ const parsed = await parseStructuredTaskOutput(assistantText, task.taskType, {
16874
+ model: opts.model,
16875
+ input: task.input
16876
+ });
16760
16877
  parsedOutput = parsed.output;
16761
16878
  parsedOutputCid = parsed.outputCid;
16762
16879
  parseError = parsed.error;
@@ -16842,6 +16959,18 @@ async function executePiTask(claimedTask, reporter, opts) {
16842
16959
  }
16843
16960
  }
16844
16961
  }
16962
+ function applyExecutionPlanSandboxOverrides(sandboxConfig, executionPlan) {
16963
+ const shadowWrites = executionPlan?.workspaceAttachment?.shadowWrites;
16964
+ if (!shadowWrites) return sandboxConfig;
16965
+ return {
16966
+ ...sandboxConfig,
16967
+ vfs: {
16968
+ ...sandboxConfig?.vfs,
16969
+ shadow: ["**"],
16970
+ shadowMode: shadowWrites
16971
+ }
16972
+ };
16973
+ }
16845
16974
  function emptyUsage(provider, model) {
16846
16975
  return {
16847
16976
  inputTokens: 0,
@@ -17066,11 +17195,12 @@ var DaemonSlotRegistryError = class extends Error {
17066
17195
  this.name = "DaemonSlotRegistryError";
17067
17196
  }
17068
17197
  };
17198
+ var SqliteDatabaseSync = DatabaseSync;
17069
17199
  var DaemonSlotRegistry = class {
17070
17200
  db;
17071
17201
  constructor(dbPath) {
17072
17202
  try {
17073
- this.db = new DatabaseSync(dbPath);
17203
+ this.db = new SqliteDatabaseSync(dbPath);
17074
17204
  this.withDb("initialize schema", () => {
17075
17205
  this.db.exec(`
17076
17206
  PRAGMA journal_mode = WAL;
@@ -17119,6 +17249,9 @@ var DaemonSlotRegistry = class {
17119
17249
 
17120
17250
  CREATE INDEX IF NOT EXISTS daemon_slots_expires_idx
17121
17251
  ON daemon_slots (expires_at_ms);
17252
+
17253
+ CREATE INDEX IF NOT EXISTS daemon_slots_task_attempt_idx
17254
+ ON daemon_slots (last_task_id, last_attempt_n, last_used_at_ms DESC);
17122
17255
  `);
17123
17256
  });
17124
17257
  } catch (error) {
@@ -17208,6 +17341,30 @@ var DaemonSlotRegistry = class {
17208
17341
  SET session_path = ?
17209
17342
  WHERE agent_name = ? AND provider = ? AND model = ? AND slot_key = ?`).run(sessionPath, identity.agentName, identity.provider, identity.model, slotKey));
17210
17343
  }
17344
+ findLatestProducerSlotByTaskAttempt(taskId, attemptN) {
17345
+ const slot = this.withDb("find producer slot by task attempt", () => this.db.prepare(`SELECT
17346
+ agent_name as agentName,
17347
+ provider,
17348
+ model,
17349
+ slot_key as slotKey,
17350
+ task_type as taskType,
17351
+ state,
17352
+ last_task_id as lastTaskId,
17353
+ last_attempt_n as lastAttemptN,
17354
+ created_at_ms as createdAtMs,
17355
+ last_used_at_ms as lastUsedAtMs,
17356
+ expires_at_ms as expiresAtMs
17357
+ FROM daemon_slots
17358
+ WHERE last_task_id = ? AND last_attempt_n = ?
17359
+ ORDER BY last_used_at_ms DESC
17360
+ LIMIT 1`).get(taskId, attemptN) ?? null);
17361
+ if (!slot) return null;
17362
+ return {
17363
+ slot,
17364
+ session: this.lookupSession(slot),
17365
+ workspace: this.lookupWorkspace(slot)
17366
+ };
17367
+ }
17211
17368
  reapExpiredSlots(now = Date.now()) {
17212
17369
  this.withDb("begin reap transaction", () => this.db.exec("BEGIN IMMEDIATE"));
17213
17370
  try {
@@ -17251,8 +17408,8 @@ var DaemonSlotRegistry = class {
17251
17408
  const deleteStmt = this.withDb("prepare expired slot delete", () => this.db.prepare("DELETE FROM daemon_slots WHERE agent_name = ? AND provider = ? AND model = ? AND slot_key = ?"));
17252
17409
  const out = [];
17253
17410
  for (const slot of slots) {
17254
- const session = this.withDb("select slot session", () => selectSession.get(slot.agentName, slot.provider, slot.model, slot.slotKey) ?? null);
17255
- const workspace = this.withDb("select slot workspace", () => selectWorkspace.get(slot.agentName, slot.provider, slot.model, slot.slotKey) ?? null);
17411
+ const session = this.lookupSession(slot, selectSession);
17412
+ const workspace = this.lookupWorkspace(slot, selectWorkspace);
17256
17413
  out.push({
17257
17414
  slot,
17258
17415
  session,
@@ -17280,6 +17437,29 @@ var DaemonSlotRegistry = class {
17280
17437
  throw new DaemonSlotRegistryError(operation, error);
17281
17438
  }
17282
17439
  }
17440
+ lookupSession(slot, stmt = this.withDb("prepare slot session lookup", () => this.db.prepare(`SELECT
17441
+ agent_name as agentName,
17442
+ provider,
17443
+ model,
17444
+ slot_key as slotKey,
17445
+ session_dir as sessionDir,
17446
+ session_path as sessionPath
17447
+ FROM daemon_slot_sessions
17448
+ WHERE agent_name = ? AND provider = ? AND model = ? AND slot_key = ?`))) {
17449
+ return this.withDb("select slot session", () => stmt.get(slot.agentName, slot.provider, slot.model, slot.slotKey) ?? null);
17450
+ }
17451
+ lookupWorkspace(slot, stmt = this.withDb("prepare slot workspace lookup", () => this.db.prepare(`SELECT
17452
+ agent_name as agentName,
17453
+ provider,
17454
+ model,
17455
+ slot_key as slotKey,
17456
+ workspace_id as workspaceId,
17457
+ worktree_path as worktreePath,
17458
+ worktree_branch as worktreeBranch
17459
+ FROM daemon_slot_workspaces
17460
+ WHERE agent_name = ? AND provider = ? AND model = ? AND slot_key = ?`))) {
17461
+ return this.withDb("select slot workspace", () => stmt.get(slot.agentName, slot.provider, slot.model, slot.slotKey) ?? null);
17462
+ }
17283
17463
  };
17284
17464
  function resolveLatestPiSessionPath(sessionDir) {
17285
17465
  try {
@@ -17380,11 +17560,6 @@ function buildCustomSessionKey(task) {
17380
17560
  if (!task.correlationId || !variantLabel) return null;
17381
17561
  return `run_eval:correlation:${task.correlationId}:variant:${slugifySessionComponent(variantLabel)}`;
17382
17562
  }
17383
- case "judge_eval_variant": {
17384
- const runTaskIds = Array.isArray(task.input.runTaskIds) ? task.input.runTaskIds.filter((value) => typeof value === "string") : [];
17385
- if (runTaskIds.length < 1) return null;
17386
- return `judge_eval_variant:run_tasks:${[...runTaskIds].sort().join(",")}`;
17387
- }
17388
17563
  default: return null;
17389
17564
  }
17390
17565
  }
@@ -17395,18 +17570,20 @@ function slugifySessionComponent(input) {
17395
17570
  //#region src/lib/task-execution-plan.ts
17396
17571
  function buildDaemonTaskExecutionPlan(task, stateDirs, identity, warmSessionTtlSec) {
17397
17572
  const descriptor = deriveTaskSessionDescriptor(task);
17573
+ const workspaceMode = resolveTaskWorkspaceMode(task, descriptor.policy);
17398
17574
  const slotKey = warmSessionTtlSec > 0 ? descriptor.sessionKey : null;
17399
17575
  const workspaceScope = slotKey !== null ? descriptor.policy.workspaceScope : "attempt";
17400
17576
  const slotId = slotKey ? buildDaemonSlotId(identity, slotKey) : null;
17401
17577
  const sessionDir = slotId ? `${stateDirs.piSessionsDir}/${encodeURIComponent(slotId)}` : null;
17402
- const worktreeBranch = resolveTaskWorktreeBranch(task, descriptor.policy);
17403
- const workspaceId = worktreeBranch !== null ? resolveTaskWorkspaceId(task, {
17578
+ const worktreeBranch = resolveTaskWorktreeBranch(task, workspaceMode);
17579
+ const workspaceId = workspaceMode !== "shared_mount" ? resolveTaskWorkspaceId(task, {
17404
17580
  sessionKey: slotId,
17405
17581
  workspaceScope,
17406
17582
  sessionPersistence: sessionDir ? { sessionDir } : null
17407
17583
  }) : null;
17408
17584
  return {
17409
17585
  descriptor,
17586
+ workspaceMode,
17410
17587
  sessionKey: slotId,
17411
17588
  slotKey,
17412
17589
  slotId,
@@ -17435,8 +17612,8 @@ function slugSlotIdentityComponent(input) {
17435
17612
  "-"
17436
17613
  ]);
17437
17614
  }
17438
- function resolveTaskWorktreeBranch(task, policy) {
17439
- if (policy.workspaceMode !== "dedicated_worktree") return null;
17615
+ function resolveTaskWorktreeBranch(task, workspaceMode) {
17616
+ if (workspaceMode !== "dedicated_worktree") return null;
17440
17617
  if (task.taskType === "fulfill_brief") {
17441
17618
  const input = task.input;
17442
17619
  const slug = slugifyAsciiLower(typeof input.title === "string" && input.title.trim().length > 0 ? input.title : typeof input.brief === "string" && input.brief.trim().length > 0 ? input.brief : task.taskType, 60) || "task";
@@ -17445,12 +17622,27 @@ function resolveTaskWorktreeBranch(task, policy) {
17445
17622
  }
17446
17623
  return `task/${slugifyAsciiLower(task.taskType, 60) || "task"}-${task.id.slice(0, 8)}`;
17447
17624
  }
17625
+ function resolveTaskWorkspaceMode(task, policy) {
17626
+ if (task.taskType !== "run_eval") return policy.workspaceMode;
17627
+ switch (typeof task.input.execution?.workspace === "string" ? task.input.execution.workspace : null) {
17628
+ case "none": return "scratch_mount";
17629
+ case "shared_mount": return "shared_mount";
17630
+ case "dedicated_worktree": return "dedicated_worktree";
17631
+ default: return policy.workspaceMode;
17632
+ }
17633
+ }
17448
17634
  function resolveTaskWorkspaceId(task, executionPlan) {
17449
17635
  if (executionPlan.workspaceScope === "session" && executionPlan.sessionKey !== null) return `session-${encodeURIComponent(executionPlan.sessionKey)}`;
17450
17636
  return `task-${task.id}`;
17451
17637
  }
17452
17638
  //#endregion
17453
17639
  //#region src/lib/execution-plan-cache.ts
17640
+ var ProducerContextResolutionError = class extends Error {
17641
+ constructor(message) {
17642
+ super(message);
17643
+ this.name = "ProducerContextResolutionError";
17644
+ }
17645
+ };
17454
17646
  function createExecutionPlanCache(args) {
17455
17647
  const cache = /* @__PURE__ */ new Map();
17456
17648
  return {
@@ -17458,7 +17650,7 @@ function createExecutionPlanCache(args) {
17458
17650
  const key = buildClaimedTaskKey(claimedTask);
17459
17651
  const existing = cache.get(key);
17460
17652
  if (existing) return existing;
17461
- const plan = buildDaemonTaskExecutionPlan(claimedTask.task, args.stateDirs, args.slotIdentity, args.warmSessionTtlSec);
17653
+ const plan = maybeAttachProducerContext(claimedTask, buildDaemonTaskExecutionPlan(claimedTask.task, args.stateDirs, args.slotIdentity, args.warmSessionTtlSec), args.stateDirs, args.slotRegistry);
17462
17654
  cache.set(key, plan);
17463
17655
  return plan;
17464
17656
  },
@@ -17470,17 +17662,90 @@ function createExecutionPlanCache(args) {
17470
17662
  function buildClaimedTaskKey(task) {
17471
17663
  return `${task.task.id}:${task.attemptN}`;
17472
17664
  }
17665
+ function maybeAttachProducerContext(claimedTask, basePlan, stateDirs, slotRegistry) {
17666
+ if (claimedTask.task.taskType !== "judge_eval_attempt") return basePlan;
17667
+ const targetTaskId = typeof claimedTask.task.input.targetTaskId === "string" ? claimedTask.task.input.targetTaskId : null;
17668
+ const targetAttemptN = typeof claimedTask.task.input.targetAttemptN === "number" ? claimedTask.task.input.targetAttemptN : null;
17669
+ if (!targetTaskId || !targetAttemptN) throw new ProducerContextResolutionError("judge_eval_attempt is missing targetTaskId/targetAttemptN");
17670
+ const producer = slotRegistry.findLatestProducerSlotByTaskAttempt(targetTaskId, targetAttemptN);
17671
+ if (!producer) throw new ProducerContextResolutionError(`No persisted producer daemon slot found for task ${targetTaskId} attempt ${targetAttemptN}`);
17672
+ const sourceSessionPath = resolveProducerSessionPath(producer);
17673
+ if (!sourceSessionPath) throw new ProducerContextResolutionError(`Producer task ${targetTaskId} attempt ${targetAttemptN} has no persisted Pi session path`);
17674
+ const attachedWorkspace = resolveProducerWorkspaceAttachment(producer, stateDirs);
17675
+ return {
17676
+ ...basePlan,
17677
+ workspaceMode: attachedWorkspace.mode,
17678
+ worktreeBranch: attachedWorkspace.branch,
17679
+ workspaceAttachment: {
17680
+ mountPath: attachedWorkspace.mountPath,
17681
+ cwdPath: attachedWorkspace.cwdPath,
17682
+ shadowWrites: "tmpfs"
17683
+ },
17684
+ sessionPersistence: {
17685
+ sessionDir: `${stateDirs.piSessionsDir}/judge-${claimedTask.task.id}-attempt-${claimedTask.attemptN}`,
17686
+ forkFromSessionPath: sourceSessionPath
17687
+ }
17688
+ };
17689
+ }
17690
+ function resolveProducerSessionPath(producer) {
17691
+ const explicit = producer.session?.sessionPath ?? null;
17692
+ if (explicit && existsSync(explicit)) return explicit;
17693
+ const sessionDir = producer.session?.sessionDir ?? null;
17694
+ if (!sessionDir || !existsSync(sessionDir)) return null;
17695
+ const latest = resolveLatestPiSessionPath(sessionDir);
17696
+ return latest && existsSync(latest) ? latest : null;
17697
+ }
17698
+ function resolveProducerWorkspaceAttachment(producer, stateDirs) {
17699
+ const workspacePath = producer.workspace?.worktreePath ?? null;
17700
+ if (workspacePath) {
17701
+ if (existsSync(workspacePath)) return {
17702
+ mountPath: workspacePath,
17703
+ cwdPath: workspacePath,
17704
+ mode: producer.workspace?.worktreeBranch ? "dedicated_worktree" : "scratch_mount",
17705
+ branch: producer.workspace?.worktreeBranch ?? null
17706
+ };
17707
+ const recoveredPath = recoverScratchWorkspacePath(producer, stateDirs);
17708
+ if (recoveredPath) return {
17709
+ mountPath: recoveredPath,
17710
+ cwdPath: recoveredPath,
17711
+ mode: "scratch_mount",
17712
+ branch: null
17713
+ };
17714
+ throw new ProducerContextResolutionError(`Producer workspace path is missing on disk: ${workspacePath}`);
17715
+ }
17716
+ const sharedMountRoot = dirname(dirname(stateDirs.rootDir));
17717
+ if (!existsSync(sharedMountRoot)) throw new ProducerContextResolutionError(`Shared producer mount root is missing on disk: ${sharedMountRoot}`);
17718
+ return {
17719
+ mountPath: sharedMountRoot,
17720
+ cwdPath: sharedMountRoot,
17721
+ mode: "shared_mount",
17722
+ branch: null
17723
+ };
17724
+ }
17725
+ function recoverScratchWorkspacePath(producer, stateDirs) {
17726
+ if (producer.workspace?.worktreeBranch) return null;
17727
+ if (!producer.workspace?.workspaceId) return null;
17728
+ const fallback = join(stateDirs.rootDir, "task-workspaces", producer.workspace.workspaceId);
17729
+ return existsSync(fallback) ? fallback : null;
17730
+ }
17473
17731
  //#endregion
17474
17732
  //#region src/lib/finalize.ts
17475
17733
  async function finalizeTask(agent, output, ctx = {}) {
17476
17734
  if (output.status === "cancelled") return;
17477
17735
  if (output.status === "completed" && output.output && output.outputCid) {
17478
- await agent.tasks.complete(output.taskId, output.attemptN, {
17479
- output: output.output,
17480
- outputCid: output.outputCid,
17481
- usage: output.usage,
17482
- ...output.contentSignature ? { contentSignature: output.contentSignature } : {}
17483
- });
17736
+ try {
17737
+ await agent.tasks.complete(output.taskId, output.attemptN, {
17738
+ output: output.output,
17739
+ outputCid: output.outputCid,
17740
+ usage: output.usage,
17741
+ ...output.contentSignature ? { contentSignature: output.contentSignature } : {}
17742
+ });
17743
+ } catch (err) {
17744
+ const reason = errorToFailReason(err);
17745
+ ctx.log?.("complete-rejected-falling-back-to-fail", err);
17746
+ await agent.tasks.fail(output.taskId, output.attemptN, { error: reason });
17747
+ return;
17748
+ }
17484
17749
  await maybeWriteAnchors(output, ctx);
17485
17750
  return;
17486
17751
  }
@@ -17492,6 +17757,21 @@ async function finalizeTask(agent, output, ctx = {}) {
17492
17757
  if ((await agent.tasks.heartbeat(output.taskId, output.attemptN, {})).cancelled) return;
17493
17758
  await agent.tasks.fail(output.taskId, output.attemptN, { error });
17494
17759
  }
17760
+ function errorToFailReason(err) {
17761
+ if (err instanceof MoltNetError) {
17762
+ const fields = err.validationErrors?.length ? "; " + err.validationErrors.map((e) => `${e.field}: ${e.message}`).join(" | ") : "";
17763
+ return {
17764
+ code: "output_rejected_by_server",
17765
+ message: `Server rejected tasks.complete (${err.code}, status ${err.statusCode ?? "?"}): ${err.detail ?? err.message}${fields}`,
17766
+ retryable: false
17767
+ };
17768
+ }
17769
+ return {
17770
+ code: "complete_call_failed",
17771
+ message: err instanceof Error ? err.message : String(err),
17772
+ retryable: false
17773
+ };
17774
+ }
17495
17775
  async function maybeWriteAnchors(output, ctx) {
17496
17776
  const { task, writeCorrelationAnchors, log } = ctx;
17497
17777
  if (!task || task.taskType !== "fulfill_brief") return;
@@ -17793,7 +18073,8 @@ async function runPolling(opts) {
17793
18073
  const executionPlans = createExecutionPlanCache({
17794
18074
  stateDirs,
17795
18075
  slotIdentity,
17796
- warmSessionTtlSec: common.warmSessionTtlSec
18076
+ warmSessionTtlSec: common.warmSessionTtlSec,
18077
+ slotRegistry
17797
18078
  });
17798
18079
  const ctx = await resolveAgentContext(common.agent);
17799
18080
  const cfg = loadConfig();
@@ -17838,15 +18119,9 @@ async function runPolling(opts) {
17838
18119
  maxPollIntervalMs
17839
18120
  }, "agent-daemon.starting");
17840
18121
  const outputs = [];
17841
- const subagentContractRegistry = createSubagentContractRegistry([{
17842
- name: "judge_eval_variant_result",
17843
- description: "Per-variant grading result produced by a subagent of judge_eval_variant: scores against the shared rubric, composite, and a 1-3 sentence verdict for a single variant.",
17844
- parametersSchema: JudgeEvalVariantResult
17845
- }]);
17846
18122
  try {
17847
18123
  const executeTask = createPiTaskExecutor({
17848
18124
  agentName: common.agent,
17849
- subagentContractRegistry,
17850
18125
  mountPath: sandbox.rootDir,
17851
18126
  provider: common.provider,
17852
18127
  model: common.model,
@@ -17890,7 +18165,34 @@ async function runPolling(opts) {
17890
18165
  log: (msg, err) => rootLogger.warn({ err }, msg)
17891
18166
  }),
17892
18167
  executeTask: async (claimedTask, reporter) => {
17893
- const executionPlan = executionPlans.getOrCreate(claimedTask);
18168
+ let executionPlan;
18169
+ try {
18170
+ executionPlan = executionPlans.getOrCreate(claimedTask);
18171
+ } catch (err) {
18172
+ const message = err instanceof Error ? err.message : String(err);
18173
+ rootLogger.warn({
18174
+ taskId: claimedTask.task.id,
18175
+ attemptN: claimedTask.attemptN,
18176
+ err: message
18177
+ }, "agent-daemon.execution_plan_failed");
18178
+ return {
18179
+ taskId: claimedTask.task.id,
18180
+ attemptN: claimedTask.attemptN,
18181
+ status: "failed",
18182
+ output: null,
18183
+ outputCid: null,
18184
+ usage: {
18185
+ inputTokens: 0,
18186
+ outputTokens: 0
18187
+ },
18188
+ durationMs: 0,
18189
+ error: {
18190
+ code: err instanceof ProducerContextResolutionError ? "producer_context_missing" : "execution_plan_failed",
18191
+ message,
18192
+ retryable: false
18193
+ }
18194
+ };
18195
+ }
17894
18196
  const sessionDescriptor = executionPlan.descriptor;
17895
18197
  let expired;
17896
18198
  try {
@@ -17909,7 +18211,7 @@ async function runPolling(opts) {
17909
18211
  taskId: claimedTask.task.id,
17910
18212
  taskType: claimedTask.task.taskType,
17911
18213
  resumable: sessionDescriptor.policy.resumable,
17912
- workspaceMode: sessionDescriptor.policy.workspaceMode,
18214
+ workspaceMode: executionPlan.workspaceMode,
17913
18215
  workspaceScope: sessionDescriptor.policy.workspaceScope,
17914
18216
  sessionScope: sessionDescriptor.policy.sessionScope,
17915
18217
  slotKey: executionPlan.slotKey,
@@ -17959,7 +18261,7 @@ async function runPolling(opts) {
17959
18261
  sessionDir: executionPlan.sessionPersistence.sessionDir,
17960
18262
  sessionPath: resolveLatestPiSessionPath(executionPlan.sessionPersistence.sessionDir),
17961
18263
  workspaceId: executionPlan.workspaceId,
17962
- worktreePath: executionPlan.workspaceId ? resolveTaskWorktreePath(mainRepo, executionPlan.workspaceId) : null,
18264
+ worktreePath: resolveRecordedWorkspacePath$1(mainRepo, stateDirs.rootDir, executionPlan),
17963
18265
  worktreeBranch: executionPlan.worktreeBranch,
17964
18266
  lastTaskId: claimedTask.task.id,
17965
18267
  lastAttemptN: claimedTask.attemptN,
@@ -17983,6 +18285,10 @@ async function runPolling(opts) {
17983
18285
  await shutdownLogger();
17984
18286
  }
17985
18287
  }
18288
+ function resolveRecordedWorkspacePath$1(mainRepo, stateRootDir, executionPlan) {
18289
+ if (!executionPlan.workspaceId) return null;
18290
+ return executionPlan.workspaceMode === "scratch_mount" ? join(stateRootDir, "task-workspaces", executionPlan.workspaceId) : join(mainRepo, ".worktrees", executionPlan.workspaceId);
18291
+ }
17986
18292
  function parseCsv(raw) {
17987
18293
  return (raw ?? "").split(",").map((s) => s.trim()).filter((s) => s.length > 0);
17988
18294
  }
@@ -18050,7 +18356,8 @@ async function runOnce(argv) {
18050
18356
  const executionPlans = createExecutionPlanCache({
18051
18357
  stateDirs,
18052
18358
  slotIdentity,
18053
- warmSessionTtlSec: opts.warmSessionTtlSec
18359
+ warmSessionTtlSec: opts.warmSessionTtlSec,
18360
+ slotRegistry
18054
18361
  });
18055
18362
  const ctx = await resolveAgentContext(opts.agent);
18056
18363
  const cfg = loadConfig();
@@ -18099,15 +18406,9 @@ async function runOnce(argv) {
18099
18406
  process.on("SIGTERM", () => {
18100
18407
  onSignal("SIGTERM");
18101
18408
  });
18102
- const subagentContractRegistry = createSubagentContractRegistry([{
18103
- name: "judge_eval_variant_result",
18104
- description: "Per-variant grading result produced by a subagent of judge_eval_variant: scores against the shared rubric, composite, and a 1-3 sentence verdict for a single variant.",
18105
- parametersSchema: JudgeEvalVariantResult
18106
- }]);
18107
18409
  try {
18108
18410
  const rawExecuteTask = createPiTaskExecutor({
18109
18411
  agentName: opts.agent,
18110
- subagentContractRegistry,
18111
18412
  mountPath: sandbox.rootDir,
18112
18413
  provider: opts.provider,
18113
18414
  model: opts.model,
@@ -18130,7 +18431,34 @@ async function runOnce(argv) {
18130
18431
  err: err instanceof Error ? err.message : String(err)
18131
18432
  }, "agent-daemon.daemon_slot_reap_failed");
18132
18433
  }
18133
- const executionPlan = executionPlans.getOrCreate(claimedTask);
18434
+ let executionPlan;
18435
+ try {
18436
+ executionPlan = executionPlans.getOrCreate(claimedTask);
18437
+ } catch (err) {
18438
+ const message = err instanceof Error ? err.message : String(err);
18439
+ rootLogger.warn({
18440
+ taskId: claimedTask.task.id,
18441
+ attemptN: claimedTask.attemptN,
18442
+ err: message
18443
+ }, "agent-daemon.execution_plan_failed");
18444
+ return {
18445
+ taskId: claimedTask.task.id,
18446
+ attemptN: claimedTask.attemptN,
18447
+ status: "failed",
18448
+ output: null,
18449
+ outputCid: null,
18450
+ usage: {
18451
+ inputTokens: 0,
18452
+ outputTokens: 0
18453
+ },
18454
+ durationMs: 0,
18455
+ error: {
18456
+ code: err instanceof ProducerContextResolutionError ? "producer_context_missing" : "execution_plan_failed",
18457
+ message,
18458
+ retryable: false
18459
+ }
18460
+ };
18461
+ }
18134
18462
  if (executionPlan.slotKey && executionPlan.sessionPersistence) slotRegistry.beginSlot({
18135
18463
  ...slotIdentity,
18136
18464
  slotKey: executionPlan.slotKey,
@@ -18138,7 +18466,7 @@ async function runOnce(argv) {
18138
18466
  sessionDir: executionPlan.sessionPersistence.sessionDir,
18139
18467
  sessionPath: resolveLatestPiSessionPath(executionPlan.sessionPersistence.sessionDir),
18140
18468
  workspaceId: executionPlan.workspaceId,
18141
- worktreePath: executionPlan.workspaceId ? resolveTaskWorktreePath(mainRepo, executionPlan.workspaceId) : null,
18469
+ worktreePath: resolveRecordedWorkspacePath(mainRepo, stateDirs.rootDir, executionPlan),
18142
18470
  worktreeBranch: executionPlan.worktreeBranch,
18143
18471
  lastTaskId: claimedTask.task.id,
18144
18472
  lastAttemptN: claimedTask.attemptN,
@@ -18190,6 +18518,10 @@ async function runOnce(argv) {
18190
18518
  await shutdownLogger();
18191
18519
  }
18192
18520
  }
18521
+ function resolveRecordedWorkspacePath(mainRepo, stateRootDir, executionPlan) {
18522
+ if (!executionPlan.workspaceId) return null;
18523
+ return executionPlan.workspaceMode === "scratch_mount" ? join(stateRootDir, "task-workspaces", executionPlan.workspaceId) : join(mainRepo, ".worktrees", executionPlan.workspaceId);
18524
+ }
18193
18525
  //#endregion
18194
18526
  //#region src/cli/poll.ts
18195
18527
  function runPoll(argv) {