@themoltnet/agent-daemon 0.9.0 → 0.10.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +15 -0
- package/dist/main.js +797 -442
- package/package.json +7 -7
package/dist/main.js
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import crypto, { createHash } from "crypto";
|
|
3
|
+
import path, { dirname, isAbsolute, join, relative, resolve } from "node:path";
|
|
3
4
|
import { parseArgs, parseEnv, promisify } from "node:util";
|
|
4
5
|
import { existsSync, mkdirSync, readFileSync, readdirSync, rmSync, statSync } from "node:fs";
|
|
5
6
|
import { ROOT_CONTEXT, SpanStatusCode, context, metrics, propagation, trace } from "@opentelemetry/api";
|
|
6
7
|
import { pino, transport } from "pino";
|
|
7
8
|
import { readFile } from "node:fs/promises";
|
|
8
9
|
import { createHash as createHash$1 } from "node:crypto";
|
|
9
|
-
import path, { dirname, isAbsolute, join, relative, resolve } from "node:path";
|
|
10
10
|
import { homedir } from "node:os";
|
|
11
11
|
import { execFile, execFileSync } from "node:child_process";
|
|
12
12
|
import { DefaultResourceLoader, SessionManager, createAgentSession, createBashToolDefinition, createEditToolDefinition, createReadToolDefinition, createSyntheticSourceInfo, createWriteToolDefinition, defineTool, parseFrontmatter } from "@earendil-works/pi-coding-agent";
|
|
@@ -2896,6 +2896,7 @@ if (!Has$1("date-time")) Set$1("date-time", (v) => !Number.isNaN(Date.parse(v)))
|
|
|
2896
2896
|
*/
|
|
2897
2897
|
var ContextBinding = Type$2.Union([
|
|
2898
2898
|
Type$2.Literal("skill"),
|
|
2899
|
+
Type$2.Literal("context_inline"),
|
|
2899
2900
|
Type$2.Literal("prompt_prefix"),
|
|
2900
2901
|
Type$2.Literal("user_inline")
|
|
2901
2902
|
], { $id: "ContextBinding" });
|
|
@@ -2912,9 +2913,14 @@ var ContextBinding = Type$2.Union([
|
|
|
2912
2913
|
* name under the runtime's skill discovery path. Must be
|
|
2913
2914
|
* kebab-case-safe (alphanumeric + dashes/underscores).
|
|
2914
2915
|
* - `binding` — how the bytes are delivered to the LLM (see above).
|
|
2915
|
-
* - `content` — the actual bytes (UTF-8 text). Capped at
|
|
2916
|
+
* - `content` — the actual bytes (UTF-8 text). Capped at 64 KiB per
|
|
2916
2917
|
* entry; total per-task context bytes are bounded by the
|
|
2917
2918
|
* soft `maxItems` cap and per-binding daemon limits.
|
|
2919
|
+
* Raised from 32 KiB in 2026-05 — protocol-heavy operator
|
|
2920
|
+
* skills (e.g. `.claude/skills/legreffier/SKILL.md`) ship
|
|
2921
|
+
* at ~35 KiB inline, and the original cap was sized for
|
|
2922
|
+
* short example skills, not the kind of skill the eval
|
|
2923
|
+
* substrate is dogfooded on (#943, #823).
|
|
2918
2924
|
*/
|
|
2919
2925
|
var ContextRef = Type$2.Object({
|
|
2920
2926
|
slug: Type$2.String({
|
|
@@ -2925,7 +2931,7 @@ var ContextRef = Type$2.Object({
|
|
|
2925
2931
|
binding: ContextBinding,
|
|
2926
2932
|
content: Type$2.String({
|
|
2927
2933
|
minLength: 1,
|
|
2928
|
-
maxLength:
|
|
2934
|
+
maxLength: 65536
|
|
2929
2935
|
})
|
|
2930
2936
|
}, {
|
|
2931
2937
|
$id: "ContextRef",
|
|
@@ -4330,61 +4336,33 @@ async function validateJudgePackInputAsync(input, ctx) {
|
|
|
4330
4336
|
return errors;
|
|
4331
4337
|
}
|
|
4332
4338
|
//#endregion
|
|
4333
|
-
//#region ../../libs/tasks/src/task-types/judge-eval-
|
|
4339
|
+
//#region ../../libs/tasks/src/task-types/judge-eval-attempt.ts
|
|
4334
4340
|
/**
|
|
4335
|
-
* `
|
|
4336
|
-
*
|
|
4337
|
-
* isolation.
|
|
4341
|
+
* `judge_eval_attempt` — score one completed `run_eval` attempt against a
|
|
4342
|
+
* hidden judge rubric.
|
|
4338
4343
|
*
|
|
4339
4344
|
* output_kind: judgment
|
|
4340
|
-
* criteria: required (`successCriteria.rubric`
|
|
4341
|
-
*
|
|
4342
|
-
*
|
|
4343
|
-
* pin the targets being graded.
|
|
4344
|
-
*
|
|
4345
|
-
* Slice 2 of #943. The parent task carries the rubric and the list of
|
|
4346
|
-
* variant `run_eval` task ids. The pi executor registers the generic
|
|
4347
|
-
* `subagent` custom tool (#1087), and the parent LLM calls
|
|
4348
|
-
* `subagent({ task, output_schema: 'judge_eval_variant_result' })` once
|
|
4349
|
-
* per variant — each child session has fresh context, fetches the
|
|
4350
|
-
* variant's accepted attempt output via `moltnet_get_task` /
|
|
4351
|
-
* `moltnet_list_task_attempts`, and grades against the rubric.
|
|
4345
|
+
* criteria: required (`successCriteria.rubric`)
|
|
4346
|
+
* references: not required at the input layer — `targetTaskId` +
|
|
4347
|
+
* `targetAttemptN` pin the producer attempt being judged.
|
|
4352
4348
|
*
|
|
4353
|
-
*
|
|
4354
|
-
*
|
|
4355
|
-
*
|
|
4356
|
-
*
|
|
4357
|
-
|
|
4358
|
-
|
|
4359
|
-
|
|
4360
|
-
|
|
4361
|
-
|
|
4362
|
-
* which the task service runs at create time (#1096 wiring). The
|
|
4363
|
-
* TypeBox layer here only enforces shape: UUID format,
|
|
4364
|
-
* minItems/maxItems, rubric presence + weight invariant.
|
|
4365
|
-
*/
|
|
4366
|
-
var JUDGE_EVAL_VARIANT_TYPE = "judge_eval_variant";
|
|
4367
|
-
var JudgeEvalVariantInput = Type$2.Object({
|
|
4368
|
-
runTaskIds: Type$2.Array(Type$2.String({ format: "uuid" }), {
|
|
4369
|
-
minItems: 2,
|
|
4370
|
-
maxItems: 10
|
|
4371
|
-
}),
|
|
4349
|
+
* This replaces the earlier parent/subagent `judge_eval_variant` design.
|
|
4350
|
+
* The unit of judgment is one producer attempt. Cross-variant deltas can be
|
|
4351
|
+
* computed later at read time from stored scores, rather than materialized as
|
|
4352
|
+
* their own task output.
|
|
4353
|
+
*/
|
|
4354
|
+
var JUDGE_EVAL_ATTEMPT_TYPE = "judge_eval_attempt";
|
|
4355
|
+
var JudgeEvalAttemptInput = Type$2.Object({
|
|
4356
|
+
targetTaskId: Type$2.String({ format: "uuid" }),
|
|
4357
|
+
targetAttemptN: Type$2.Integer({ minimum: 1 }),
|
|
4372
4358
|
successCriteria: SuccessCriteria
|
|
4373
4359
|
}, {
|
|
4374
|
-
$id: "
|
|
4360
|
+
$id: "JudgeEvalAttemptInput",
|
|
4375
4361
|
additionalProperties: false
|
|
4376
4362
|
});
|
|
4377
|
-
|
|
4378
|
-
|
|
4379
|
-
|
|
4380
|
-
* deterministic_*). Reuse the type rather than re-declare.
|
|
4381
|
-
*
|
|
4382
|
-
* This is also the **subagent output contract** — the parent's
|
|
4383
|
-
* `subagent` tool resolves the contract name `judge_eval_variant_result`
|
|
4384
|
-
* to this schema. See `agent-runtime`'s subagent contract registry.
|
|
4385
|
-
*/
|
|
4386
|
-
var JudgeEvalVariantResult = Type$2.Object({
|
|
4387
|
-
runTaskId: Type$2.String({ format: "uuid" }),
|
|
4363
|
+
var JudgeEvalAttemptOutput = Type$2.Object({
|
|
4364
|
+
targetTaskId: Type$2.String({ format: "uuid" }),
|
|
4365
|
+
targetAttemptN: Type$2.Integer({ minimum: 1 }),
|
|
4388
4366
|
variantLabel: Type$2.String({
|
|
4389
4367
|
minLength: 1,
|
|
4390
4368
|
maxLength: 64,
|
|
@@ -4395,216 +4373,126 @@ var JudgeEvalVariantResult = Type$2.Object({
|
|
|
4395
4373
|
minimum: 0,
|
|
4396
4374
|
maximum: 1
|
|
4397
4375
|
}),
|
|
4398
|
-
verdict: Type$2.String({ minLength: 1 })
|
|
4399
|
-
}, {
|
|
4400
|
-
$id: "JudgeEvalVariantResult",
|
|
4401
|
-
additionalProperties: false
|
|
4402
|
-
});
|
|
4403
|
-
var JudgeEvalVariantOutput = Type$2.Object({
|
|
4404
|
-
results: Type$2.Array(JudgeEvalVariantResult, { minItems: 2 }),
|
|
4405
|
-
deltas: Type$2.Optional(Type$2.Record(Type$2.String(), Type$2.Number({
|
|
4406
|
-
minimum: -1,
|
|
4407
|
-
maximum: 1
|
|
4408
|
-
}))),
|
|
4376
|
+
verdict: Type$2.String({ minLength: 1 }),
|
|
4409
4377
|
judgeModel: Type$2.Optional(Type$2.String({ minLength: 1 })),
|
|
4410
4378
|
traceparent: Type$2.String({ minLength: 1 })
|
|
4411
4379
|
}, {
|
|
4412
|
-
$id: "
|
|
4380
|
+
$id: "JudgeEvalAttemptOutput",
|
|
4413
4381
|
additionalProperties: false
|
|
4414
4382
|
});
|
|
4415
|
-
|
|
4416
|
-
* Synchronous input invariants beyond TypeBox shape: rubric must be
|
|
4417
|
-
* present (already required by the schema, but the rubric body has
|
|
4418
|
-
* its own per-criterion weight invariant) and the rubric's weights
|
|
4419
|
-
* must sum to 1.
|
|
4420
|
-
*
|
|
4421
|
-
* Cross-task invariants (all targets are `run_eval`, all completed,
|
|
4422
|
-
* share `correlation_id`, byte-identical `input.successCriteria`)
|
|
4423
|
-
* are NOT checked here — they require async DB lookups against
|
|
4424
|
-
* `runTaskIds` and live in `validateJudgeEvalVariantInputAsync`
|
|
4425
|
-
* below, invoked by the task service at create time (#1096).
|
|
4426
|
-
*/
|
|
4427
|
-
function validateJudgeEvalVariantInput(input) {
|
|
4383
|
+
function validateJudgeEvalAttemptInput(input) {
|
|
4428
4384
|
const sc = input.successCriteria;
|
|
4429
|
-
if (!sc) return "successCriteria is required for
|
|
4430
|
-
if (!sc.rubric) return "successCriteria.rubric is required for
|
|
4385
|
+
if (!sc) return "successCriteria is required for judge_eval_attempt";
|
|
4386
|
+
if (!sc.rubric) return "successCriteria.rubric is required for judge_eval_attempt";
|
|
4431
4387
|
return validateRubricWeights(sc.rubric);
|
|
4432
4388
|
}
|
|
4433
|
-
|
|
4434
|
-
* Output cross-field invariants the schema cannot express:
|
|
4435
|
-
*
|
|
4436
|
-
* 1. `results.length === input.runTaskIds.length` — every variant
|
|
4437
|
-
* the imposer asked for must be graded. Partial grading
|
|
4438
|
-
* invalidates cross-variant comparison; fail the whole task
|
|
4439
|
-
* rather than silently report a subset.
|
|
4440
|
-
*
|
|
4441
|
-
* 2. `results[i].runTaskId === input.runTaskIds[i]` — order is
|
|
4442
|
-
* load-bearing for downstream consumers (e.g. deltas keyed by
|
|
4443
|
-
* adjacent pairs). Mismatch is an LLM bug; reject loudly.
|
|
4444
|
-
*
|
|
4445
|
-
* 3. Each `result.scores` follows the same `llm_checklist` rule
|
|
4446
|
-
* `judge_pack` enforces (#999): if a score has an `assertions`
|
|
4447
|
-
* array, the numeric score MUST be `1` iff every assertion
|
|
4448
|
-
* passes. Inconsistent payloads pollute attestations.
|
|
4449
|
-
*
|
|
4450
|
-
* 4. Each `result.composite` MUST equal the rubric-weighted sum
|
|
4451
|
-
* `Σ(weight_j × scores[j].score)`. The parent (and any subagent
|
|
4452
|
-
* it delegated to) is supposed to compute this; surfacing a
|
|
4453
|
-
* drift here catches LLMs that hand-wave the arithmetic.
|
|
4454
|
-
*
|
|
4455
|
-
* 5. Optional `deltas` keys MUST be of the form `"A - B"` where
|
|
4456
|
-
* both `A` and `B` are variantLabels present in `results`.
|
|
4457
|
-
* Values are not range-checked (any float in [-1, 1] is
|
|
4458
|
-
* arithmetically possible).
|
|
4459
|
-
*/
|
|
4460
|
-
function validateJudgeEvalVariantOutput(output, input) {
|
|
4389
|
+
function validateJudgeEvalAttemptOutput(output, input) {
|
|
4461
4390
|
const out = output;
|
|
4462
4391
|
const inp = input;
|
|
4463
4392
|
if (inp) {
|
|
4464
|
-
if (out.
|
|
4465
|
-
|
|
4466
|
-
}
|
|
4467
|
-
for (let
|
|
4468
|
-
const
|
|
4469
|
-
|
|
4470
|
-
|
|
4471
|
-
|
|
4472
|
-
|
|
4473
|
-
const expected = allPassed ? 1 : 0;
|
|
4474
|
-
if (sc.score !== expected) return `results[${r}].scores[${s}] (criterionId="${sc.criterionId}"): assertions ${allPassed ? "all pass" : "have at least one fail"} but score=${sc.score}. Score must be derived: 1 iff every assertion passes, else 0 (#999 llm_checklist rule).`;
|
|
4475
|
-
}
|
|
4393
|
+
if (out.targetTaskId !== inp.targetTaskId) return `output.targetTaskId (${out.targetTaskId}) does not match input.targetTaskId (${inp.targetTaskId})`;
|
|
4394
|
+
if (out.targetAttemptN !== inp.targetAttemptN) return `output.targetAttemptN (${out.targetAttemptN}) does not match input.targetAttemptN (${inp.targetAttemptN})`;
|
|
4395
|
+
}
|
|
4396
|
+
for (let s = 0; s < out.scores.length; s++) {
|
|
4397
|
+
const sc = out.scores[s];
|
|
4398
|
+
if (!sc.assertions) continue;
|
|
4399
|
+
const allPassed = sc.assertions.every((a) => a.passed);
|
|
4400
|
+
const expected = allPassed ? 1 : 0;
|
|
4401
|
+
if (sc.score !== expected) return `scores[${s}] (criterionId="${sc.criterionId}"): assertions ${allPassed ? "all pass" : "have at least one fail"} but score=${sc.score}. Score must be 1 iff every assertion passes, else 0.`;
|
|
4476
4402
|
}
|
|
4477
4403
|
if (inp?.successCriteria?.rubric) {
|
|
4478
4404
|
const criteria = inp.successCriteria.rubric.criteria;
|
|
4479
4405
|
const weightById = new Map(criteria.map((c) => [c.id, c.weight]));
|
|
4480
|
-
|
|
4481
|
-
|
|
4482
|
-
|
|
4483
|
-
|
|
4484
|
-
|
|
4485
|
-
if (w === void 0) return `results[${r}].scores: criterionId "${sc.criterionId}" is not in the input rubric (known: ${Array.from(weightById.keys()).join(", ")}). Score every rubric criterion exactly once; do not invent new ids.`;
|
|
4486
|
-
sum += w * sc.score;
|
|
4487
|
-
}
|
|
4488
|
-
if (Math.abs(sum - result.composite) > .001) return `results[${r}].composite (${result.composite}) does not match Σ(weight × score) (${sum.toFixed(6)}). Composite must be the rubric-weighted sum of per-criterion scores (drift > 0.001).`;
|
|
4489
|
-
}
|
|
4490
|
-
}
|
|
4491
|
-
if (out.deltas) {
|
|
4492
|
-
const labels = new Set(out.results.map((r) => r.variantLabel));
|
|
4493
|
-
for (const key of Object.keys(out.deltas)) {
|
|
4494
|
-
const m = /^(.+?) - (.+)$/.exec(key);
|
|
4495
|
-
if (!m) return `deltas key "${key}" is not of the form "<variantLabel-A> - <variantLabel-B>". Use a single space-hyphen-space separator between labels.`;
|
|
4496
|
-
const [, a, b] = m;
|
|
4497
|
-
if (!labels.has(a) || !labels.has(b)) return `deltas key "${key}" references variantLabel(s) not present in results: ${!labels.has(a) ? `"${a}" missing` : ""}${!labels.has(a) && !labels.has(b) ? ", " : ""}${!labels.has(b) ? `"${b}" missing` : ""}`;
|
|
4406
|
+
let sum = 0;
|
|
4407
|
+
for (const sc of out.scores) {
|
|
4408
|
+
const w = weightById.get(sc.criterionId);
|
|
4409
|
+
if (w === void 0) return `scores references unknown criterionId "${sc.criterionId}"`;
|
|
4410
|
+
sum += w * sc.score;
|
|
4498
4411
|
}
|
|
4412
|
+
const rounded = Math.round(sum * 1e3) / 1e3;
|
|
4413
|
+
if (Math.abs(rounded - out.composite) > .001) return `composite (${out.composite}) does not match weighted rubric sum (${rounded})`;
|
|
4499
4414
|
}
|
|
4500
4415
|
return null;
|
|
4501
4416
|
}
|
|
4502
|
-
|
|
4503
|
-
|
|
4504
|
-
* equality. Recursively sorts object keys; arrays preserve order
|
|
4505
|
-
* (intentional — rubric criteria order is semantically meaningful).
|
|
4506
|
-
* Mirrors the canonical-JSON shape `crypto-service` uses for CIDs,
|
|
4507
|
-
* without taking on a crypto-service dep just for this comparison.
|
|
4508
|
-
*/
|
|
4509
|
-
function stableStringify(value) {
|
|
4510
|
-
if (value === null || typeof value !== "object") return JSON.stringify(value);
|
|
4511
|
-
if (Array.isArray(value)) return "[" + value.map(stableStringify).join(",") + "]";
|
|
4512
|
-
const obj = value;
|
|
4513
|
-
return "{" + Object.keys(obj).sort().map((k) => JSON.stringify(k) + ":" + stableStringify(obj[k])).join(",") + "}";
|
|
4514
|
-
}
|
|
4515
|
-
/**
|
|
4516
|
-
* Async preflight for `judge_eval_variant` (#1096 + #943):
|
|
4517
|
-
*
|
|
4518
|
-
* 1. Every `runTaskIds[i]` resolves to a task the caller can read.
|
|
4519
|
-
* 2. Every resolved task is `taskType === 'run_eval'`.
|
|
4520
|
-
* 3. Every resolved task is `status === 'completed'` with a
|
|
4521
|
-
* non-null `acceptedAttemptN` — grading an unaccepted attempt
|
|
4522
|
-
* races with re-attempts and pollutes the judge attestation.
|
|
4523
|
-
* 4. Every resolved task shares a non-null `correlationId`, and all
|
|
4524
|
-
* `correlationId`s are equal. Without this an imposer could
|
|
4525
|
-
* fabricate a "variant set" by stapling unrelated runs together.
|
|
4526
|
-
* 5. The shared `correlationId` is NOT already sealed. A previous
|
|
4527
|
-
* judge_eval_variant against the same group is final; produce a
|
|
4528
|
-
* fresh correlation_id for a new judging round rather than
|
|
4529
|
-
* adding contradictory verdicts to a sealed group.
|
|
4530
|
-
* 6. Every variant's `input.successCriteria` is byte-identical (via
|
|
4531
|
-
* stable-stringify). Different rubrics across "variants" makes
|
|
4532
|
-
* the comparison meaningless.
|
|
4533
|
-
*/
|
|
4534
|
-
async function validateJudgeEvalVariantInputAsync(input, ctx) {
|
|
4535
|
-
const { runTaskIds } = input;
|
|
4417
|
+
async function validateJudgeEvalAttemptInputAsync(input, ctx) {
|
|
4418
|
+
const inp = input;
|
|
4536
4419
|
const errors = [];
|
|
4537
|
-
const
|
|
4538
|
-
|
|
4539
|
-
|
|
4540
|
-
|
|
4541
|
-
|
|
4542
|
-
|
|
4543
|
-
|
|
4544
|
-
|
|
4545
|
-
field: `runTaskIds[${i}]`,
|
|
4546
|
-
message: `runTaskIds[${i}]=${runTaskIds[i]} does not resolve to a task you can read`
|
|
4547
|
-
});
|
|
4548
|
-
continue;
|
|
4549
|
-
}
|
|
4550
|
-
presentTargets.push(t);
|
|
4551
|
-
if (t.taskType !== "run_eval") errors.push({
|
|
4552
|
-
field: `runTaskIds[${i}]`,
|
|
4553
|
-
message: `runTaskIds[${i}]=${runTaskIds[i]} is a ${t.taskType}, not a run_eval`
|
|
4554
|
-
});
|
|
4555
|
-
if (t.status !== "completed" || t.acceptedAttemptN === null) errors.push({
|
|
4556
|
-
field: `runTaskIds[${i}]`,
|
|
4557
|
-
message: `runTaskIds[${i}]=${runTaskIds[i]} is not completed with an accepted attempt (status=${t.status}, acceptedAttemptN=${t.acceptedAttemptN})`
|
|
4558
|
-
});
|
|
4559
|
-
}
|
|
4560
|
-
if (missingTargets || presentTargets.length === 0) return errors;
|
|
4561
|
-
const correlationIds = new Set(presentTargets.map((t) => t.correlationId ?? "__null__"));
|
|
4562
|
-
if (correlationIds.has("__null__")) errors.push({
|
|
4563
|
-
field: "runTaskIds",
|
|
4564
|
-
message: "one or more run_eval targets have no correlation_id; cannot group as variants"
|
|
4420
|
+
const target = await ctx.resolveTask(inp.targetTaskId);
|
|
4421
|
+
if (!target) return [{
|
|
4422
|
+
field: "targetTaskId",
|
|
4423
|
+
message: `targetTaskId=${inp.targetTaskId} does not resolve to a task you can read`
|
|
4424
|
+
}];
|
|
4425
|
+
if (target.taskType !== "run_eval") errors.push({
|
|
4426
|
+
field: "targetTaskId",
|
|
4427
|
+
message: `targetTaskId=${inp.targetTaskId} is a ${target.taskType}, not a run_eval`
|
|
4565
4428
|
});
|
|
4566
|
-
if (
|
|
4567
|
-
field: "
|
|
4568
|
-
message: `
|
|
4429
|
+
if (target.status !== "completed" || target.acceptedAttemptN === null) errors.push({
|
|
4430
|
+
field: "targetTaskId",
|
|
4431
|
+
message: `targetTaskId=${inp.targetTaskId} is not completed with an accepted attempt (status=${target.status}, acceptedAttemptN=${target.acceptedAttemptN})`
|
|
4569
4432
|
});
|
|
4570
|
-
if (
|
|
4571
|
-
|
|
4572
|
-
|
|
4573
|
-
|
|
4574
|
-
if (
|
|
4575
|
-
field: "
|
|
4576
|
-
message:
|
|
4433
|
+
else if (target.acceptedAttemptN !== inp.targetAttemptN) errors.push({
|
|
4434
|
+
field: "targetAttemptN",
|
|
4435
|
+
message: `targetAttemptN=${inp.targetAttemptN} does not match the producer's acceptedAttemptN=${target.acceptedAttemptN}`
|
|
4436
|
+
});
|
|
4437
|
+
if (!target.correlationId) errors.push({
|
|
4438
|
+
field: "targetTaskId",
|
|
4439
|
+
message: "target run_eval has no correlation_id; cannot enforce duplicate-judge protection"
|
|
4440
|
+
});
|
|
4441
|
+
if (errors.length > 0 || !target.correlationId) return errors;
|
|
4442
|
+
const rubric = inp.successCriteria.rubric;
|
|
4443
|
+
const duplicate = (await ctx.listTasksByCorrelation(target.correlationId)).find((task) => {
|
|
4444
|
+
if (task.taskType !== "judge_eval_attempt") return false;
|
|
4445
|
+
if (task.status === "failed" || task.status === "cancelled" || task.status === "expired") return false;
|
|
4446
|
+
const existing = task.input;
|
|
4447
|
+
const existingRubric = existing.successCriteria?.rubric;
|
|
4448
|
+
return existing.targetTaskId === inp.targetTaskId && existing.targetAttemptN === inp.targetAttemptN && existingRubric?.rubricId === rubric?.rubricId && existingRubric?.version === rubric?.version;
|
|
4449
|
+
});
|
|
4450
|
+
if (duplicate) errors.push({
|
|
4451
|
+
field: "targetTaskId",
|
|
4452
|
+
message: `judge task ${duplicate.id} already exists for (${inp.targetTaskId}, attempt ${inp.targetAttemptN}, rubric ${rubric?.rubricId}@${rubric?.version})`
|
|
4577
4453
|
});
|
|
4578
|
-
const first = stableStringify(presentTargets[0].input.successCriteria);
|
|
4579
|
-
for (let i = 1; i < presentTargets.length; i++) if (stableStringify(presentTargets[i].input.successCriteria) !== first) {
|
|
4580
|
-
errors.push({
|
|
4581
|
-
field: `runTaskIds[${i}]`,
|
|
4582
|
-
message: `runTaskIds[${i}] has a different input.successCriteria than runTaskIds[0]; all variants must share the rubric and gates`
|
|
4583
|
-
});
|
|
4584
|
-
break;
|
|
4585
|
-
}
|
|
4586
4454
|
return errors;
|
|
4587
4455
|
}
|
|
4588
|
-
|
|
4589
|
-
|
|
4590
|
-
|
|
4591
|
-
|
|
4592
|
-
* concurrent second `judge_eval_variant` against the same group
|
|
4593
|
-
* loses the race and is rejected with a clean conflict error.
|
|
4594
|
-
*
|
|
4595
|
-
* The seal applies to the SHARED correlation_id of the targets —
|
|
4596
|
-
* NOT to the judge task's own correlationId (which is typically
|
|
4597
|
-
* null or distinct). The task service derives the correlationId
|
|
4598
|
-
* for the effect from the resolved targets, not from the judge
|
|
4599
|
-
* task row.
|
|
4600
|
-
*/
|
|
4601
|
-
async function onCreateJudgeEvalVariant(input, ctx) {
|
|
4602
|
-
const { runTaskIds } = input;
|
|
4603
|
-
const first = await ctx.resolveTask(runTaskIds[0]);
|
|
4604
|
-
if (!first?.correlationId) return [];
|
|
4456
|
+
async function onCreateJudgeEvalAttempt(input, _ctx) {
|
|
4457
|
+
const judge = input;
|
|
4458
|
+
const rubric = judge.successCriteria.rubric;
|
|
4459
|
+
if (!rubric) return [];
|
|
4605
4460
|
return [{
|
|
4606
|
-
kind: "
|
|
4607
|
-
|
|
4461
|
+
kind: "guardTaskUniqueness",
|
|
4462
|
+
taskType: JUDGE_EVAL_ATTEMPT_TYPE,
|
|
4463
|
+
lockKey: [
|
|
4464
|
+
JUDGE_EVAL_ATTEMPT_TYPE,
|
|
4465
|
+
judge.targetTaskId,
|
|
4466
|
+
String(judge.targetAttemptN),
|
|
4467
|
+
rubric.rubricId,
|
|
4468
|
+
rubric.version
|
|
4469
|
+
].join(":"),
|
|
4470
|
+
inputMatches: [
|
|
4471
|
+
{
|
|
4472
|
+
path: ["targetTaskId"],
|
|
4473
|
+
value: judge.targetTaskId
|
|
4474
|
+
},
|
|
4475
|
+
{
|
|
4476
|
+
path: ["targetAttemptN"],
|
|
4477
|
+
value: judge.targetAttemptN
|
|
4478
|
+
},
|
|
4479
|
+
{
|
|
4480
|
+
path: [
|
|
4481
|
+
"successCriteria",
|
|
4482
|
+
"rubric",
|
|
4483
|
+
"rubricId"
|
|
4484
|
+
],
|
|
4485
|
+
value: rubric.rubricId
|
|
4486
|
+
},
|
|
4487
|
+
{
|
|
4488
|
+
path: [
|
|
4489
|
+
"successCriteria",
|
|
4490
|
+
"rubric",
|
|
4491
|
+
"version"
|
|
4492
|
+
],
|
|
4493
|
+
value: rubric.version
|
|
4494
|
+
}
|
|
4495
|
+
]
|
|
4608
4496
|
}];
|
|
4609
4497
|
}
|
|
4610
4498
|
//#endregion
|
|
@@ -4728,14 +4616,43 @@ async function validateRenderPackInputAsync(input, ctx) {
|
|
|
4728
4616
|
//#region ../../libs/tasks/src/task-types/run-eval.ts
|
|
4729
4617
|
/**
|
|
4730
4618
|
* `run_eval` — execute a scenario prompt under a named variant for
|
|
4731
|
-
* later
|
|
4619
|
+
* later per-attempt grading by `judge_eval_attempt` tasks.
|
|
4732
4620
|
*
|
|
4733
4621
|
* output_kind: artifact
|
|
4734
|
-
* criteria: optional (when set,
|
|
4735
|
-
*
|
|
4622
|
+
* criteria: optional producer-only checks (when set,
|
|
4623
|
+
* output.verification is required — the judge rubric remains hidden
|
|
4624
|
+
* on downstream `judge_eval_attempt` tasks)
|
|
4736
4625
|
* references: not required (scenario lives entirely in input)
|
|
4737
4626
|
*/
|
|
4738
4627
|
var RUN_EVAL_TYPE = "run_eval";
|
|
4628
|
+
var RunEvalMode = Type$2.Union([Type$2.Literal("vitro"), Type$2.Literal("vivo")], { $id: "RunEvalMode" });
|
|
4629
|
+
var RunEvalWorkspace = Type$2.Union([
|
|
4630
|
+
Type$2.Literal("none"),
|
|
4631
|
+
Type$2.Literal("shared_mount"),
|
|
4632
|
+
Type$2.Literal("dedicated_worktree")
|
|
4633
|
+
], { $id: "RunEvalWorkspace" });
|
|
4634
|
+
var RunEvalExecution = Type$2.Object({
|
|
4635
|
+
mode: RunEvalMode,
|
|
4636
|
+
workspace: RunEvalWorkspace
|
|
4637
|
+
}, {
|
|
4638
|
+
$id: "RunEvalExecution",
|
|
4639
|
+
additionalProperties: false
|
|
4640
|
+
});
|
|
4641
|
+
/**
|
|
4642
|
+
* Producer-visible checks for `run_eval`. Deliberately forbids `rubric`
|
|
4643
|
+
* so the variant runner cannot see the downstream judge's answer key.
|
|
4644
|
+
* Keep the rest of the SuccessCriteria envelope available for generic
|
|
4645
|
+
* process / structure checks (`gates`, `assertions`, `sideEffects`).
|
|
4646
|
+
*/
|
|
4647
|
+
var RunEvalSuccessCriteria = Type$2.Object({
|
|
4648
|
+
version: Type$2.Literal(1),
|
|
4649
|
+
gates: Type$2.Optional(SuccessCriteria.properties.gates),
|
|
4650
|
+
assertions: Type$2.Optional(SuccessCriteria.properties.assertions),
|
|
4651
|
+
sideEffects: Type$2.Optional(SuccessCriteria.properties.sideEffects)
|
|
4652
|
+
}, {
|
|
4653
|
+
$id: "RunEvalSuccessCriteria",
|
|
4654
|
+
additionalProperties: false
|
|
4655
|
+
});
|
|
4739
4656
|
var RunEvalInput = Type$2.Object({
|
|
4740
4657
|
scenario: Type$2.Object({
|
|
4741
4658
|
prompt: Type$2.String({ minLength: 1 }),
|
|
@@ -4745,8 +4662,9 @@ var RunEvalInput = Type$2.Object({
|
|
|
4745
4662
|
minLength: 1,
|
|
4746
4663
|
maxLength: 64
|
|
4747
4664
|
}),
|
|
4665
|
+
execution: RunEvalExecution,
|
|
4748
4666
|
context: TaskContext,
|
|
4749
|
-
successCriteria: Type$2.Optional(
|
|
4667
|
+
successCriteria: Type$2.Optional(RunEvalSuccessCriteria)
|
|
4750
4668
|
}, {
|
|
4751
4669
|
$id: "RunEvalInput",
|
|
4752
4670
|
additionalProperties: false
|
|
@@ -4774,8 +4692,8 @@ var RunEvalOutput = Type$2.Object({
|
|
|
4774
4692
|
function validateRunEvalOutput(output, input) {
|
|
4775
4693
|
const hasCriteria = input !== null && input !== void 0 && input.successCriteria !== void 0;
|
|
4776
4694
|
const hasVerification = output !== null && output !== void 0 && output.verification !== void 0;
|
|
4777
|
-
if (hasCriteria && !hasVerification) return "output.verification is required because input.successCriteria is set; the producer LLM must self-assess against the
|
|
4778
|
-
if (!hasCriteria && hasVerification) return "output.verification was supplied but input.successCriteria is unset; omit verification when there are no
|
|
4695
|
+
if (hasCriteria && !hasVerification) return "output.verification is required because input.successCriteria is set; the producer LLM must self-assess against the producer checks";
|
|
4696
|
+
if (!hasCriteria && hasVerification) return "output.verification was supplied but input.successCriteria is unset; omit verification when there are no producer checks to assess against";
|
|
4779
4697
|
return null;
|
|
4780
4698
|
}
|
|
4781
4699
|
//#endregion
|
|
@@ -4891,24 +4809,24 @@ var BUILT_IN_TASK_TYPES = {
|
|
|
4891
4809
|
inputSchema: RunEvalInput,
|
|
4892
4810
|
outputSchema: RunEvalOutput,
|
|
4893
4811
|
outputKind: "artifact",
|
|
4894
|
-
|
|
4812
|
+
resumable: true,
|
|
4813
|
+
workspaceScope: "session",
|
|
4895
4814
|
sessionScope: "custom",
|
|
4896
4815
|
requiresReferences: false,
|
|
4897
4816
|
validateOutput: validateRunEvalOutput
|
|
4898
4817
|
},
|
|
4899
|
-
[
|
|
4900
|
-
name:
|
|
4901
|
-
inputSchema:
|
|
4902
|
-
outputSchema:
|
|
4818
|
+
[JUDGE_EVAL_ATTEMPT_TYPE]: {
|
|
4819
|
+
name: JUDGE_EVAL_ATTEMPT_TYPE,
|
|
4820
|
+
inputSchema: JudgeEvalAttemptInput,
|
|
4821
|
+
outputSchema: JudgeEvalAttemptOutput,
|
|
4903
4822
|
outputKind: "judgment",
|
|
4904
4823
|
workspaceScope: "attempt",
|
|
4905
|
-
sessionScope: "
|
|
4824
|
+
sessionScope: "none",
|
|
4906
4825
|
requiresReferences: false,
|
|
4907
|
-
validateInput:
|
|
4908
|
-
validateOutput:
|
|
4909
|
-
validateInputAsync:
|
|
4910
|
-
onCreate:
|
|
4911
|
-
usesSubagents: true
|
|
4826
|
+
validateInput: validateJudgeEvalAttemptInput,
|
|
4827
|
+
validateOutput: validateJudgeEvalAttemptOutput,
|
|
4828
|
+
validateInputAsync: validateJudgeEvalAttemptInputAsync,
|
|
4829
|
+
onCreate: onCreateJudgeEvalAttempt
|
|
4912
4830
|
}
|
|
4913
4831
|
};
|
|
4914
4832
|
//#endregion
|
|
@@ -6258,6 +6176,11 @@ var PROMPT_SEPARATOR = "\n\n---\n\n";
|
|
|
6258
6176
|
* - `skill` → `deliver.skill({ slug, content })` once per ref.
|
|
6259
6177
|
* Slug collisions on distinct contents are
|
|
6260
6178
|
* refused loudly.
|
|
6179
|
+
* - `context_inline`→ persist raw bytes via `deliver.contextFile(...)`
|
|
6180
|
+
* and inject them into the prompt in an explicit,
|
|
6181
|
+
* named block. Intended for eval/context experiments
|
|
6182
|
+
* where the content must be in the model context
|
|
6183
|
+
* window, not merely discoverable as a skill.
|
|
6261
6184
|
* - `prompt_prefix` → content appended to `systemPromptPrefix` with
|
|
6262
6185
|
* the canonical `\n\n---\n\n` separator (in
|
|
6263
6186
|
* declared order).
|
|
@@ -6290,6 +6213,13 @@ async function resolveTaskContext(args) {
|
|
|
6290
6213
|
slug: ref.slug,
|
|
6291
6214
|
content: ref.content
|
|
6292
6215
|
});
|
|
6216
|
+
} else if (ref.binding === "context_inline") {
|
|
6217
|
+
await args.deliver.contextFile({
|
|
6218
|
+
slug: ref.slug,
|
|
6219
|
+
content: ref.content,
|
|
6220
|
+
suggestedFileName: `${ref.slug}.md`
|
|
6221
|
+
});
|
|
6222
|
+
promptParts.push(formatInlineContextBlock(ref.slug, ref.content));
|
|
6293
6223
|
} else if (ref.binding === "prompt_prefix") promptParts.push(ref.content);
|
|
6294
6224
|
else userParts.push(ref.content);
|
|
6295
6225
|
injected.push(ref);
|
|
@@ -6300,6 +6230,23 @@ async function resolveTaskContext(args) {
|
|
|
6300
6230
|
userInlineSuffix: userParts.join(PROMPT_SEPARATOR)
|
|
6301
6231
|
};
|
|
6302
6232
|
}
|
|
6233
|
+
function formatInlineContextBlock(slug, content) {
|
|
6234
|
+
return [
|
|
6235
|
+
"### Injected Task Context",
|
|
6236
|
+
"",
|
|
6237
|
+
`Context id: \`${slug}\``,
|
|
6238
|
+
"The following raw context was supplied by the task creator. Treat it",
|
|
6239
|
+
"as task-relevant background that may override generic coding instincts",
|
|
6240
|
+
"when it contains repo- or workflow-specific constraints.",
|
|
6241
|
+
"The same content is also materialized in the workspace as",
|
|
6242
|
+
"`/workspace/context-pack.md` and mirrored in `AGENTS.md` for",
|
|
6243
|
+
"repo-context discovery.",
|
|
6244
|
+
"",
|
|
6245
|
+
"<context>",
|
|
6246
|
+
content,
|
|
6247
|
+
"</context>"
|
|
6248
|
+
].join("\n");
|
|
6249
|
+
}
|
|
6303
6250
|
//#endregion
|
|
6304
6251
|
//#region ../../libs/agent-runtime/src/output-tools.ts
|
|
6305
6252
|
/**
|
|
@@ -6365,20 +6312,16 @@ function buildFinalOutputBlock(opts) {
|
|
|
6365
6312
|
"## Final output (read this carefully)",
|
|
6366
6313
|
"",
|
|
6367
6314
|
`Your VERY LAST action in this conversation MUST report the structured`,
|
|
6368
|
-
`output matching \`${outputSchemaName}
|
|
6369
|
-
`preference:`,
|
|
6315
|
+
`output matching \`${outputSchemaName}\`.`,
|
|
6370
6316
|
"",
|
|
6371
|
-
`
|
|
6372
|
-
`
|
|
6373
|
-
`
|
|
6374
|
-
`
|
|
6375
|
-
`
|
|
6376
|
-
` \`${outputSchemaName}\`. No prose before or after. No code fences.`,
|
|
6377
|
-
` No "ok" or "done". The runtime parses the last balanced top-level`,
|
|
6378
|
-
` JSON object as the output.`,
|
|
6317
|
+
`Call \`${submitTool}\` exactly once with the payload.`,
|
|
6318
|
+
`The runtime captures the validated arguments and ends the session.`,
|
|
6319
|
+
`Do NOT emit the output as plain assistant text. Do NOT rely on a`,
|
|
6320
|
+
`JSON-in-message fallback. If you do not call \`${submitTool}\`, the`,
|
|
6321
|
+
`attempt fails even if the underlying work succeeded.`,
|
|
6379
6322
|
"",
|
|
6380
|
-
`
|
|
6381
|
-
`
|
|
6323
|
+
`Your final assistant text before that tool call may explain your work,`,
|
|
6324
|
+
`but the submit-tool call itself must be your VERY LAST action.`,
|
|
6382
6325
|
"",
|
|
6383
6326
|
`Output shape:`,
|
|
6384
6327
|
"",
|
|
@@ -6516,21 +6459,30 @@ function buildAssessBriefUserPrompt(input, ctx) {
|
|
|
6516
6459
|
}
|
|
6517
6460
|
//#endregion
|
|
6518
6461
|
//#region ../../libs/agent-runtime/src/prompts/self-verification.ts
|
|
6519
|
-
function buildSelfVerificationBlock(taskId) {
|
|
6462
|
+
function buildSelfVerificationBlock(taskId, criteriaField = "successCriteria") {
|
|
6520
6463
|
return [
|
|
6521
6464
|
"## Self-verification",
|
|
6522
6465
|
"",
|
|
6523
|
-
`
|
|
6466
|
+
`If \`input.${criteriaField}\` is set on this task, your final output MUST`,
|
|
6467
|
+
"include a `verification` block. **The runtime/server rejects task",
|
|
6468
|
+
`submission without \`verification\` when \`${criteriaField}\` is present**`,
|
|
6469
|
+
"— the request fails validation and the attempt is discarded, even if the",
|
|
6470
|
+
"underlying work succeeded. Do not call the submit tool until you have",
|
|
6471
|
+
"computed the verification payload.",
|
|
6472
|
+
"",
|
|
6473
|
+
`Call \`moltnet_get_task\` with task id \`${taskId}\` and read \`input.${criteriaField}\`.`,
|
|
6524
6474
|
"",
|
|
6525
|
-
|
|
6475
|
+
`- If \`input.${criteriaField}\` is **absent**, omit \`verification\` from your`,
|
|
6526
6476
|
" final output entirely.",
|
|
6527
|
-
|
|
6528
|
-
" `verification` block in your final output. Evaluate every applicable",
|
|
6477
|
+
`- If \`input.${criteriaField}\` is **present**, evaluate every applicable`,
|
|
6529
6478
|
" item — `gates`, `assertions`, `rubric` criteria, `sideEffects` — against",
|
|
6530
6479
|
" your produced work and emit one result per id. Be honest: a `fail` with",
|
|
6531
6480
|
" a one-line reason is more useful than a false `pass`. Use `skip` (with a",
|
|
6532
6481
|
" `detail`) when you genuinely could not determine a result. Compute",
|
|
6533
6482
|
" `passed = results.every(r => r.status !== 'fail')`.",
|
|
6483
|
+
"- `verification` MUST be a JSON object. Never send a string, markdown",
|
|
6484
|
+
" block, null, or an empty placeholder. The submit tool expects an object",
|
|
6485
|
+
" with `inputCid`, `results`, and `passed` fields.",
|
|
6534
6486
|
"",
|
|
6535
6487
|
"Verification shape:",
|
|
6536
6488
|
"",
|
|
@@ -6544,6 +6496,23 @@ function buildSelfVerificationBlock(taskId) {
|
|
|
6544
6496
|
" \"passed\": <boolean>",
|
|
6545
6497
|
"}",
|
|
6546
6498
|
"```",
|
|
6499
|
+
"",
|
|
6500
|
+
"Minimal valid example:",
|
|
6501
|
+
"",
|
|
6502
|
+
"```json",
|
|
6503
|
+
"{",
|
|
6504
|
+
" \"inputCid\": \"<task inputCid>\",",
|
|
6505
|
+
" \"results\": [",
|
|
6506
|
+
" {",
|
|
6507
|
+
" \"id\": \"<criterion id>\",",
|
|
6508
|
+
" \"kind\": \"rubric\",",
|
|
6509
|
+
" \"status\": \"pass\",",
|
|
6510
|
+
" \"detail\": \"one-line reason\"",
|
|
6511
|
+
" }",
|
|
6512
|
+
" ],",
|
|
6513
|
+
" \"passed\": true",
|
|
6514
|
+
"}",
|
|
6515
|
+
"```",
|
|
6547
6516
|
""
|
|
6548
6517
|
].join("\n");
|
|
6549
6518
|
}
|
|
@@ -6794,69 +6763,62 @@ function buildFulfillBriefUserPrompt(input, ctx) {
|
|
|
6794
6763
|
].filter(Boolean).join("\n");
|
|
6795
6764
|
}
|
|
6796
6765
|
//#endregion
|
|
6797
|
-
//#region ../../libs/agent-runtime/src/prompts/judge-eval-
|
|
6798
|
-
|
|
6799
|
-
|
|
6800
|
-
|
|
6801
|
-
*
|
|
6802
|
-
* The parent agent's job is **fan-out-and-collect**: for each
|
|
6803
|
-
* `runTaskIds[i]`, spawn an isolated subagent via the `subagent` custom
|
|
6804
|
-
* tool (#1087), have it grade that variant against the shared rubric,
|
|
6805
|
-
* and collect each subagent's structured `judge_eval_variant_result`
|
|
6806
|
-
* payload. The parent does NOT grade itself; it composes the per-
|
|
6807
|
-
* variant results into the final `judge_eval_variant` output (results
|
|
6808
|
-
* array + optional deltas + verdicts).
|
|
6809
|
-
*
|
|
6810
|
-
* Isolation is the point: each variant gets a fresh subagent session
|
|
6811
|
-
* with no carryover context from sibling variants, so per-variant
|
|
6812
|
-
* grading is independent. Cost is bounded by `maxItems: 10` on
|
|
6813
|
-
* runTaskIds.
|
|
6814
|
-
*/
|
|
6815
|
-
function buildJudgeEvalVariantUserPrompt(input, ctx) {
|
|
6816
|
-
const { runTaskIds, successCriteria } = input;
|
|
6817
|
-
const rubric = successCriteria.rubric;
|
|
6818
|
-
if (!rubric) throw new Error("judge_eval_variant requires successCriteria.rubric — none present");
|
|
6766
|
+
//#region ../../libs/agent-runtime/src/prompts/judge-eval-attempt.ts
|
|
6767
|
+
function buildJudgeEvalAttemptUserPrompt(input, ctx) {
|
|
6768
|
+
const rubric = input.successCriteria.rubric;
|
|
6769
|
+
if (!rubric) throw new Error("judge_eval_attempt requires successCriteria.rubric — none present");
|
|
6819
6770
|
const escapeCell = (s) => s.replace(/\\/g, "\\\\").replace(/\|/g, "\\|").replace(/\r?\n/g, " ");
|
|
6820
6771
|
const criteriaTable = rubric.criteria.map((c) => `| \`${c.id}\` | ${c.weight.toFixed(3)} | ${c.scoring} | ${escapeCell(c.description)} |`).join("\n");
|
|
6821
|
-
const targetsBlock = runTaskIds.map((id, i) => `${i + 1}. \`${id}\``).join("\n");
|
|
6822
6772
|
const finalOutputBlock = buildFinalOutputBlock({
|
|
6823
|
-
taskType: "
|
|
6824
|
-
outputSchemaName: "
|
|
6773
|
+
taskType: "judge_eval_attempt",
|
|
6774
|
+
outputSchemaName: "JudgeEvalAttemptOutput",
|
|
6825
6775
|
shapeSketch: [
|
|
6826
6776
|
"{",
|
|
6827
|
-
|
|
6828
|
-
"
|
|
6829
|
-
"
|
|
6830
|
-
"
|
|
6831
|
-
"
|
|
6832
|
-
"
|
|
6833
|
-
" \"verdict\": \"<1-3 sentences>\"",
|
|
6834
|
-
" },",
|
|
6835
|
-
" ...one entry per runTaskIds[i], same order",
|
|
6836
|
-
" ],",
|
|
6837
|
-
" \"deltas\": { \"<labelA> - <labelB>\": <composite(A) - composite(B)> }, // optional",
|
|
6777
|
+
` "targetTaskId": "${input.targetTaskId}",`,
|
|
6778
|
+
` "targetAttemptN": ${input.targetAttemptN},`,
|
|
6779
|
+
" \"variantLabel\": \"<from producer input>\",",
|
|
6780
|
+
" \"scores\": [ { \"criterionId\": \"...\", \"score\": 0..1, \"rationale\": \"...\", \"assertions\": [...]? } ],",
|
|
6781
|
+
" \"composite\": <Σ(weight × score), 0..1>,",
|
|
6782
|
+
" \"verdict\": \"<1-3 sentences>\",",
|
|
6838
6783
|
" \"judgeModel\": \"<id>\", // optional",
|
|
6839
6784
|
" \"traceparent\": \"<from claim>\"",
|
|
6840
6785
|
"}"
|
|
6841
6786
|
].join("\n")
|
|
6842
6787
|
});
|
|
6788
|
+
const workspaceSection = ctx.workspace?.attached === true ? [
|
|
6789
|
+
"### Workspace",
|
|
6790
|
+
"",
|
|
6791
|
+
"Your current workspace is already attached to the producer attempt",
|
|
6792
|
+
"you are judging. Inspect files directly from the current workspace",
|
|
6793
|
+
"root instead of inventing synthetic `artifact_<taskId>` paths.",
|
|
6794
|
+
"If the accepted attempt output lists `artifacts[].path`, treat those",
|
|
6795
|
+
"paths as relative to the current workspace root unless the output",
|
|
6796
|
+
"explicitly says otherwise.",
|
|
6797
|
+
ctx.workspace.mode === "dedicated_worktree" ? `This attachment is a dedicated producer worktree${ctx.workspace.branch ? ` on branch \`${ctx.workspace.branch}\`` : ""}.` : ctx.workspace.mode === "scratch_mount" ? "This attachment is the producer scratch workspace mounted with shadow writes for safe inspection." : "This attachment is the producer shared workspace mounted with shadow writes for safe inspection.",
|
|
6798
|
+
""
|
|
6799
|
+
].join("\n") : "";
|
|
6843
6800
|
return [
|
|
6844
|
-
"# Judge Eval
|
|
6845
|
-
|
|
6846
|
-
"
|
|
6847
|
-
"grade yourself.",
|
|
6801
|
+
"# Judge Eval Attempt\n",
|
|
6802
|
+
"You are grading one accepted `run_eval` producer attempt against a hidden",
|
|
6803
|
+
"judge rubric. Do not delegate to subagents. Grade in this session only.",
|
|
6848
6804
|
"",
|
|
6849
6805
|
`Task id: \`${ctx.taskId}\``,
|
|
6850
6806
|
`Diary: \`${ctx.diaryId}\``,
|
|
6807
|
+
`Producer task: \`${input.targetTaskId}\``,
|
|
6808
|
+
`Producer attempt: \`${input.targetAttemptN}\``,
|
|
6851
6809
|
"",
|
|
6852
|
-
"###
|
|
6810
|
+
"### Evidence gathering",
|
|
6853
6811
|
"",
|
|
6854
|
-
|
|
6855
|
-
|
|
6856
|
-
|
|
6857
|
-
"
|
|
6858
|
-
"
|
|
6812
|
+
`1. Call \`moltnet_get_task\` with taskId=\`${input.targetTaskId}\`.`,
|
|
6813
|
+
`2. Call \`moltnet_list_task_attempts\` with taskId=\`${input.targetTaskId}\` and inspect the accepted attempt matching \`${input.targetAttemptN}\`.`,
|
|
6814
|
+
`3. Call \`moltnet_list_task_messages\` with taskId=\`${input.targetTaskId}\`, attemptN=\`${input.targetAttemptN}\` to inspect the producer's turn-by-turn behavior.`,
|
|
6815
|
+
"4. Use the accepted attempt output, attempt messages, and any accessible",
|
|
6816
|
+
" artifacts or workspace evidence available in your environment.",
|
|
6817
|
+
" Read artifact files from the mounted producer workspace when present;",
|
|
6818
|
+
" do not assume detached `artifact_<taskId>` directories exist.",
|
|
6819
|
+
"5. Score strictly against the rubric below.",
|
|
6859
6820
|
"",
|
|
6821
|
+
workspaceSection,
|
|
6860
6822
|
"### Rubric",
|
|
6861
6823
|
"",
|
|
6862
6824
|
rubric.preamble ? `${rubric.preamble}\n` : "",
|
|
@@ -6864,34 +6826,10 @@ function buildJudgeEvalVariantUserPrompt(input, ctx) {
|
|
|
6864
6826
|
"| --- | --- | --- | --- |",
|
|
6865
6827
|
criteriaTable,
|
|
6866
6828
|
"",
|
|
6867
|
-
"### How to grade",
|
|
6868
|
-
"",
|
|
6869
|
-
"For EACH `runTaskIds[i]`:",
|
|
6870
|
-
"",
|
|
6871
|
-
"1. Call the `subagent` custom tool with:",
|
|
6872
|
-
" - `task`: a brief instructing the subagent to grade ONLY that variant",
|
|
6873
|
-
" against the rubric above; include the target task id and the rubric",
|
|
6874
|
-
" verbatim. The subagent has the same MoltNet tools and can fetch the",
|
|
6875
|
-
" accepted attempt output independently.",
|
|
6876
|
-
" - `output_schema`: `\"judge_eval_variant_result\"`",
|
|
6877
|
-
"2. Receive the subagent's structured `judge_eval_variant_result` payload.",
|
|
6878
|
-
"3. Append it to your `results[]` array, **in the same order as input.runTaskIds**.",
|
|
6879
|
-
"",
|
|
6880
|
-
"Do NOT score any variant in your own session. The whole point of the",
|
|
6881
|
-
"subagent fan-out is per-variant context isolation — grading two variants",
|
|
6882
|
-
"back-to-back in one session lets the second be biased by the first.",
|
|
6883
|
-
"",
|
|
6884
6829
|
"### Composite arithmetic",
|
|
6885
6830
|
"",
|
|
6886
|
-
"
|
|
6887
|
-
"criteria. Drift > 0.001 is rejected.
|
|
6888
|
-
"themselves; double-check before assembling the final output.",
|
|
6889
|
-
"",
|
|
6890
|
-
"### Deltas (optional)",
|
|
6891
|
-
"",
|
|
6892
|
-
"If useful, populate `deltas` with pairwise composite differences keyed by",
|
|
6893
|
-
"`\"<variantLabel-A> - <variantLabel-B>\"` (single space-hyphen-space). Both",
|
|
6894
|
-
"labels must appear in `results`. Omit `deltas` entirely if not used.",
|
|
6831
|
+
"Your `composite` MUST equal `Σ(criterion.weight × score)` over the rubric",
|
|
6832
|
+
"criteria. Drift > 0.001 is rejected.",
|
|
6895
6833
|
"",
|
|
6896
6834
|
finalOutputBlock
|
|
6897
6835
|
].filter((s) => s !== "").join("\n");
|
|
@@ -7164,6 +7102,29 @@ function buildRenderPackUserPrompt(input, ctx) {
|
|
|
7164
7102
|
"- Do NOT write diary entries unless a genuine incident occurs",
|
|
7165
7103
|
" (rendering failure, invariant violation).",
|
|
7166
7104
|
"",
|
|
7105
|
+
"## Fidelity Discipline",
|
|
7106
|
+
"",
|
|
7107
|
+
"These rules apply when you are producing the markdown yourself rather",
|
|
7108
|
+
"than relying on a deterministic `server:*` renderer.",
|
|
7109
|
+
"",
|
|
7110
|
+
"1. Preserve hedges and qualifiers verbatim.",
|
|
7111
|
+
" Source phrases like \"typically\", \"roughly\", \"about half\",",
|
|
7112
|
+
" \"in this codebase\", \"for this speaker\", and \"on most slides\"",
|
|
7113
|
+
" are load-bearing. Keep them. Do not turn a partial observation",
|
|
7114
|
+
" into a universal claim by stripping the hedge.",
|
|
7115
|
+
"2. Signal list truncation explicitly.",
|
|
7116
|
+
" If a source entry enumerates examples and you shorten the list,",
|
|
7117
|
+
" say so with markers like `e.g.`, `among others`, or",
|
|
7118
|
+
" `[truncated - see source for full list]`. Do not silently drop",
|
|
7119
|
+
" items from an enumeration in a way that looks lossless.",
|
|
7120
|
+
"3. Calibrate against fidelity scoring.",
|
|
7121
|
+
" A paraphrased rendered pack will be audited claim-by-claim for",
|
|
7122
|
+
" drift on quotes, numbers, file paths, hedges, polarity, and list",
|
|
7123
|
+
" completeness. Optimize for \"no detectable drift across a",
|
|
7124
|
+
" claim-by-claim audit\", not \"shorter at any cost\". When compressing, prefer",
|
|
7125
|
+
" tightening prose around a quote rather than altering the quote,",
|
|
7126
|
+
" and prefer summarising a list over silently truncating it.",
|
|
7127
|
+
"",
|
|
7167
7128
|
buildSelfVerificationBlock(ctx.taskId),
|
|
7168
7129
|
buildFinalOutputBlock({
|
|
7169
7130
|
taskType: "render_pack",
|
|
@@ -7188,8 +7149,9 @@ function buildRenderPackUserPrompt(input, ctx) {
|
|
|
7188
7149
|
* Build the first user-message prompt for a `run_eval` task.
|
|
7189
7150
|
*
|
|
7190
7151
|
* Free-form: no git workflow, no commit ceremony. The executor produces
|
|
7191
|
-
* a textual response (and optional file artifacts) that
|
|
7192
|
-
* `
|
|
7152
|
+
* a textual response (and optional file artifacts) that later
|
|
7153
|
+
* `judge_eval_attempt` task(s) grade against their own hidden
|
|
7154
|
+
* rubric.
|
|
7193
7155
|
*
|
|
7194
7156
|
* Context delivery is handled by `resolveTaskContext` (see
|
|
7195
7157
|
* libs/agent-runtime/src/context-bindings.ts) and runs BEFORE this
|
|
@@ -7199,7 +7161,9 @@ function buildRenderPackUserPrompt(input, ctx) {
|
|
|
7199
7161
|
* builder does NOT inline `input.context[]` itself.
|
|
7200
7162
|
*/
|
|
7201
7163
|
function buildRunEvalUserPrompt(input, ctx) {
|
|
7202
|
-
const { scenario, variantLabel, successCriteria } = input;
|
|
7164
|
+
const { scenario, variantLabel, execution, successCriteria } = input;
|
|
7165
|
+
const hasContext = input.context.length > 0;
|
|
7166
|
+
const hasInlineContext = input.context.some((entry) => entry.binding === "context_inline");
|
|
7203
7167
|
const inputFilesSection = scenario.inputFiles?.length ? [
|
|
7204
7168
|
"### Input files",
|
|
7205
7169
|
"",
|
|
@@ -7212,9 +7176,30 @@ function buildRunEvalUserPrompt(input, ctx) {
|
|
|
7212
7176
|
"",
|
|
7213
7177
|
`This task carries correlationId \`${ctx.correlationId}\`. It joins`,
|
|
7214
7178
|
"this variant to its sibling `run_eval` tasks (other variants of the",
|
|
7215
|
-
"same scenario
|
|
7216
|
-
"
|
|
7217
|
-
"
|
|
7179
|
+
"same scenario and to any later `judge_eval_attempt` tasks created",
|
|
7180
|
+
"against those variants. You do not need to act on it directly — it",
|
|
7181
|
+
"is recorded for cross-variant aggregation at query time.",
|
|
7182
|
+
""
|
|
7183
|
+
].join("\n") : "";
|
|
7184
|
+
const executionSection = [
|
|
7185
|
+
"### Execution mode",
|
|
7186
|
+
"",
|
|
7187
|
+
`Mode: \`${execution.mode}\``,
|
|
7188
|
+
`Workspace: \`${execution.workspace}\``,
|
|
7189
|
+
execution.workspace === "none" ? "You are running in a scratch workspace with no repository checkout mounted. Do not assume git history or repo files are present unless the scenario provided them explicitly." : execution.workspace === "shared_mount" ? "You are running against the daemon shared mount. Treat any repository mutations as affecting the mounted checkout directly." : "You are running in a dedicated disposable git worktree isolated from the daemon shared checkout.",
|
|
7190
|
+
""
|
|
7191
|
+
].join("\n");
|
|
7192
|
+
const contextDisciplineSection = hasContext ? [
|
|
7193
|
+
"### Injected context discipline",
|
|
7194
|
+
"",
|
|
7195
|
+
"This task includes extra injected context from the task creator.",
|
|
7196
|
+
"You MUST inspect and use that context BEFORE you write solution",
|
|
7197
|
+
"files or draft your final answer.",
|
|
7198
|
+
"Do not solve first and only review the context afterward.",
|
|
7199
|
+
hasInlineContext ? "For `context_inline`, your FIRST content-inspection step should be a `read` of `/workspace/context-pack.md` before your first `write` call. The same content is also mirrored in `/workspace/AGENTS.md` and may be referenced from `/workspace/.claude/CLAUDE.md`." : "If injected context was provided as a skill, inspect that task-injected context before solving.",
|
|
7200
|
+
hasInlineContext ? "If `/workspace/context-pack.md` exists and you skip reading it before writing solution files, you are not following the task instructions." : "Do not rely on memory alone when task-injected context is available; inspect it first.",
|
|
7201
|
+
"If the injected context contains repo- or workflow-specific rules,",
|
|
7202
|
+
"those rules override your generic instincts.",
|
|
7218
7203
|
""
|
|
7219
7204
|
].join("\n") : "";
|
|
7220
7205
|
const finalOutputBlock = buildFinalOutputBlock({
|
|
@@ -7227,7 +7212,13 @@ function buildRunEvalUserPrompt(input, ctx) {
|
|
|
7227
7212
|
" \"totalTokens\": <int>,",
|
|
7228
7213
|
" \"durationMs\": <int>,",
|
|
7229
7214
|
" \"traceparent\": \"<from claim>\",",
|
|
7230
|
-
" \"verification\":
|
|
7215
|
+
" \"verification\": {",
|
|
7216
|
+
" \"inputCid\": \"<task inputCid>\",",
|
|
7217
|
+
" \"results\": [",
|
|
7218
|
+
" { \"id\": \"<criterion id>\", \"kind\": \"rubric\", \"status\": \"pass|fail|skip\", \"detail\": \"<optional one-liner>\" }",
|
|
7219
|
+
" ],",
|
|
7220
|
+
" \"passed\": <boolean>",
|
|
7221
|
+
" } // required iff input.successCriteria; must be an object, never a string",
|
|
7231
7222
|
"}"
|
|
7232
7223
|
].join("\n")
|
|
7233
7224
|
});
|
|
@@ -7235,6 +7226,8 @@ function buildRunEvalUserPrompt(input, ctx) {
|
|
|
7235
7226
|
"# Run Eval Agent\n",
|
|
7236
7227
|
`You are running an evaluation scenario as variant \`${variantLabel}\`.\nTask id: \`${ctx.taskId}\`\n`,
|
|
7237
7228
|
correlationSection,
|
|
7229
|
+
executionSection,
|
|
7230
|
+
contextDisciplineSection,
|
|
7238
7231
|
`### Scenario\n\n${scenario.prompt}\n`,
|
|
7239
7232
|
inputFilesSection,
|
|
7240
7233
|
verificationSection,
|
|
@@ -7306,6 +7299,16 @@ function buildTaskUserPrompt(task, ctx) {
|
|
|
7306
7299
|
diaryId: ctx.diaryId,
|
|
7307
7300
|
taskId: ctx.taskId
|
|
7308
7301
|
});
|
|
7302
|
+
case JUDGE_EVAL_ATTEMPT_TYPE:
|
|
7303
|
+
if (!Check(JudgeEvalAttemptInput, task.input)) {
|
|
7304
|
+
const errors = [...Errors(JudgeEvalAttemptInput, task.input)];
|
|
7305
|
+
throw new Error(`judge_eval_attempt input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
|
|
7306
|
+
}
|
|
7307
|
+
return buildJudgeEvalAttemptUserPrompt(task.input, {
|
|
7308
|
+
diaryId: ctx.diaryId,
|
|
7309
|
+
taskId: ctx.taskId,
|
|
7310
|
+
workspace: ctx.workspace
|
|
7311
|
+
});
|
|
7309
7312
|
case PR_REVIEW_TYPE:
|
|
7310
7313
|
if (!Check(PrReviewInput, task.input)) {
|
|
7311
7314
|
const errors = [...Errors(PrReviewInput, task.input)];
|
|
@@ -7316,15 +7319,6 @@ function buildTaskUserPrompt(task, ctx) {
|
|
|
7316
7319
|
taskId: ctx.taskId,
|
|
7317
7320
|
workspace: ctx.workspace
|
|
7318
7321
|
});
|
|
7319
|
-
case JUDGE_EVAL_VARIANT_TYPE:
|
|
7320
|
-
if (!Check(JudgeEvalVariantInput, task.input)) {
|
|
7321
|
-
const errors = [...Errors(JudgeEvalVariantInput, task.input)];
|
|
7322
|
-
throw new Error(`judge_eval_variant input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
|
|
7323
|
-
}
|
|
7324
|
-
return buildJudgeEvalVariantUserPrompt(task.input, {
|
|
7325
|
-
diaryId: ctx.diaryId,
|
|
7326
|
-
taskId: ctx.taskId
|
|
7327
|
-
});
|
|
7328
7322
|
case RUN_EVAL_TYPE:
|
|
7329
7323
|
if (!Check(RunEvalInput, task.input)) {
|
|
7330
7324
|
const errors = [...Errors(RunEvalInput, task.input)];
|
|
@@ -10077,12 +10071,20 @@ var MoltNetError = class extends Error {
|
|
|
10077
10071
|
code;
|
|
10078
10072
|
statusCode;
|
|
10079
10073
|
detail;
|
|
10074
|
+
/**
|
|
10075
|
+
* Populated when the server returned a `VALIDATION_FAILED` problem
|
|
10076
|
+
* (status 400) with field-level errors. Empty / undefined for every
|
|
10077
|
+
* other problem kind. Imposer scripts surface these to operators so
|
|
10078
|
+
* they don't have to re-run with curl to see what was rejected.
|
|
10079
|
+
*/
|
|
10080
|
+
validationErrors;
|
|
10080
10081
|
constructor(message, options) {
|
|
10081
10082
|
super(message);
|
|
10082
10083
|
this.name = "MoltNetError";
|
|
10083
10084
|
this.code = options.code;
|
|
10084
10085
|
this.statusCode = options.statusCode;
|
|
10085
10086
|
this.detail = options.detail;
|
|
10087
|
+
this.validationErrors = options.validationErrors;
|
|
10086
10088
|
}
|
|
10087
10089
|
};
|
|
10088
10090
|
var NetworkError = class extends MoltNetError {
|
|
@@ -10106,10 +10108,14 @@ var AuthenticationError = class extends MoltNetError {
|
|
|
10106
10108
|
};
|
|
10107
10109
|
function problemToError(problem, statusCode) {
|
|
10108
10110
|
const title = problem.title ?? "Request failed";
|
|
10109
|
-
|
|
10111
|
+
const message = problem.detail ? `${title}: ${problem.detail}` : title;
|
|
10112
|
+
const rawErrors = problem.errors;
|
|
10113
|
+
const validationErrors = Array.isArray(rawErrors) ? rawErrors.filter((e) => typeof e === "object" && e !== null && typeof e.field === "string" && typeof e.message === "string") : void 0;
|
|
10114
|
+
return new MoltNetError(message, {
|
|
10110
10115
|
code: problem.type ?? problem.code ?? "UNKNOWN",
|
|
10111
10116
|
statusCode,
|
|
10112
|
-
detail: problem.detail
|
|
10117
|
+
detail: problem.detail,
|
|
10118
|
+
validationErrors
|
|
10113
10119
|
});
|
|
10114
10120
|
}
|
|
10115
10121
|
//#endregion
|
|
@@ -14034,31 +14040,6 @@ function abortableSleep(ms, signal) {
|
|
|
14034
14040
|
});
|
|
14035
14041
|
}
|
|
14036
14042
|
//#endregion
|
|
14037
|
-
//#region ../../libs/agent-runtime/src/subagent-output-contracts.ts
|
|
14038
|
-
/**
|
|
14039
|
-
* Construct an immutable contract registry from a static list.
|
|
14040
|
-
*
|
|
14041
|
-
* The resulting registry is safe to share across sessions and
|
|
14042
|
-
* invocations — no mutation is possible after construction.
|
|
14043
|
-
*/
|
|
14044
|
-
function createSubagentContractRegistry(contracts) {
|
|
14045
|
-
const lookup = /* @__PURE__ */ new Map();
|
|
14046
|
-
for (const c of contracts) {
|
|
14047
|
-
if (!c.name || c.name.trim().length === 0) throw new Error("subagent output contract name is required");
|
|
14048
|
-
if (!/^[a-z][a-z0-9_]*$/.test(c.name)) throw new Error(`subagent output contract name '${c.name}' must be lower_snake_case (starts with a letter, then [a-z0-9_]+)`);
|
|
14049
|
-
if (lookup.has(c.name)) throw new Error(`duplicate subagent output contract name '${c.name}' in constructor args`);
|
|
14050
|
-
lookup.set(c.name, c);
|
|
14051
|
-
}
|
|
14052
|
-
return {
|
|
14053
|
-
get(name) {
|
|
14054
|
-
return lookup.get(name) ?? null;
|
|
14055
|
-
},
|
|
14056
|
-
list() {
|
|
14057
|
-
return [...lookup.values()];
|
|
14058
|
-
}
|
|
14059
|
-
};
|
|
14060
|
-
}
|
|
14061
|
-
//#endregion
|
|
14062
14043
|
//#region ../../libs/pi-extension/src/moltnet/render-phase6.ts
|
|
14063
14044
|
function slugToTitle(value) {
|
|
14064
14045
|
return value.split(/[:/_-]+/).filter(Boolean).map((part) => part[0]?.toUpperCase() + part.slice(1)).join(" ");
|
|
@@ -14669,6 +14650,41 @@ function createMoltNetTools(config) {
|
|
|
14669
14650
|
};
|
|
14670
14651
|
}
|
|
14671
14652
|
});
|
|
14653
|
+
const listTaskMessages = defineTool({
|
|
14654
|
+
name: "moltnet_list_task_messages",
|
|
14655
|
+
label: "List MoltNet Task Attempt Messages",
|
|
14656
|
+
description: "List messages for a specific task attempt. Use this when you need the turn-by-turn execution record behind an accepted attempt — tool calls, text deltas, and error/info events that do not appear in the attempt output alone.",
|
|
14657
|
+
parameters: Type.Object({
|
|
14658
|
+
taskId: Type.String({ description: "Task ID (UUID)." }),
|
|
14659
|
+
attemptN: Type.Integer({
|
|
14660
|
+
minimum: 1,
|
|
14661
|
+
description: "Attempt number to inspect."
|
|
14662
|
+
}),
|
|
14663
|
+
afterSeq: Type.Optional(Type.Integer({
|
|
14664
|
+
minimum: 0,
|
|
14665
|
+
description: "Optional cursor: only return messages with seq > afterSeq."
|
|
14666
|
+
})),
|
|
14667
|
+
limit: Type.Optional(Type.Integer({
|
|
14668
|
+
minimum: 1,
|
|
14669
|
+
maximum: 500,
|
|
14670
|
+
description: "Optional maximum messages to return. Defaults to the API value."
|
|
14671
|
+
}))
|
|
14672
|
+
}),
|
|
14673
|
+
async execute(_id, params) {
|
|
14674
|
+
const { agent } = ensureConnected(config);
|
|
14675
|
+
const messages = await agent.tasks.listMessages(params.taskId, params.attemptN, {
|
|
14676
|
+
afterSeq: params.afterSeq,
|
|
14677
|
+
limit: params.limit
|
|
14678
|
+
});
|
|
14679
|
+
return {
|
|
14680
|
+
content: [{
|
|
14681
|
+
type: "text",
|
|
14682
|
+
text: JSON.stringify(messages, null, 2)
|
|
14683
|
+
}],
|
|
14684
|
+
details: {}
|
|
14685
|
+
};
|
|
14686
|
+
}
|
|
14687
|
+
});
|
|
14672
14688
|
const reviewSessionErrors = defineTool({
|
|
14673
14689
|
name: "moltnet_review_session_errors",
|
|
14674
14690
|
label: "Review Session Tool Errors",
|
|
@@ -14717,6 +14733,7 @@ function createMoltNetTools(config) {
|
|
|
14717
14733
|
createEntry,
|
|
14718
14734
|
getTask,
|
|
14719
14735
|
listTaskAttempts,
|
|
14736
|
+
listTaskMessages,
|
|
14720
14737
|
reviewSessionErrors,
|
|
14721
14738
|
defineTool({
|
|
14722
14739
|
name: "moltnet_host_exec",
|
|
@@ -15015,6 +15032,12 @@ var GUEST_WORKSPACE$1 = "/workspace";
|
|
|
15015
15032
|
* investigation and the alternatives we rejected.
|
|
15016
15033
|
*/
|
|
15017
15034
|
var GUEST_TASK_SKILLS_MOUNT = "/moltnet-task-skills";
|
|
15035
|
+
function shouldRunResumeCommand(entry, ctx) {
|
|
15036
|
+
if (typeof entry === "string") return true;
|
|
15037
|
+
const workspaceModes = entry.when?.workspaceMode;
|
|
15038
|
+
if (workspaceModes && !workspaceModes.includes(ctx.workspaceMode)) return false;
|
|
15039
|
+
return true;
|
|
15040
|
+
}
|
|
15018
15041
|
/**
|
|
15019
15042
|
* Resolve the main worktree root (where .moltnet/ lives — it's untracked,
|
|
15020
15043
|
* only exists in the main worktree, not in git worktrees).
|
|
@@ -15160,6 +15183,7 @@ async function resumeVm(config) {
|
|
|
15160
15183
|
...envOverrides
|
|
15161
15184
|
};
|
|
15162
15185
|
const resources = config.sandboxConfig?.resources;
|
|
15186
|
+
const workspaceMode = config.workspaceMode ?? "shared_mount";
|
|
15163
15187
|
const vm = await VmCheckpoint.load(config.checkpointPath).resume({
|
|
15164
15188
|
httpHooks,
|
|
15165
15189
|
env: vmEnv,
|
|
@@ -15178,7 +15202,32 @@ async function resumeVm(config) {
|
|
|
15178
15202
|
'`);
|
|
15179
15203
|
await vmRun(vm, "DNS resolvers", `printf 'nameserver 8.8.8.8\\nnameserver 1.1.1.1\\n' > /etc/resolv.conf`);
|
|
15180
15204
|
await vmRun(vm, "git safe.directory", `git config --system --add safe.directory '*'`);
|
|
15181
|
-
for (const [i,
|
|
15205
|
+
for (const [i, entry] of (config.sandboxConfig?.resumeCommands ?? []).entries()) {
|
|
15206
|
+
if (!shouldRunResumeCommand(entry, { workspaceMode })) continue;
|
|
15207
|
+
const { run, retries, backoffMs } = typeof entry === "string" ? {
|
|
15208
|
+
run: entry,
|
|
15209
|
+
retries: 0,
|
|
15210
|
+
backoffMs: 2e3
|
|
15211
|
+
} : {
|
|
15212
|
+
run: entry.run,
|
|
15213
|
+
retries: entry.retries ?? 0,
|
|
15214
|
+
backoffMs: entry.retryBackoffMs ?? 2e3
|
|
15215
|
+
};
|
|
15216
|
+
const label = `resumeCommands[${i}]`;
|
|
15217
|
+
let lastErr;
|
|
15218
|
+
for (let attempt = 0; attempt <= retries; attempt++) try {
|
|
15219
|
+
await vmRun(vm, label, run);
|
|
15220
|
+
lastErr = void 0;
|
|
15221
|
+
break;
|
|
15222
|
+
} catch (err) {
|
|
15223
|
+
lastErr = err;
|
|
15224
|
+
if (attempt === retries) break;
|
|
15225
|
+
await new Promise((resolve) => {
|
|
15226
|
+
setTimeout(resolve, (attempt + 1) * backoffMs);
|
|
15227
|
+
});
|
|
15228
|
+
}
|
|
15229
|
+
if (lastErr) throw lastErr instanceof Error ? lastErr : new Error(String(lastErr));
|
|
15230
|
+
}
|
|
15182
15231
|
const vmSshDir = `${vmAgentDir}/ssh`;
|
|
15183
15232
|
await vm.exec(`mkdir -p ${vmAgentDir}/ssh /home/agent/.pi/agent`);
|
|
15184
15233
|
if (creds.piAuthJson !== null) await vm.fs.writeFile("/home/agent/.pi/agent/auth.json", creds.piAuthJson, { mode: 384 });
|
|
@@ -15557,7 +15606,8 @@ async function buildAgentSession(args) {
|
|
|
15557
15606
|
await resourceLoader.reload();
|
|
15558
15607
|
const sessionManager = args.sessionPersistence ? await resolvePersistentSessionManager({
|
|
15559
15608
|
cwd: args.cwdPath,
|
|
15560
|
-
sessionDir: args.sessionPersistence.sessionDir
|
|
15609
|
+
sessionDir: args.sessionPersistence.sessionDir,
|
|
15610
|
+
forkFromSessionPath: args.sessionPersistence.forkFromSessionPath
|
|
15561
15611
|
}) : SessionManager.inMemory(args.cwdPath);
|
|
15562
15612
|
return (await createAgentSession({
|
|
15563
15613
|
agentDir: args.piAuthDir,
|
|
@@ -15569,6 +15619,7 @@ async function buildAgentSession(args) {
|
|
|
15569
15619
|
})).session;
|
|
15570
15620
|
}
|
|
15571
15621
|
async function resolvePersistentSessionManager(args) {
|
|
15622
|
+
if (args.forkFromSessionPath) return SessionManager.forkFrom(args.forkFromSessionPath, args.cwd, args.sessionDir);
|
|
15572
15623
|
await SessionManager.list(args.cwd, args.sessionDir);
|
|
15573
15624
|
return SessionManager.continueRecent(args.cwd, args.sessionDir);
|
|
15574
15625
|
}
|
|
@@ -15609,6 +15660,11 @@ async function resolvePersistentSessionManager(args) {
|
|
|
15609
15660
|
* paths under this mount via `toGuestPath` in `tool-operations.ts`.
|
|
15610
15661
|
*/
|
|
15611
15662
|
var SKILL_ROOT_IN_VM = GUEST_TASK_SKILLS_MOUNT;
|
|
15663
|
+
var INLINE_CONTEXT_ROOT_IN_VM = "/workspace/.moltnet/context";
|
|
15664
|
+
var WORKSPACE_CONTEXT_PACK = "/workspace/context-pack.md";
|
|
15665
|
+
var WORKSPACE_AGENTS_MD = "/workspace/AGENTS.md";
|
|
15666
|
+
var WORKSPACE_CLAUDE_DIR = "/workspace/.claude";
|
|
15667
|
+
var WORKSPACE_CLAUDE_MD = "/workspace/.claude/CLAUDE.md";
|
|
15612
15668
|
/** Bounds borrowed from pi's skill validation; conservative caps so a
|
|
15613
15669
|
* malformed SKILL.md doesn't bloat the system prompt. */
|
|
15614
15670
|
var MAX_SKILL_NAME = 64;
|
|
@@ -15619,21 +15675,40 @@ var MAX_SKILL_DESCRIPTION = 1024;
|
|
|
15619
15675
|
*/
|
|
15620
15676
|
async function injectTaskContext(args) {
|
|
15621
15677
|
const skills = [];
|
|
15678
|
+
const inlineContexts = [];
|
|
15622
15679
|
const resolved = await resolveTaskContext({
|
|
15623
15680
|
context: args.context,
|
|
15624
|
-
deliver: {
|
|
15625
|
-
|
|
15626
|
-
|
|
15627
|
-
|
|
15628
|
-
|
|
15629
|
-
|
|
15630
|
-
|
|
15631
|
-
|
|
15632
|
-
|
|
15633
|
-
|
|
15634
|
-
|
|
15635
|
-
|
|
15681
|
+
deliver: {
|
|
15682
|
+
skill: async ({ slug, content }) => {
|
|
15683
|
+
const dir = `${SKILL_ROOT_IN_VM}/${slug}`;
|
|
15684
|
+
const filePath = `${dir}/SKILL.md`;
|
|
15685
|
+
await args.fs.mkdir(dir, { recursive: true });
|
|
15686
|
+
await args.fs.writeFile(filePath, content, { mode: 420 });
|
|
15687
|
+
skills.push(buildSyntheticSkill({
|
|
15688
|
+
slug,
|
|
15689
|
+
content,
|
|
15690
|
+
filePath,
|
|
15691
|
+
dir
|
|
15692
|
+
}));
|
|
15693
|
+
},
|
|
15694
|
+
contextFile: async ({ suggestedFileName, content }) => {
|
|
15695
|
+
await args.fs.mkdir(INLINE_CONTEXT_ROOT_IN_VM, { recursive: true });
|
|
15696
|
+
const filePath = `${INLINE_CONTEXT_ROOT_IN_VM}/${suggestedFileName}`;
|
|
15697
|
+
await args.fs.writeFile(filePath, content, { mode: 420 });
|
|
15698
|
+
inlineContexts.push({
|
|
15699
|
+
slug: suggestedFileName.replace(/\.md$/u, ""),
|
|
15700
|
+
content
|
|
15701
|
+
});
|
|
15702
|
+
}
|
|
15703
|
+
}
|
|
15636
15704
|
});
|
|
15705
|
+
if (inlineContexts.length > 0) {
|
|
15706
|
+
const packContent = buildWorkspaceContextPack(inlineContexts);
|
|
15707
|
+
await args.fs.writeFile(WORKSPACE_CONTEXT_PACK, packContent, { mode: 420 });
|
|
15708
|
+
await args.fs.writeFile(WORKSPACE_AGENTS_MD, packContent, { mode: 420 });
|
|
15709
|
+
await args.fs.mkdir(WORKSPACE_CLAUDE_DIR, { recursive: true });
|
|
15710
|
+
await args.fs.writeFile(WORKSPACE_CLAUDE_MD, "@../context-pack.md\n", { mode: 420 });
|
|
15711
|
+
}
|
|
15637
15712
|
return {
|
|
15638
15713
|
injected: resolved.injected,
|
|
15639
15714
|
skills,
|
|
@@ -15641,6 +15716,17 @@ async function injectTaskContext(args) {
|
|
|
15641
15716
|
userInlineSuffix: resolved.userInlineSuffix
|
|
15642
15717
|
};
|
|
15643
15718
|
}
|
|
15719
|
+
function buildWorkspaceContextPack(contexts) {
|
|
15720
|
+
return [
|
|
15721
|
+
"# Context Pack",
|
|
15722
|
+
"",
|
|
15723
|
+
...contexts.map(({ slug, content }) => [
|
|
15724
|
+
`## ${slug}`,
|
|
15725
|
+
"",
|
|
15726
|
+
content.trimEnd()
|
|
15727
|
+
].join("\n"))
|
|
15728
|
+
].join("\n\n").trimEnd() + "\n";
|
|
15729
|
+
}
|
|
15644
15730
|
/**
|
|
15645
15731
|
* Build a `Skill` object pi will faithfully render in
|
|
15646
15732
|
* `<available_skills>`. We extract `name` and `description` from the
|
|
@@ -16004,7 +16090,7 @@ async function parseStructuredTaskOutput(assistantText, taskType, opts = {}) {
|
|
|
16004
16090
|
}
|
|
16005
16091
|
};
|
|
16006
16092
|
}
|
|
16007
|
-
const errors = validateTaskOutput(taskType, extracted);
|
|
16093
|
+
const errors = validateTaskOutput(taskType, extracted, opts.input);
|
|
16008
16094
|
if (errors.length > 0) {
|
|
16009
16095
|
const details = errors.slice(0, 3).map((error) => `${error.field}: ${error.message}`);
|
|
16010
16096
|
const [firstError] = errors;
|
|
@@ -16118,7 +16204,7 @@ function createSubmitOutputTool(taskType, opts = {}) {
|
|
|
16118
16204
|
description: contract.description,
|
|
16119
16205
|
parameters: schema,
|
|
16120
16206
|
async execute(_id, params) {
|
|
16121
|
-
const errors = validateTaskOutput(taskType, params);
|
|
16207
|
+
const errors = validateTaskOutput(taskType, params, opts.input);
|
|
16122
16208
|
if (errors.length > 0) {
|
|
16123
16209
|
const detailMsg = errors.slice(0, 3).map((err) => `${err.field}: ${err.message}`).join("; ");
|
|
16124
16210
|
const details = {
|
|
@@ -16187,6 +16273,39 @@ function resolveSubmitTools(taskType, opts = {}) {
|
|
|
16187
16273
|
//#region ../../libs/pi-extension/src/runtime/task-workspace.ts
|
|
16188
16274
|
function prepareTaskWorkspace(task, requestedMountPath, executionPlan) {
|
|
16189
16275
|
const branch = executionPlan?.worktreeBranch ?? null;
|
|
16276
|
+
const workspaceMode = executionPlan?.workspaceMode ?? "shared_mount";
|
|
16277
|
+
const attachedWorkspace = executionPlan?.workspaceAttachment ?? null;
|
|
16278
|
+
if (attachedWorkspace) return {
|
|
16279
|
+
mountPath: attachedWorkspace.mountPath,
|
|
16280
|
+
cwdPath: attachedWorkspace.cwdPath,
|
|
16281
|
+
mode: workspaceMode,
|
|
16282
|
+
branch,
|
|
16283
|
+
cleanup: () => {}
|
|
16284
|
+
};
|
|
16285
|
+
if (workspaceMode === "scratch_mount") {
|
|
16286
|
+
const scratchDir = resolveTaskScratchPath(findMainWorktree(), executionPlan?.workspaceId ?? `task-${task.id}`);
|
|
16287
|
+
const keepWorkspace = executionPlan?.workspaceScope === "session" && executionPlan.sessionKey !== null;
|
|
16288
|
+
if (keepWorkspace) mkdirSync(scratchDir, { recursive: true });
|
|
16289
|
+
else {
|
|
16290
|
+
rmSync(scratchDir, {
|
|
16291
|
+
recursive: true,
|
|
16292
|
+
force: true
|
|
16293
|
+
});
|
|
16294
|
+
mkdirSync(scratchDir, { recursive: true });
|
|
16295
|
+
}
|
|
16296
|
+
return {
|
|
16297
|
+
mountPath: scratchDir,
|
|
16298
|
+
cwdPath: scratchDir,
|
|
16299
|
+
mode: "scratch_mount",
|
|
16300
|
+
branch: null,
|
|
16301
|
+
cleanup: keepWorkspace ? () => {} : () => {
|
|
16302
|
+
rmSync(scratchDir, {
|
|
16303
|
+
recursive: true,
|
|
16304
|
+
force: true
|
|
16305
|
+
});
|
|
16306
|
+
}
|
|
16307
|
+
};
|
|
16308
|
+
}
|
|
16190
16309
|
if (!branch) return {
|
|
16191
16310
|
mountPath: requestedMountPath,
|
|
16192
16311
|
cwdPath: requestedMountPath,
|
|
@@ -16224,6 +16343,9 @@ function prepareTaskWorkspace(task, requestedMountPath, executionPlan) {
|
|
|
16224
16343
|
function resolveTaskWorktreePath(mainRepo, workspaceId) {
|
|
16225
16344
|
return join(mainRepo, ".worktrees", workspaceId);
|
|
16226
16345
|
}
|
|
16346
|
+
function resolveTaskScratchPath(mainRepo, workspaceId) {
|
|
16347
|
+
return join(mainRepo, ".moltnet", "d", "task-workspaces", workspaceId);
|
|
16348
|
+
}
|
|
16227
16349
|
function ensureReusableTaskWorktree(mainRepo, worktreeDir, branch) {
|
|
16228
16350
|
if (isRegisteredWorktree$1(mainRepo, worktreeDir)) return;
|
|
16229
16351
|
if (existsSync(worktreeDir)) throw new Error(`Expected reusable worktree ${worktreeDir} to be git-managed, but it exists outside git worktree metadata.`);
|
|
@@ -16460,12 +16582,14 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
16460
16582
|
return makeFailedOutput("worktree_setup_failed", message);
|
|
16461
16583
|
}
|
|
16462
16584
|
try {
|
|
16585
|
+
const sandboxConfig = applyExecutionPlanSandboxOverrides(opts.sandboxConfig, executionPlan);
|
|
16463
16586
|
managed = await resumeVm({
|
|
16464
16587
|
checkpointPath,
|
|
16465
16588
|
agentName: opts.agentName,
|
|
16466
16589
|
mountPath,
|
|
16590
|
+
workspaceMode: workspace.mode,
|
|
16467
16591
|
extraAllowedHosts: opts.extraAllowedHosts,
|
|
16468
|
-
sandboxConfig
|
|
16592
|
+
sandboxConfig
|
|
16469
16593
|
});
|
|
16470
16594
|
} catch (err) {
|
|
16471
16595
|
const message = err instanceof Error ? err.message : String(err);
|
|
@@ -16494,7 +16618,8 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
16494
16618
|
taskId: task.id,
|
|
16495
16619
|
workspace: {
|
|
16496
16620
|
mode: activeWorkspace.mode,
|
|
16497
|
-
branch: activeWorkspace.branch
|
|
16621
|
+
branch: activeWorkspace.branch,
|
|
16622
|
+
attached: executionPlan?.workspaceAttachment !== void 0
|
|
16498
16623
|
},
|
|
16499
16624
|
extras: opts.promptExtras
|
|
16500
16625
|
});
|
|
@@ -16536,7 +16661,10 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
16536
16661
|
createEditToolDefinition(mountPath, { operations: createGondolinEditOps(managed.vm, mountPath) }),
|
|
16537
16662
|
createBashToolDefinition(mountPath, { operations: createGondolinBashOps(managed.vm, mountPath) })
|
|
16538
16663
|
];
|
|
16539
|
-
const { handle: submitToolHandle, tools: submitToolDefs } = resolveSubmitTools(task.taskType, {
|
|
16664
|
+
const { handle: submitToolHandle, tools: submitToolDefs } = resolveSubmitTools(task.taskType, {
|
|
16665
|
+
model: opts.model,
|
|
16666
|
+
input: task.input
|
|
16667
|
+
});
|
|
16540
16668
|
const submitTools = submitToolDefs;
|
|
16541
16669
|
try {
|
|
16542
16670
|
const moltnetAgent = await connect({ configDir: managed.agentDir });
|
|
@@ -16755,8 +16883,20 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
16755
16883
|
phase: "output_validation"
|
|
16756
16884
|
});
|
|
16757
16885
|
}
|
|
16758
|
-
else {
|
|
16759
|
-
|
|
16886
|
+
else if (submitToolHandle) {
|
|
16887
|
+
parseError = {
|
|
16888
|
+
code: "output_missing",
|
|
16889
|
+
message: "Agent did not submit output through the task submit tool. A valid submit tool call is required to complete this task type."
|
|
16890
|
+
};
|
|
16891
|
+
await emit("error", {
|
|
16892
|
+
message: parseError.message,
|
|
16893
|
+
phase: "output_validation"
|
|
16894
|
+
});
|
|
16895
|
+
} else {
|
|
16896
|
+
const parsed = await parseStructuredTaskOutput(assistantText, task.taskType, {
|
|
16897
|
+
model: opts.model,
|
|
16898
|
+
input: task.input
|
|
16899
|
+
});
|
|
16760
16900
|
parsedOutput = parsed.output;
|
|
16761
16901
|
parsedOutputCid = parsed.outputCid;
|
|
16762
16902
|
parseError = parsed.error;
|
|
@@ -16842,6 +16982,18 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
16842
16982
|
}
|
|
16843
16983
|
}
|
|
16844
16984
|
}
|
|
16985
|
+
function applyExecutionPlanSandboxOverrides(sandboxConfig, executionPlan) {
|
|
16986
|
+
const shadowWrites = executionPlan?.workspaceAttachment?.shadowWrites;
|
|
16987
|
+
if (!shadowWrites) return sandboxConfig;
|
|
16988
|
+
return {
|
|
16989
|
+
...sandboxConfig,
|
|
16990
|
+
vfs: {
|
|
16991
|
+
...sandboxConfig?.vfs,
|
|
16992
|
+
shadow: ["**"],
|
|
16993
|
+
shadowMode: shadowWrites
|
|
16994
|
+
}
|
|
16995
|
+
};
|
|
16996
|
+
}
|
|
16845
16997
|
function emptyUsage(provider, model) {
|
|
16846
16998
|
return {
|
|
16847
16999
|
inputTokens: 0,
|
|
@@ -17066,11 +17218,12 @@ var DaemonSlotRegistryError = class extends Error {
|
|
|
17066
17218
|
this.name = "DaemonSlotRegistryError";
|
|
17067
17219
|
}
|
|
17068
17220
|
};
|
|
17221
|
+
var SqliteDatabaseSync = DatabaseSync;
|
|
17069
17222
|
var DaemonSlotRegistry = class {
|
|
17070
17223
|
db;
|
|
17071
17224
|
constructor(dbPath) {
|
|
17072
17225
|
try {
|
|
17073
|
-
this.db = new
|
|
17226
|
+
this.db = new SqliteDatabaseSync(dbPath);
|
|
17074
17227
|
this.withDb("initialize schema", () => {
|
|
17075
17228
|
this.db.exec(`
|
|
17076
17229
|
PRAGMA journal_mode = WAL;
|
|
@@ -17119,6 +17272,9 @@ var DaemonSlotRegistry = class {
|
|
|
17119
17272
|
|
|
17120
17273
|
CREATE INDEX IF NOT EXISTS daemon_slots_expires_idx
|
|
17121
17274
|
ON daemon_slots (expires_at_ms);
|
|
17275
|
+
|
|
17276
|
+
CREATE INDEX IF NOT EXISTS daemon_slots_task_attempt_idx
|
|
17277
|
+
ON daemon_slots (last_task_id, last_attempt_n, last_used_at_ms DESC);
|
|
17122
17278
|
`);
|
|
17123
17279
|
});
|
|
17124
17280
|
} catch (error) {
|
|
@@ -17208,6 +17364,30 @@ var DaemonSlotRegistry = class {
|
|
|
17208
17364
|
SET session_path = ?
|
|
17209
17365
|
WHERE agent_name = ? AND provider = ? AND model = ? AND slot_key = ?`).run(sessionPath, identity.agentName, identity.provider, identity.model, slotKey));
|
|
17210
17366
|
}
|
|
17367
|
+
findLatestProducerSlotByTaskAttempt(taskId, attemptN) {
|
|
17368
|
+
const slot = this.withDb("find producer slot by task attempt", () => this.db.prepare(`SELECT
|
|
17369
|
+
agent_name as agentName,
|
|
17370
|
+
provider,
|
|
17371
|
+
model,
|
|
17372
|
+
slot_key as slotKey,
|
|
17373
|
+
task_type as taskType,
|
|
17374
|
+
state,
|
|
17375
|
+
last_task_id as lastTaskId,
|
|
17376
|
+
last_attempt_n as lastAttemptN,
|
|
17377
|
+
created_at_ms as createdAtMs,
|
|
17378
|
+
last_used_at_ms as lastUsedAtMs,
|
|
17379
|
+
expires_at_ms as expiresAtMs
|
|
17380
|
+
FROM daemon_slots
|
|
17381
|
+
WHERE last_task_id = ? AND last_attempt_n = ?
|
|
17382
|
+
ORDER BY last_used_at_ms DESC
|
|
17383
|
+
LIMIT 1`).get(taskId, attemptN) ?? null);
|
|
17384
|
+
if (!slot) return null;
|
|
17385
|
+
return {
|
|
17386
|
+
slot,
|
|
17387
|
+
session: this.lookupSession(slot),
|
|
17388
|
+
workspace: this.lookupWorkspace(slot)
|
|
17389
|
+
};
|
|
17390
|
+
}
|
|
17211
17391
|
reapExpiredSlots(now = Date.now()) {
|
|
17212
17392
|
this.withDb("begin reap transaction", () => this.db.exec("BEGIN IMMEDIATE"));
|
|
17213
17393
|
try {
|
|
@@ -17251,8 +17431,8 @@ var DaemonSlotRegistry = class {
|
|
|
17251
17431
|
const deleteStmt = this.withDb("prepare expired slot delete", () => this.db.prepare("DELETE FROM daemon_slots WHERE agent_name = ? AND provider = ? AND model = ? AND slot_key = ?"));
|
|
17252
17432
|
const out = [];
|
|
17253
17433
|
for (const slot of slots) {
|
|
17254
|
-
const session = this.
|
|
17255
|
-
const workspace = this.
|
|
17434
|
+
const session = this.lookupSession(slot, selectSession);
|
|
17435
|
+
const workspace = this.lookupWorkspace(slot, selectWorkspace);
|
|
17256
17436
|
out.push({
|
|
17257
17437
|
slot,
|
|
17258
17438
|
session,
|
|
@@ -17280,6 +17460,29 @@ var DaemonSlotRegistry = class {
|
|
|
17280
17460
|
throw new DaemonSlotRegistryError(operation, error);
|
|
17281
17461
|
}
|
|
17282
17462
|
}
|
|
17463
|
+
lookupSession(slot, stmt = this.withDb("prepare slot session lookup", () => this.db.prepare(`SELECT
|
|
17464
|
+
agent_name as agentName,
|
|
17465
|
+
provider,
|
|
17466
|
+
model,
|
|
17467
|
+
slot_key as slotKey,
|
|
17468
|
+
session_dir as sessionDir,
|
|
17469
|
+
session_path as sessionPath
|
|
17470
|
+
FROM daemon_slot_sessions
|
|
17471
|
+
WHERE agent_name = ? AND provider = ? AND model = ? AND slot_key = ?`))) {
|
|
17472
|
+
return this.withDb("select slot session", () => stmt.get(slot.agentName, slot.provider, slot.model, slot.slotKey) ?? null);
|
|
17473
|
+
}
|
|
17474
|
+
lookupWorkspace(slot, stmt = this.withDb("prepare slot workspace lookup", () => this.db.prepare(`SELECT
|
|
17475
|
+
agent_name as agentName,
|
|
17476
|
+
provider,
|
|
17477
|
+
model,
|
|
17478
|
+
slot_key as slotKey,
|
|
17479
|
+
workspace_id as workspaceId,
|
|
17480
|
+
worktree_path as worktreePath,
|
|
17481
|
+
worktree_branch as worktreeBranch
|
|
17482
|
+
FROM daemon_slot_workspaces
|
|
17483
|
+
WHERE agent_name = ? AND provider = ? AND model = ? AND slot_key = ?`))) {
|
|
17484
|
+
return this.withDb("select slot workspace", () => stmt.get(slot.agentName, slot.provider, slot.model, slot.slotKey) ?? null);
|
|
17485
|
+
}
|
|
17283
17486
|
};
|
|
17284
17487
|
function resolveLatestPiSessionPath(sessionDir) {
|
|
17285
17488
|
try {
|
|
@@ -17380,11 +17583,6 @@ function buildCustomSessionKey(task) {
|
|
|
17380
17583
|
if (!task.correlationId || !variantLabel) return null;
|
|
17381
17584
|
return `run_eval:correlation:${task.correlationId}:variant:${slugifySessionComponent(variantLabel)}`;
|
|
17382
17585
|
}
|
|
17383
|
-
case "judge_eval_variant": {
|
|
17384
|
-
const runTaskIds = Array.isArray(task.input.runTaskIds) ? task.input.runTaskIds.filter((value) => typeof value === "string") : [];
|
|
17385
|
-
if (runTaskIds.length < 1) return null;
|
|
17386
|
-
return `judge_eval_variant:run_tasks:${[...runTaskIds].sort().join(",")}`;
|
|
17387
|
-
}
|
|
17388
17586
|
default: return null;
|
|
17389
17587
|
}
|
|
17390
17588
|
}
|
|
@@ -17395,18 +17593,20 @@ function slugifySessionComponent(input) {
|
|
|
17395
17593
|
//#region src/lib/task-execution-plan.ts
|
|
17396
17594
|
function buildDaemonTaskExecutionPlan(task, stateDirs, identity, warmSessionTtlSec) {
|
|
17397
17595
|
const descriptor = deriveTaskSessionDescriptor(task);
|
|
17596
|
+
const workspaceMode = resolveTaskWorkspaceMode(task, descriptor.policy);
|
|
17398
17597
|
const slotKey = warmSessionTtlSec > 0 ? descriptor.sessionKey : null;
|
|
17399
17598
|
const workspaceScope = slotKey !== null ? descriptor.policy.workspaceScope : "attempt";
|
|
17400
17599
|
const slotId = slotKey ? buildDaemonSlotId(identity, slotKey) : null;
|
|
17401
17600
|
const sessionDir = slotId ? `${stateDirs.piSessionsDir}/${encodeURIComponent(slotId)}` : null;
|
|
17402
|
-
const worktreeBranch = resolveTaskWorktreeBranch(task,
|
|
17403
|
-
const workspaceId =
|
|
17601
|
+
const worktreeBranch = resolveTaskWorktreeBranch(task, workspaceMode);
|
|
17602
|
+
const workspaceId = workspaceMode !== "shared_mount" ? resolveTaskWorkspaceId(task, {
|
|
17404
17603
|
sessionKey: slotId,
|
|
17405
17604
|
workspaceScope,
|
|
17406
17605
|
sessionPersistence: sessionDir ? { sessionDir } : null
|
|
17407
17606
|
}) : null;
|
|
17408
17607
|
return {
|
|
17409
17608
|
descriptor,
|
|
17609
|
+
workspaceMode,
|
|
17410
17610
|
sessionKey: slotId,
|
|
17411
17611
|
slotKey,
|
|
17412
17612
|
slotId,
|
|
@@ -17435,8 +17635,8 @@ function slugSlotIdentityComponent(input) {
|
|
|
17435
17635
|
"-"
|
|
17436
17636
|
]);
|
|
17437
17637
|
}
|
|
17438
|
-
function resolveTaskWorktreeBranch(task,
|
|
17439
|
-
if (
|
|
17638
|
+
function resolveTaskWorktreeBranch(task, workspaceMode) {
|
|
17639
|
+
if (workspaceMode !== "dedicated_worktree") return null;
|
|
17440
17640
|
if (task.taskType === "fulfill_brief") {
|
|
17441
17641
|
const input = task.input;
|
|
17442
17642
|
const slug = slugifyAsciiLower(typeof input.title === "string" && input.title.trim().length > 0 ? input.title : typeof input.brief === "string" && input.brief.trim().length > 0 ? input.brief : task.taskType, 60) || "task";
|
|
@@ -17445,12 +17645,27 @@ function resolveTaskWorktreeBranch(task, policy) {
|
|
|
17445
17645
|
}
|
|
17446
17646
|
return `task/${slugifyAsciiLower(task.taskType, 60) || "task"}-${task.id.slice(0, 8)}`;
|
|
17447
17647
|
}
|
|
17648
|
+
function resolveTaskWorkspaceMode(task, policy) {
|
|
17649
|
+
if (task.taskType !== "run_eval") return policy.workspaceMode;
|
|
17650
|
+
switch (typeof task.input.execution?.workspace === "string" ? task.input.execution.workspace : null) {
|
|
17651
|
+
case "none": return "scratch_mount";
|
|
17652
|
+
case "shared_mount": return "shared_mount";
|
|
17653
|
+
case "dedicated_worktree": return "dedicated_worktree";
|
|
17654
|
+
default: return policy.workspaceMode;
|
|
17655
|
+
}
|
|
17656
|
+
}
|
|
17448
17657
|
function resolveTaskWorkspaceId(task, executionPlan) {
|
|
17449
17658
|
if (executionPlan.workspaceScope === "session" && executionPlan.sessionKey !== null) return `session-${encodeURIComponent(executionPlan.sessionKey)}`;
|
|
17450
17659
|
return `task-${task.id}`;
|
|
17451
17660
|
}
|
|
17452
17661
|
//#endregion
|
|
17453
17662
|
//#region src/lib/execution-plan-cache.ts
|
|
17663
|
+
var ProducerContextResolutionError = class extends Error {
|
|
17664
|
+
constructor(message) {
|
|
17665
|
+
super(message);
|
|
17666
|
+
this.name = "ProducerContextResolutionError";
|
|
17667
|
+
}
|
|
17668
|
+
};
|
|
17454
17669
|
function createExecutionPlanCache(args) {
|
|
17455
17670
|
const cache = /* @__PURE__ */ new Map();
|
|
17456
17671
|
return {
|
|
@@ -17458,7 +17673,7 @@ function createExecutionPlanCache(args) {
|
|
|
17458
17673
|
const key = buildClaimedTaskKey(claimedTask);
|
|
17459
17674
|
const existing = cache.get(key);
|
|
17460
17675
|
if (existing) return existing;
|
|
17461
|
-
const plan = buildDaemonTaskExecutionPlan(claimedTask.task, args.stateDirs, args.slotIdentity, args.warmSessionTtlSec);
|
|
17676
|
+
const plan = maybeAttachProducerContext(claimedTask, buildDaemonTaskExecutionPlan(claimedTask.task, args.stateDirs, args.slotIdentity, args.warmSessionTtlSec), args.stateDirs, args.slotRegistry);
|
|
17462
17677
|
cache.set(key, plan);
|
|
17463
17678
|
return plan;
|
|
17464
17679
|
},
|
|
@@ -17470,17 +17685,90 @@ function createExecutionPlanCache(args) {
|
|
|
17470
17685
|
function buildClaimedTaskKey(task) {
|
|
17471
17686
|
return `${task.task.id}:${task.attemptN}`;
|
|
17472
17687
|
}
|
|
17688
|
+
function maybeAttachProducerContext(claimedTask, basePlan, stateDirs, slotRegistry) {
|
|
17689
|
+
if (claimedTask.task.taskType !== "judge_eval_attempt") return basePlan;
|
|
17690
|
+
const targetTaskId = typeof claimedTask.task.input.targetTaskId === "string" ? claimedTask.task.input.targetTaskId : null;
|
|
17691
|
+
const targetAttemptN = typeof claimedTask.task.input.targetAttemptN === "number" ? claimedTask.task.input.targetAttemptN : null;
|
|
17692
|
+
if (!targetTaskId || !targetAttemptN) throw new ProducerContextResolutionError("judge_eval_attempt is missing targetTaskId/targetAttemptN");
|
|
17693
|
+
const producer = slotRegistry.findLatestProducerSlotByTaskAttempt(targetTaskId, targetAttemptN);
|
|
17694
|
+
if (!producer) throw new ProducerContextResolutionError(`No persisted producer daemon slot found for task ${targetTaskId} attempt ${targetAttemptN}`);
|
|
17695
|
+
const sourceSessionPath = resolveProducerSessionPath(producer);
|
|
17696
|
+
if (!sourceSessionPath) throw new ProducerContextResolutionError(`Producer task ${targetTaskId} attempt ${targetAttemptN} has no persisted Pi session path`);
|
|
17697
|
+
const attachedWorkspace = resolveProducerWorkspaceAttachment(producer, stateDirs);
|
|
17698
|
+
return {
|
|
17699
|
+
...basePlan,
|
|
17700
|
+
workspaceMode: attachedWorkspace.mode,
|
|
17701
|
+
worktreeBranch: attachedWorkspace.branch,
|
|
17702
|
+
workspaceAttachment: {
|
|
17703
|
+
mountPath: attachedWorkspace.mountPath,
|
|
17704
|
+
cwdPath: attachedWorkspace.cwdPath,
|
|
17705
|
+
shadowWrites: "tmpfs"
|
|
17706
|
+
},
|
|
17707
|
+
sessionPersistence: {
|
|
17708
|
+
sessionDir: `${stateDirs.piSessionsDir}/judge-${claimedTask.task.id}-attempt-${claimedTask.attemptN}`,
|
|
17709
|
+
forkFromSessionPath: sourceSessionPath
|
|
17710
|
+
}
|
|
17711
|
+
};
|
|
17712
|
+
}
|
|
17713
|
+
function resolveProducerSessionPath(producer) {
|
|
17714
|
+
const explicit = producer.session?.sessionPath ?? null;
|
|
17715
|
+
if (explicit && existsSync(explicit)) return explicit;
|
|
17716
|
+
const sessionDir = producer.session?.sessionDir ?? null;
|
|
17717
|
+
if (!sessionDir || !existsSync(sessionDir)) return null;
|
|
17718
|
+
const latest = resolveLatestPiSessionPath(sessionDir);
|
|
17719
|
+
return latest && existsSync(latest) ? latest : null;
|
|
17720
|
+
}
|
|
17721
|
+
function resolveProducerWorkspaceAttachment(producer, stateDirs) {
|
|
17722
|
+
const workspacePath = producer.workspace?.worktreePath ?? null;
|
|
17723
|
+
if (workspacePath) {
|
|
17724
|
+
if (existsSync(workspacePath)) return {
|
|
17725
|
+
mountPath: workspacePath,
|
|
17726
|
+
cwdPath: workspacePath,
|
|
17727
|
+
mode: producer.workspace?.worktreeBranch ? "dedicated_worktree" : "scratch_mount",
|
|
17728
|
+
branch: producer.workspace?.worktreeBranch ?? null
|
|
17729
|
+
};
|
|
17730
|
+
const recoveredPath = recoverScratchWorkspacePath(producer, stateDirs);
|
|
17731
|
+
if (recoveredPath) return {
|
|
17732
|
+
mountPath: recoveredPath,
|
|
17733
|
+
cwdPath: recoveredPath,
|
|
17734
|
+
mode: "scratch_mount",
|
|
17735
|
+
branch: null
|
|
17736
|
+
};
|
|
17737
|
+
throw new ProducerContextResolutionError(`Producer workspace path is missing on disk: ${workspacePath}`);
|
|
17738
|
+
}
|
|
17739
|
+
const sharedMountRoot = dirname(dirname(stateDirs.rootDir));
|
|
17740
|
+
if (!existsSync(sharedMountRoot)) throw new ProducerContextResolutionError(`Shared producer mount root is missing on disk: ${sharedMountRoot}`);
|
|
17741
|
+
return {
|
|
17742
|
+
mountPath: sharedMountRoot,
|
|
17743
|
+
cwdPath: sharedMountRoot,
|
|
17744
|
+
mode: "shared_mount",
|
|
17745
|
+
branch: null
|
|
17746
|
+
};
|
|
17747
|
+
}
|
|
17748
|
+
function recoverScratchWorkspacePath(producer, stateDirs) {
|
|
17749
|
+
if (producer.workspace?.worktreeBranch) return null;
|
|
17750
|
+
if (!producer.workspace?.workspaceId) return null;
|
|
17751
|
+
const fallback = join(stateDirs.rootDir, "task-workspaces", producer.workspace.workspaceId);
|
|
17752
|
+
return existsSync(fallback) ? fallback : null;
|
|
17753
|
+
}
|
|
17473
17754
|
//#endregion
|
|
17474
17755
|
//#region src/lib/finalize.ts
|
|
17475
17756
|
async function finalizeTask(agent, output, ctx = {}) {
|
|
17476
17757
|
if (output.status === "cancelled") return;
|
|
17477
17758
|
if (output.status === "completed" && output.output && output.outputCid) {
|
|
17478
|
-
|
|
17479
|
-
output
|
|
17480
|
-
|
|
17481
|
-
|
|
17482
|
-
|
|
17483
|
-
|
|
17759
|
+
try {
|
|
17760
|
+
await agent.tasks.complete(output.taskId, output.attemptN, {
|
|
17761
|
+
output: output.output,
|
|
17762
|
+
outputCid: output.outputCid,
|
|
17763
|
+
usage: output.usage,
|
|
17764
|
+
...output.contentSignature ? { contentSignature: output.contentSignature } : {}
|
|
17765
|
+
});
|
|
17766
|
+
} catch (err) {
|
|
17767
|
+
const reason = errorToFailReason(err);
|
|
17768
|
+
ctx.log?.("complete-rejected-falling-back-to-fail", err);
|
|
17769
|
+
await agent.tasks.fail(output.taskId, output.attemptN, { error: reason });
|
|
17770
|
+
return;
|
|
17771
|
+
}
|
|
17484
17772
|
await maybeWriteAnchors(output, ctx);
|
|
17485
17773
|
return;
|
|
17486
17774
|
}
|
|
@@ -17492,6 +17780,21 @@ async function finalizeTask(agent, output, ctx = {}) {
|
|
|
17492
17780
|
if ((await agent.tasks.heartbeat(output.taskId, output.attemptN, {})).cancelled) return;
|
|
17493
17781
|
await agent.tasks.fail(output.taskId, output.attemptN, { error });
|
|
17494
17782
|
}
|
|
17783
|
+
function errorToFailReason(err) {
|
|
17784
|
+
if (err instanceof MoltNetError) {
|
|
17785
|
+
const fields = err.validationErrors?.length ? "; " + err.validationErrors.map((e) => `${e.field}: ${e.message}`).join(" | ") : "";
|
|
17786
|
+
return {
|
|
17787
|
+
code: "output_rejected_by_server",
|
|
17788
|
+
message: `Server rejected tasks.complete (${err.code}, status ${err.statusCode ?? "?"}): ${err.detail ?? err.message}${fields}`,
|
|
17789
|
+
retryable: false
|
|
17790
|
+
};
|
|
17791
|
+
}
|
|
17792
|
+
return {
|
|
17793
|
+
code: "complete_call_failed",
|
|
17794
|
+
message: err instanceof Error ? err.message : String(err),
|
|
17795
|
+
retryable: false
|
|
17796
|
+
};
|
|
17797
|
+
}
|
|
17495
17798
|
async function maybeWriteAnchors(output, ctx) {
|
|
17496
17799
|
const { task, writeCorrelationAnchors, log } = ctx;
|
|
17497
17800
|
if (!task || task.taskType !== "fulfill_brief") return;
|
|
@@ -17793,7 +18096,8 @@ async function runPolling(opts) {
|
|
|
17793
18096
|
const executionPlans = createExecutionPlanCache({
|
|
17794
18097
|
stateDirs,
|
|
17795
18098
|
slotIdentity,
|
|
17796
|
-
warmSessionTtlSec: common.warmSessionTtlSec
|
|
18099
|
+
warmSessionTtlSec: common.warmSessionTtlSec,
|
|
18100
|
+
slotRegistry
|
|
17797
18101
|
});
|
|
17798
18102
|
const ctx = await resolveAgentContext(common.agent);
|
|
17799
18103
|
const cfg = loadConfig();
|
|
@@ -17838,15 +18142,9 @@ async function runPolling(opts) {
|
|
|
17838
18142
|
maxPollIntervalMs
|
|
17839
18143
|
}, "agent-daemon.starting");
|
|
17840
18144
|
const outputs = [];
|
|
17841
|
-
const subagentContractRegistry = createSubagentContractRegistry([{
|
|
17842
|
-
name: "judge_eval_variant_result",
|
|
17843
|
-
description: "Per-variant grading result produced by a subagent of judge_eval_variant: scores against the shared rubric, composite, and a 1-3 sentence verdict for a single variant.",
|
|
17844
|
-
parametersSchema: JudgeEvalVariantResult
|
|
17845
|
-
}]);
|
|
17846
18145
|
try {
|
|
17847
18146
|
const executeTask = createPiTaskExecutor({
|
|
17848
18147
|
agentName: common.agent,
|
|
17849
|
-
subagentContractRegistry,
|
|
17850
18148
|
mountPath: sandbox.rootDir,
|
|
17851
18149
|
provider: common.provider,
|
|
17852
18150
|
model: common.model,
|
|
@@ -17890,7 +18188,34 @@ async function runPolling(opts) {
|
|
|
17890
18188
|
log: (msg, err) => rootLogger.warn({ err }, msg)
|
|
17891
18189
|
}),
|
|
17892
18190
|
executeTask: async (claimedTask, reporter) => {
|
|
17893
|
-
|
|
18191
|
+
let executionPlan;
|
|
18192
|
+
try {
|
|
18193
|
+
executionPlan = executionPlans.getOrCreate(claimedTask);
|
|
18194
|
+
} catch (err) {
|
|
18195
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
18196
|
+
rootLogger.warn({
|
|
18197
|
+
taskId: claimedTask.task.id,
|
|
18198
|
+
attemptN: claimedTask.attemptN,
|
|
18199
|
+
err: message
|
|
18200
|
+
}, "agent-daemon.execution_plan_failed");
|
|
18201
|
+
return {
|
|
18202
|
+
taskId: claimedTask.task.id,
|
|
18203
|
+
attemptN: claimedTask.attemptN,
|
|
18204
|
+
status: "failed",
|
|
18205
|
+
output: null,
|
|
18206
|
+
outputCid: null,
|
|
18207
|
+
usage: {
|
|
18208
|
+
inputTokens: 0,
|
|
18209
|
+
outputTokens: 0
|
|
18210
|
+
},
|
|
18211
|
+
durationMs: 0,
|
|
18212
|
+
error: {
|
|
18213
|
+
code: err instanceof ProducerContextResolutionError ? "producer_context_missing" : "execution_plan_failed",
|
|
18214
|
+
message,
|
|
18215
|
+
retryable: false
|
|
18216
|
+
}
|
|
18217
|
+
};
|
|
18218
|
+
}
|
|
17894
18219
|
const sessionDescriptor = executionPlan.descriptor;
|
|
17895
18220
|
let expired;
|
|
17896
18221
|
try {
|
|
@@ -17909,7 +18234,7 @@ async function runPolling(opts) {
|
|
|
17909
18234
|
taskId: claimedTask.task.id,
|
|
17910
18235
|
taskType: claimedTask.task.taskType,
|
|
17911
18236
|
resumable: sessionDescriptor.policy.resumable,
|
|
17912
|
-
workspaceMode:
|
|
18237
|
+
workspaceMode: executionPlan.workspaceMode,
|
|
17913
18238
|
workspaceScope: sessionDescriptor.policy.workspaceScope,
|
|
17914
18239
|
sessionScope: sessionDescriptor.policy.sessionScope,
|
|
17915
18240
|
slotKey: executionPlan.slotKey,
|
|
@@ -17959,7 +18284,7 @@ async function runPolling(opts) {
|
|
|
17959
18284
|
sessionDir: executionPlan.sessionPersistence.sessionDir,
|
|
17960
18285
|
sessionPath: resolveLatestPiSessionPath(executionPlan.sessionPersistence.sessionDir),
|
|
17961
18286
|
workspaceId: executionPlan.workspaceId,
|
|
17962
|
-
worktreePath:
|
|
18287
|
+
worktreePath: resolveRecordedWorkspacePath$1(mainRepo, stateDirs.rootDir, executionPlan),
|
|
17963
18288
|
worktreeBranch: executionPlan.worktreeBranch,
|
|
17964
18289
|
lastTaskId: claimedTask.task.id,
|
|
17965
18290
|
lastAttemptN: claimedTask.attemptN,
|
|
@@ -17983,6 +18308,10 @@ async function runPolling(opts) {
|
|
|
17983
18308
|
await shutdownLogger();
|
|
17984
18309
|
}
|
|
17985
18310
|
}
|
|
18311
|
+
function resolveRecordedWorkspacePath$1(mainRepo, stateRootDir, executionPlan) {
|
|
18312
|
+
if (!executionPlan.workspaceId) return null;
|
|
18313
|
+
return executionPlan.workspaceMode === "scratch_mount" ? join(stateRootDir, "task-workspaces", executionPlan.workspaceId) : join(mainRepo, ".worktrees", executionPlan.workspaceId);
|
|
18314
|
+
}
|
|
17986
18315
|
function parseCsv(raw) {
|
|
17987
18316
|
return (raw ?? "").split(",").map((s) => s.trim()).filter((s) => s.length > 0);
|
|
17988
18317
|
}
|
|
@@ -18050,7 +18379,8 @@ async function runOnce(argv) {
|
|
|
18050
18379
|
const executionPlans = createExecutionPlanCache({
|
|
18051
18380
|
stateDirs,
|
|
18052
18381
|
slotIdentity,
|
|
18053
|
-
warmSessionTtlSec: opts.warmSessionTtlSec
|
|
18382
|
+
warmSessionTtlSec: opts.warmSessionTtlSec,
|
|
18383
|
+
slotRegistry
|
|
18054
18384
|
});
|
|
18055
18385
|
const ctx = await resolveAgentContext(opts.agent);
|
|
18056
18386
|
const cfg = loadConfig();
|
|
@@ -18099,15 +18429,9 @@ async function runOnce(argv) {
|
|
|
18099
18429
|
process.on("SIGTERM", () => {
|
|
18100
18430
|
onSignal("SIGTERM");
|
|
18101
18431
|
});
|
|
18102
|
-
const subagentContractRegistry = createSubagentContractRegistry([{
|
|
18103
|
-
name: "judge_eval_variant_result",
|
|
18104
|
-
description: "Per-variant grading result produced by a subagent of judge_eval_variant: scores against the shared rubric, composite, and a 1-3 sentence verdict for a single variant.",
|
|
18105
|
-
parametersSchema: JudgeEvalVariantResult
|
|
18106
|
-
}]);
|
|
18107
18432
|
try {
|
|
18108
18433
|
const rawExecuteTask = createPiTaskExecutor({
|
|
18109
18434
|
agentName: opts.agent,
|
|
18110
|
-
subagentContractRegistry,
|
|
18111
18435
|
mountPath: sandbox.rootDir,
|
|
18112
18436
|
provider: opts.provider,
|
|
18113
18437
|
model: opts.model,
|
|
@@ -18130,7 +18454,34 @@ async function runOnce(argv) {
|
|
|
18130
18454
|
err: err instanceof Error ? err.message : String(err)
|
|
18131
18455
|
}, "agent-daemon.daemon_slot_reap_failed");
|
|
18132
18456
|
}
|
|
18133
|
-
|
|
18457
|
+
let executionPlan;
|
|
18458
|
+
try {
|
|
18459
|
+
executionPlan = executionPlans.getOrCreate(claimedTask);
|
|
18460
|
+
} catch (err) {
|
|
18461
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
18462
|
+
rootLogger.warn({
|
|
18463
|
+
taskId: claimedTask.task.id,
|
|
18464
|
+
attemptN: claimedTask.attemptN,
|
|
18465
|
+
err: message
|
|
18466
|
+
}, "agent-daemon.execution_plan_failed");
|
|
18467
|
+
return {
|
|
18468
|
+
taskId: claimedTask.task.id,
|
|
18469
|
+
attemptN: claimedTask.attemptN,
|
|
18470
|
+
status: "failed",
|
|
18471
|
+
output: null,
|
|
18472
|
+
outputCid: null,
|
|
18473
|
+
usage: {
|
|
18474
|
+
inputTokens: 0,
|
|
18475
|
+
outputTokens: 0
|
|
18476
|
+
},
|
|
18477
|
+
durationMs: 0,
|
|
18478
|
+
error: {
|
|
18479
|
+
code: err instanceof ProducerContextResolutionError ? "producer_context_missing" : "execution_plan_failed",
|
|
18480
|
+
message,
|
|
18481
|
+
retryable: false
|
|
18482
|
+
}
|
|
18483
|
+
};
|
|
18484
|
+
}
|
|
18134
18485
|
if (executionPlan.slotKey && executionPlan.sessionPersistence) slotRegistry.beginSlot({
|
|
18135
18486
|
...slotIdentity,
|
|
18136
18487
|
slotKey: executionPlan.slotKey,
|
|
@@ -18138,7 +18489,7 @@ async function runOnce(argv) {
|
|
|
18138
18489
|
sessionDir: executionPlan.sessionPersistence.sessionDir,
|
|
18139
18490
|
sessionPath: resolveLatestPiSessionPath(executionPlan.sessionPersistence.sessionDir),
|
|
18140
18491
|
workspaceId: executionPlan.workspaceId,
|
|
18141
|
-
worktreePath:
|
|
18492
|
+
worktreePath: resolveRecordedWorkspacePath(mainRepo, stateDirs.rootDir, executionPlan),
|
|
18142
18493
|
worktreeBranch: executionPlan.worktreeBranch,
|
|
18143
18494
|
lastTaskId: claimedTask.task.id,
|
|
18144
18495
|
lastAttemptN: claimedTask.attemptN,
|
|
@@ -18190,6 +18541,10 @@ async function runOnce(argv) {
|
|
|
18190
18541
|
await shutdownLogger();
|
|
18191
18542
|
}
|
|
18192
18543
|
}
|
|
18544
|
+
function resolveRecordedWorkspacePath(mainRepo, stateRootDir, executionPlan) {
|
|
18545
|
+
if (!executionPlan.workspaceId) return null;
|
|
18546
|
+
return executionPlan.workspaceMode === "scratch_mount" ? join(stateRootDir, "task-workspaces", executionPlan.workspaceId) : join(mainRepo, ".worktrees", executionPlan.workspaceId);
|
|
18547
|
+
}
|
|
18193
18548
|
//#endregion
|
|
18194
18549
|
//#region src/cli/poll.ts
|
|
18195
18550
|
function runPoll(argv) {
|