@themoltnet/agent-daemon 0.8.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +42 -0
- package/dist/main.js +983 -453
- package/package.json +6 -6
package/dist/main.js
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import crypto, { createHash } from "crypto";
|
|
3
|
+
import path, { dirname, isAbsolute, join, relative, resolve } from "node:path";
|
|
3
4
|
import { parseArgs, parseEnv, promisify } from "node:util";
|
|
4
5
|
import { existsSync, mkdirSync, readFileSync, readdirSync, rmSync, statSync } from "node:fs";
|
|
5
6
|
import { ROOT_CONTEXT, SpanStatusCode, context, metrics, propagation, trace } from "@opentelemetry/api";
|
|
6
7
|
import { pino, transport } from "pino";
|
|
7
8
|
import { readFile } from "node:fs/promises";
|
|
8
9
|
import { createHash as createHash$1 } from "node:crypto";
|
|
9
|
-
import path, { dirname, isAbsolute, join, relative, resolve } from "node:path";
|
|
10
10
|
import { homedir } from "node:os";
|
|
11
11
|
import { execFile, execFileSync } from "node:child_process";
|
|
12
12
|
import { DefaultResourceLoader, SessionManager, createAgentSession, createBashToolDefinition, createEditToolDefinition, createReadToolDefinition, createSyntheticSourceInfo, createWriteToolDefinition, defineTool, parseFrontmatter } from "@earendil-works/pi-coding-agent";
|
|
@@ -2896,6 +2896,7 @@ if (!Has$1("date-time")) Set$1("date-time", (v) => !Number.isNaN(Date.parse(v)))
|
|
|
2896
2896
|
*/
|
|
2897
2897
|
var ContextBinding = Type$2.Union([
|
|
2898
2898
|
Type$2.Literal("skill"),
|
|
2899
|
+
Type$2.Literal("context_inline"),
|
|
2899
2900
|
Type$2.Literal("prompt_prefix"),
|
|
2900
2901
|
Type$2.Literal("user_inline")
|
|
2901
2902
|
], { $id: "ContextBinding" });
|
|
@@ -2912,9 +2913,14 @@ var ContextBinding = Type$2.Union([
|
|
|
2912
2913
|
* name under the runtime's skill discovery path. Must be
|
|
2913
2914
|
* kebab-case-safe (alphanumeric + dashes/underscores).
|
|
2914
2915
|
* - `binding` — how the bytes are delivered to the LLM (see above).
|
|
2915
|
-
* - `content` — the actual bytes (UTF-8 text). Capped at
|
|
2916
|
+
* - `content` — the actual bytes (UTF-8 text). Capped at 64 KiB per
|
|
2916
2917
|
* entry; total per-task context bytes are bounded by the
|
|
2917
2918
|
* soft `maxItems` cap and per-binding daemon limits.
|
|
2919
|
+
* Raised from 32 KiB in 2026-05 — protocol-heavy operator
|
|
2920
|
+
* skills (e.g. `.claude/skills/legreffier/SKILL.md`) ship
|
|
2921
|
+
* at ~35 KiB inline, and the original cap was sized for
|
|
2922
|
+
* short example skills, not the kind of skill the eval
|
|
2923
|
+
* substrate is dogfooded on (#943, #823).
|
|
2918
2924
|
*/
|
|
2919
2925
|
var ContextRef = Type$2.Object({
|
|
2920
2926
|
slug: Type$2.String({
|
|
@@ -2925,7 +2931,7 @@ var ContextRef = Type$2.Object({
|
|
|
2925
2931
|
binding: ContextBinding,
|
|
2926
2932
|
content: Type$2.String({
|
|
2927
2933
|
minLength: 1,
|
|
2928
|
-
maxLength:
|
|
2934
|
+
maxLength: 65536
|
|
2929
2935
|
})
|
|
2930
2936
|
}, {
|
|
2931
2937
|
$id: "ContextRef",
|
|
@@ -4330,61 +4336,33 @@ async function validateJudgePackInputAsync(input, ctx) {
|
|
|
4330
4336
|
return errors;
|
|
4331
4337
|
}
|
|
4332
4338
|
//#endregion
|
|
4333
|
-
//#region ../../libs/tasks/src/task-types/judge-eval-
|
|
4339
|
+
//#region ../../libs/tasks/src/task-types/judge-eval-attempt.ts
|
|
4334
4340
|
/**
|
|
4335
|
-
* `
|
|
4336
|
-
*
|
|
4337
|
-
* isolation.
|
|
4341
|
+
* `judge_eval_attempt` — score one completed `run_eval` attempt against a
|
|
4342
|
+
* hidden judge rubric.
|
|
4338
4343
|
*
|
|
4339
4344
|
* output_kind: judgment
|
|
4340
|
-
* criteria: required (`successCriteria.rubric`
|
|
4341
|
-
*
|
|
4342
|
-
*
|
|
4343
|
-
* pin the targets being graded.
|
|
4344
|
-
*
|
|
4345
|
-
* Slice 2 of #943. The parent task carries the rubric and the list of
|
|
4346
|
-
* variant `run_eval` task ids. The pi executor registers the generic
|
|
4347
|
-
* `subagent` custom tool (#1087), and the parent LLM calls
|
|
4348
|
-
* `subagent({ task, output_schema: 'judge_eval_variant_result' })` once
|
|
4349
|
-
* per variant — each child session has fresh context, fetches the
|
|
4350
|
-
* variant's accepted attempt output via `moltnet_get_task` /
|
|
4351
|
-
* `moltnet_list_task_attempts`, and grades against the rubric.
|
|
4352
|
-
*
|
|
4353
|
-
* Reuses `JudgePackScore` from `judge_pack` for per-criterion scoring
|
|
4354
|
-
* (Lane 1 binary via `llm_checklist`, Lane 2 graded via `llm_score`,
|
|
4355
|
-
* deterministic_*) — the score shape is the same across judgment
|
|
4356
|
-
* tasks; only the wrapping (per-variant grouping + deltas) differs.
|
|
4345
|
+
* criteria: required (`successCriteria.rubric`)
|
|
4346
|
+
* references: not required at the input layer — `targetTaskId` +
|
|
4347
|
+
* `targetAttemptN` pin the producer attempt being judged.
|
|
4357
4348
|
*
|
|
4358
|
-
*
|
|
4359
|
-
*
|
|
4360
|
-
*
|
|
4361
|
-
*
|
|
4362
|
-
|
|
4363
|
-
|
|
4364
|
-
|
|
4365
|
-
|
|
4366
|
-
|
|
4367
|
-
var JudgeEvalVariantInput = Type$2.Object({
|
|
4368
|
-
runTaskIds: Type$2.Array(Type$2.String({ format: "uuid" }), {
|
|
4369
|
-
minItems: 2,
|
|
4370
|
-
maxItems: 10
|
|
4371
|
-
}),
|
|
4349
|
+
* This replaces the earlier parent/subagent `judge_eval_variant` design.
|
|
4350
|
+
* The unit of judgment is one producer attempt. Cross-variant deltas can be
|
|
4351
|
+
* computed later at read time from stored scores, rather than materialized as
|
|
4352
|
+
* their own task output.
|
|
4353
|
+
*/
|
|
4354
|
+
var JUDGE_EVAL_ATTEMPT_TYPE = "judge_eval_attempt";
|
|
4355
|
+
var JudgeEvalAttemptInput = Type$2.Object({
|
|
4356
|
+
targetTaskId: Type$2.String({ format: "uuid" }),
|
|
4357
|
+
targetAttemptN: Type$2.Integer({ minimum: 1 }),
|
|
4372
4358
|
successCriteria: SuccessCriteria
|
|
4373
4359
|
}, {
|
|
4374
|
-
$id: "
|
|
4360
|
+
$id: "JudgeEvalAttemptInput",
|
|
4375
4361
|
additionalProperties: false
|
|
4376
4362
|
});
|
|
4377
|
-
|
|
4378
|
-
|
|
4379
|
-
|
|
4380
|
-
* deterministic_*). Reuse the type rather than re-declare.
|
|
4381
|
-
*
|
|
4382
|
-
* This is also the **subagent output contract** — the parent's
|
|
4383
|
-
* `subagent` tool resolves the contract name `judge_eval_variant_result`
|
|
4384
|
-
* to this schema. See `agent-runtime`'s subagent contract registry.
|
|
4385
|
-
*/
|
|
4386
|
-
var JudgeEvalVariantResult = Type$2.Object({
|
|
4387
|
-
runTaskId: Type$2.String({ format: "uuid" }),
|
|
4363
|
+
var JudgeEvalAttemptOutput = Type$2.Object({
|
|
4364
|
+
targetTaskId: Type$2.String({ format: "uuid" }),
|
|
4365
|
+
targetAttemptN: Type$2.Integer({ minimum: 1 }),
|
|
4388
4366
|
variantLabel: Type$2.String({
|
|
4389
4367
|
minLength: 1,
|
|
4390
4368
|
maxLength: 64,
|
|
@@ -4395,219 +4373,195 @@ var JudgeEvalVariantResult = Type$2.Object({
|
|
|
4395
4373
|
minimum: 0,
|
|
4396
4374
|
maximum: 1
|
|
4397
4375
|
}),
|
|
4398
|
-
verdict: Type$2.String({ minLength: 1 })
|
|
4399
|
-
}, {
|
|
4400
|
-
$id: "JudgeEvalVariantResult",
|
|
4401
|
-
additionalProperties: false
|
|
4402
|
-
});
|
|
4403
|
-
var JudgeEvalVariantOutput = Type$2.Object({
|
|
4404
|
-
results: Type$2.Array(JudgeEvalVariantResult, { minItems: 2 }),
|
|
4405
|
-
deltas: Type$2.Optional(Type$2.Record(Type$2.String(), Type$2.Number({
|
|
4406
|
-
minimum: -1,
|
|
4407
|
-
maximum: 1
|
|
4408
|
-
}))),
|
|
4376
|
+
verdict: Type$2.String({ minLength: 1 }),
|
|
4409
4377
|
judgeModel: Type$2.Optional(Type$2.String({ minLength: 1 })),
|
|
4410
4378
|
traceparent: Type$2.String({ minLength: 1 })
|
|
4411
4379
|
}, {
|
|
4412
|
-
$id: "
|
|
4380
|
+
$id: "JudgeEvalAttemptOutput",
|
|
4413
4381
|
additionalProperties: false
|
|
4414
4382
|
});
|
|
4415
|
-
|
|
4416
|
-
* Synchronous input invariants beyond TypeBox shape: rubric must be
|
|
4417
|
-
* present (already required by the schema, but the rubric body has
|
|
4418
|
-
* its own per-criterion weight invariant) and the rubric's weights
|
|
4419
|
-
* must sum to 1.
|
|
4420
|
-
*
|
|
4421
|
-
* Cross-task invariants (all targets are `run_eval`, all completed,
|
|
4422
|
-
* share `correlation_id`, byte-identical `input.successCriteria`)
|
|
4423
|
-
* are NOT checked here — they require async DB lookups against
|
|
4424
|
-
* `runTaskIds` and live in `validateJudgeEvalVariantInputAsync`
|
|
4425
|
-
* below, invoked by the task service at create time (#1096).
|
|
4426
|
-
*/
|
|
4427
|
-
function validateJudgeEvalVariantInput(input) {
|
|
4383
|
+
function validateJudgeEvalAttemptInput(input) {
|
|
4428
4384
|
const sc = input.successCriteria;
|
|
4429
|
-
if (!sc) return "successCriteria is required for
|
|
4430
|
-
if (!sc.rubric) return "successCriteria.rubric is required for
|
|
4385
|
+
if (!sc) return "successCriteria is required for judge_eval_attempt";
|
|
4386
|
+
if (!sc.rubric) return "successCriteria.rubric is required for judge_eval_attempt";
|
|
4431
4387
|
return validateRubricWeights(sc.rubric);
|
|
4432
4388
|
}
|
|
4433
|
-
|
|
4434
|
-
* Output cross-field invariants the schema cannot express:
|
|
4435
|
-
*
|
|
4436
|
-
* 1. `results.length === input.runTaskIds.length` — every variant
|
|
4437
|
-
* the imposer asked for must be graded. Partial grading
|
|
4438
|
-
* invalidates cross-variant comparison; fail the whole task
|
|
4439
|
-
* rather than silently report a subset.
|
|
4440
|
-
*
|
|
4441
|
-
* 2. `results[i].runTaskId === input.runTaskIds[i]` — order is
|
|
4442
|
-
* load-bearing for downstream consumers (e.g. deltas keyed by
|
|
4443
|
-
* adjacent pairs). Mismatch is an LLM bug; reject loudly.
|
|
4444
|
-
*
|
|
4445
|
-
* 3. Each `result.scores` follows the same `llm_checklist` rule
|
|
4446
|
-
* `judge_pack` enforces (#999): if a score has an `assertions`
|
|
4447
|
-
* array, the numeric score MUST be `1` iff every assertion
|
|
4448
|
-
* passes. Inconsistent payloads pollute attestations.
|
|
4449
|
-
*
|
|
4450
|
-
* 4. Each `result.composite` MUST equal the rubric-weighted sum
|
|
4451
|
-
* `Σ(weight_j × scores[j].score)`. The parent (and any subagent
|
|
4452
|
-
* it delegated to) is supposed to compute this; surfacing a
|
|
4453
|
-
* drift here catches LLMs that hand-wave the arithmetic.
|
|
4454
|
-
*
|
|
4455
|
-
* 5. Optional `deltas` keys MUST be of the form `"A - B"` where
|
|
4456
|
-
* both `A` and `B` are variantLabels present in `results`.
|
|
4457
|
-
* Values are not range-checked (any float in [-1, 1] is
|
|
4458
|
-
* arithmetically possible).
|
|
4459
|
-
*/
|
|
4460
|
-
function validateJudgeEvalVariantOutput(output, input) {
|
|
4389
|
+
function validateJudgeEvalAttemptOutput(output, input) {
|
|
4461
4390
|
const out = output;
|
|
4462
4391
|
const inp = input;
|
|
4463
4392
|
if (inp) {
|
|
4464
|
-
if (out.
|
|
4465
|
-
|
|
4466
|
-
}
|
|
4467
|
-
for (let
|
|
4468
|
-
const
|
|
4469
|
-
|
|
4470
|
-
|
|
4471
|
-
|
|
4472
|
-
|
|
4473
|
-
const expected = allPassed ? 1 : 0;
|
|
4474
|
-
if (sc.score !== expected) return `results[${r}].scores[${s}] (criterionId="${sc.criterionId}"): assertions ${allPassed ? "all pass" : "have at least one fail"} but score=${sc.score}. Score must be derived: 1 iff every assertion passes, else 0 (#999 llm_checklist rule).`;
|
|
4475
|
-
}
|
|
4393
|
+
if (out.targetTaskId !== inp.targetTaskId) return `output.targetTaskId (${out.targetTaskId}) does not match input.targetTaskId (${inp.targetTaskId})`;
|
|
4394
|
+
if (out.targetAttemptN !== inp.targetAttemptN) return `output.targetAttemptN (${out.targetAttemptN}) does not match input.targetAttemptN (${inp.targetAttemptN})`;
|
|
4395
|
+
}
|
|
4396
|
+
for (let s = 0; s < out.scores.length; s++) {
|
|
4397
|
+
const sc = out.scores[s];
|
|
4398
|
+
if (!sc.assertions) continue;
|
|
4399
|
+
const allPassed = sc.assertions.every((a) => a.passed);
|
|
4400
|
+
const expected = allPassed ? 1 : 0;
|
|
4401
|
+
if (sc.score !== expected) return `scores[${s}] (criterionId="${sc.criterionId}"): assertions ${allPassed ? "all pass" : "have at least one fail"} but score=${sc.score}. Score must be 1 iff every assertion passes, else 0.`;
|
|
4476
4402
|
}
|
|
4477
4403
|
if (inp?.successCriteria?.rubric) {
|
|
4478
4404
|
const criteria = inp.successCriteria.rubric.criteria;
|
|
4479
4405
|
const weightById = new Map(criteria.map((c) => [c.id, c.weight]));
|
|
4480
|
-
|
|
4481
|
-
|
|
4482
|
-
|
|
4483
|
-
|
|
4484
|
-
|
|
4485
|
-
if (w === void 0) return `results[${r}].scores: criterionId "${sc.criterionId}" is not in the input rubric (known: ${Array.from(weightById.keys()).join(", ")}). Score every rubric criterion exactly once; do not invent new ids.`;
|
|
4486
|
-
sum += w * sc.score;
|
|
4487
|
-
}
|
|
4488
|
-
if (Math.abs(sum - result.composite) > .001) return `results[${r}].composite (${result.composite}) does not match Σ(weight × score) (${sum.toFixed(6)}). Composite must be the rubric-weighted sum of per-criterion scores (drift > 0.001).`;
|
|
4489
|
-
}
|
|
4490
|
-
}
|
|
4491
|
-
if (out.deltas) {
|
|
4492
|
-
const labels = new Set(out.results.map((r) => r.variantLabel));
|
|
4493
|
-
for (const key of Object.keys(out.deltas)) {
|
|
4494
|
-
const m = /^(.+?) - (.+)$/.exec(key);
|
|
4495
|
-
if (!m) return `deltas key "${key}" is not of the form "<variantLabel-A> - <variantLabel-B>". Use a single space-hyphen-space separator between labels.`;
|
|
4496
|
-
const [, a, b] = m;
|
|
4497
|
-
if (!labels.has(a) || !labels.has(b)) return `deltas key "${key}" references variantLabel(s) not present in results: ${!labels.has(a) ? `"${a}" missing` : ""}${!labels.has(a) && !labels.has(b) ? ", " : ""}${!labels.has(b) ? `"${b}" missing` : ""}`;
|
|
4406
|
+
let sum = 0;
|
|
4407
|
+
for (const sc of out.scores) {
|
|
4408
|
+
const w = weightById.get(sc.criterionId);
|
|
4409
|
+
if (w === void 0) return `scores references unknown criterionId "${sc.criterionId}"`;
|
|
4410
|
+
sum += w * sc.score;
|
|
4498
4411
|
}
|
|
4412
|
+
const rounded = Math.round(sum * 1e3) / 1e3;
|
|
4413
|
+
if (Math.abs(rounded - out.composite) > .001) return `composite (${out.composite}) does not match weighted rubric sum (${rounded})`;
|
|
4499
4414
|
}
|
|
4500
4415
|
return null;
|
|
4501
4416
|
}
|
|
4502
|
-
|
|
4503
|
-
|
|
4504
|
-
* equality. Recursively sorts object keys; arrays preserve order
|
|
4505
|
-
* (intentional — rubric criteria order is semantically meaningful).
|
|
4506
|
-
* Mirrors the canonical-JSON shape `crypto-service` uses for CIDs,
|
|
4507
|
-
* without taking on a crypto-service dep just for this comparison.
|
|
4508
|
-
*/
|
|
4509
|
-
function stableStringify(value) {
|
|
4510
|
-
if (value === null || typeof value !== "object") return JSON.stringify(value);
|
|
4511
|
-
if (Array.isArray(value)) return "[" + value.map(stableStringify).join(",") + "]";
|
|
4512
|
-
const obj = value;
|
|
4513
|
-
return "{" + Object.keys(obj).sort().map((k) => JSON.stringify(k) + ":" + stableStringify(obj[k])).join(",") + "}";
|
|
4514
|
-
}
|
|
4515
|
-
/**
|
|
4516
|
-
* Async preflight for `judge_eval_variant` (#1096 + #943):
|
|
4517
|
-
*
|
|
4518
|
-
* 1. Every `runTaskIds[i]` resolves to a task the caller can read.
|
|
4519
|
-
* 2. Every resolved task is `taskType === 'run_eval'`.
|
|
4520
|
-
* 3. Every resolved task is `status === 'completed'` with a
|
|
4521
|
-
* non-null `acceptedAttemptN` — grading an unaccepted attempt
|
|
4522
|
-
* races with re-attempts and pollutes the judge attestation.
|
|
4523
|
-
* 4. Every resolved task shares a non-null `correlationId`, and all
|
|
4524
|
-
* `correlationId`s are equal. Without this an imposer could
|
|
4525
|
-
* fabricate a "variant set" by stapling unrelated runs together.
|
|
4526
|
-
* 5. The shared `correlationId` is NOT already sealed. A previous
|
|
4527
|
-
* judge_eval_variant against the same group is final; produce a
|
|
4528
|
-
* fresh correlation_id for a new judging round rather than
|
|
4529
|
-
* adding contradictory verdicts to a sealed group.
|
|
4530
|
-
* 6. Every variant's `input.successCriteria` is byte-identical (via
|
|
4531
|
-
* stable-stringify). Different rubrics across "variants" makes
|
|
4532
|
-
* the comparison meaningless.
|
|
4533
|
-
*/
|
|
4534
|
-
async function validateJudgeEvalVariantInputAsync(input, ctx) {
|
|
4535
|
-
const { runTaskIds } = input;
|
|
4417
|
+
async function validateJudgeEvalAttemptInputAsync(input, ctx) {
|
|
4418
|
+
const inp = input;
|
|
4536
4419
|
const errors = [];
|
|
4537
|
-
const
|
|
4538
|
-
|
|
4539
|
-
|
|
4540
|
-
|
|
4541
|
-
|
|
4542
|
-
|
|
4543
|
-
|
|
4544
|
-
|
|
4545
|
-
|
|
4546
|
-
|
|
4547
|
-
|
|
4548
|
-
|
|
4549
|
-
}
|
|
4550
|
-
presentTargets.push(t);
|
|
4551
|
-
if (t.taskType !== "run_eval") errors.push({
|
|
4552
|
-
field: `runTaskIds[${i}]`,
|
|
4553
|
-
message: `runTaskIds[${i}]=${runTaskIds[i]} is a ${t.taskType}, not a run_eval`
|
|
4554
|
-
});
|
|
4555
|
-
if (t.status !== "completed" || t.acceptedAttemptN === null) errors.push({
|
|
4556
|
-
field: `runTaskIds[${i}]`,
|
|
4557
|
-
message: `runTaskIds[${i}]=${runTaskIds[i]} is not completed with an accepted attempt (status=${t.status}, acceptedAttemptN=${t.acceptedAttemptN})`
|
|
4558
|
-
});
|
|
4559
|
-
}
|
|
4560
|
-
if (missingTargets || presentTargets.length === 0) return errors;
|
|
4561
|
-
const correlationIds = new Set(presentTargets.map((t) => t.correlationId ?? "__null__"));
|
|
4562
|
-
if (correlationIds.has("__null__")) errors.push({
|
|
4563
|
-
field: "runTaskIds",
|
|
4564
|
-
message: "one or more run_eval targets have no correlation_id; cannot group as variants"
|
|
4420
|
+
const target = await ctx.resolveTask(inp.targetTaskId);
|
|
4421
|
+
if (!target) return [{
|
|
4422
|
+
field: "targetTaskId",
|
|
4423
|
+
message: `targetTaskId=${inp.targetTaskId} does not resolve to a task you can read`
|
|
4424
|
+
}];
|
|
4425
|
+
if (target.taskType !== "run_eval") errors.push({
|
|
4426
|
+
field: "targetTaskId",
|
|
4427
|
+
message: `targetTaskId=${inp.targetTaskId} is a ${target.taskType}, not a run_eval`
|
|
4428
|
+
});
|
|
4429
|
+
if (target.status !== "completed" || target.acceptedAttemptN === null) errors.push({
|
|
4430
|
+
field: "targetTaskId",
|
|
4431
|
+
message: `targetTaskId=${inp.targetTaskId} is not completed with an accepted attempt (status=${target.status}, acceptedAttemptN=${target.acceptedAttemptN})`
|
|
4565
4432
|
});
|
|
4566
|
-
if (
|
|
4567
|
-
field: "
|
|
4568
|
-
message: `
|
|
4433
|
+
else if (target.acceptedAttemptN !== inp.targetAttemptN) errors.push({
|
|
4434
|
+
field: "targetAttemptN",
|
|
4435
|
+
message: `targetAttemptN=${inp.targetAttemptN} does not match the producer's acceptedAttemptN=${target.acceptedAttemptN}`
|
|
4569
4436
|
});
|
|
4570
|
-
if (
|
|
4571
|
-
|
|
4572
|
-
|
|
4573
|
-
|
|
4574
|
-
if (
|
|
4575
|
-
|
|
4576
|
-
|
|
4437
|
+
if (!target.correlationId) errors.push({
|
|
4438
|
+
field: "targetTaskId",
|
|
4439
|
+
message: "target run_eval has no correlation_id; cannot enforce duplicate-judge protection"
|
|
4440
|
+
});
|
|
4441
|
+
if (errors.length > 0 || !target.correlationId) return errors;
|
|
4442
|
+
const rubric = inp.successCriteria.rubric;
|
|
4443
|
+
const duplicate = (await ctx.listTasksByCorrelation(target.correlationId)).find((task) => {
|
|
4444
|
+
if (task.taskType !== "judge_eval_attempt") return false;
|
|
4445
|
+
if (task.status === "failed" || task.status === "cancelled" || task.status === "expired") return false;
|
|
4446
|
+
const existing = task.input;
|
|
4447
|
+
const existingRubric = existing.successCriteria?.rubric;
|
|
4448
|
+
return existing.targetTaskId === inp.targetTaskId && existing.targetAttemptN === inp.targetAttemptN && existingRubric?.rubricId === rubric?.rubricId && existingRubric?.version === rubric?.version;
|
|
4449
|
+
});
|
|
4450
|
+
if (duplicate) errors.push({
|
|
4451
|
+
field: "targetTaskId",
|
|
4452
|
+
message: `judge task ${duplicate.id} already exists for (${inp.targetTaskId}, attempt ${inp.targetAttemptN}, rubric ${rubric?.rubricId}@${rubric?.version})`
|
|
4577
4453
|
});
|
|
4578
|
-
const first = stableStringify(presentTargets[0].input.successCriteria);
|
|
4579
|
-
for (let i = 1; i < presentTargets.length; i++) if (stableStringify(presentTargets[i].input.successCriteria) !== first) {
|
|
4580
|
-
errors.push({
|
|
4581
|
-
field: `runTaskIds[${i}]`,
|
|
4582
|
-
message: `runTaskIds[${i}] has a different input.successCriteria than runTaskIds[0]; all variants must share the rubric and gates`
|
|
4583
|
-
});
|
|
4584
|
-
break;
|
|
4585
|
-
}
|
|
4586
4454
|
return errors;
|
|
4587
4455
|
}
|
|
4588
|
-
|
|
4589
|
-
|
|
4590
|
-
|
|
4591
|
-
|
|
4592
|
-
* concurrent second `judge_eval_variant` against the same group
|
|
4593
|
-
* loses the race and is rejected with a clean conflict error.
|
|
4594
|
-
*
|
|
4595
|
-
* The seal applies to the SHARED correlation_id of the targets —
|
|
4596
|
-
* NOT to the judge task's own correlationId (which is typically
|
|
4597
|
-
* null or distinct). The task service derives the correlationId
|
|
4598
|
-
* for the effect from the resolved targets, not from the judge
|
|
4599
|
-
* task row.
|
|
4600
|
-
*/
|
|
4601
|
-
async function onCreateJudgeEvalVariant(input, ctx) {
|
|
4602
|
-
const { runTaskIds } = input;
|
|
4603
|
-
const first = await ctx.resolveTask(runTaskIds[0]);
|
|
4604
|
-
if (!first?.correlationId) return [];
|
|
4456
|
+
async function onCreateJudgeEvalAttempt(input, _ctx) {
|
|
4457
|
+
const judge = input;
|
|
4458
|
+
const rubric = judge.successCriteria.rubric;
|
|
4459
|
+
if (!rubric) return [];
|
|
4605
4460
|
return [{
|
|
4606
|
-
kind: "
|
|
4607
|
-
|
|
4461
|
+
kind: "guardTaskUniqueness",
|
|
4462
|
+
taskType: JUDGE_EVAL_ATTEMPT_TYPE,
|
|
4463
|
+
lockKey: [
|
|
4464
|
+
JUDGE_EVAL_ATTEMPT_TYPE,
|
|
4465
|
+
judge.targetTaskId,
|
|
4466
|
+
String(judge.targetAttemptN),
|
|
4467
|
+
rubric.rubricId,
|
|
4468
|
+
rubric.version
|
|
4469
|
+
].join(":"),
|
|
4470
|
+
inputMatches: [
|
|
4471
|
+
{
|
|
4472
|
+
path: ["targetTaskId"],
|
|
4473
|
+
value: judge.targetTaskId
|
|
4474
|
+
},
|
|
4475
|
+
{
|
|
4476
|
+
path: ["targetAttemptN"],
|
|
4477
|
+
value: judge.targetAttemptN
|
|
4478
|
+
},
|
|
4479
|
+
{
|
|
4480
|
+
path: [
|
|
4481
|
+
"successCriteria",
|
|
4482
|
+
"rubric",
|
|
4483
|
+
"rubricId"
|
|
4484
|
+
],
|
|
4485
|
+
value: rubric.rubricId
|
|
4486
|
+
},
|
|
4487
|
+
{
|
|
4488
|
+
path: [
|
|
4489
|
+
"successCriteria",
|
|
4490
|
+
"rubric",
|
|
4491
|
+
"version"
|
|
4492
|
+
],
|
|
4493
|
+
value: rubric.version
|
|
4494
|
+
}
|
|
4495
|
+
]
|
|
4608
4496
|
}];
|
|
4609
4497
|
}
|
|
4610
4498
|
//#endregion
|
|
4499
|
+
//#region ../../libs/tasks/src/task-types/pr-review.ts
|
|
4500
|
+
var PR_REVIEW_TYPE = "pr_review";
|
|
4501
|
+
var PrReviewSubject = Type$2.Object({
|
|
4502
|
+
title: Type$2.String({ minLength: 1 }),
|
|
4503
|
+
summary: Type$2.String({ minLength: 1 }),
|
|
4504
|
+
resourceUrls: Type$2.Optional(Type$2.Array(Type$2.String({ minLength: 1 }))),
|
|
4505
|
+
inspectionHints: Type$2.Optional(Type$2.Array(Type$2.String({ minLength: 1 })))
|
|
4506
|
+
}, {
|
|
4507
|
+
$id: "PrReviewSubject",
|
|
4508
|
+
additionalProperties: false
|
|
4509
|
+
});
|
|
4510
|
+
var PrReviewInput = Type$2.Object({
|
|
4511
|
+
subject: PrReviewSubject,
|
|
4512
|
+
taskPrompt: Type$2.Optional(Type$2.String({ minLength: 1 })),
|
|
4513
|
+
successCriteria: SuccessCriteria
|
|
4514
|
+
}, {
|
|
4515
|
+
$id: "PrReviewInput",
|
|
4516
|
+
additionalProperties: false
|
|
4517
|
+
});
|
|
4518
|
+
var PrReviewScore = Type$2.Object({
|
|
4519
|
+
criterionId: Type$2.String({ minLength: 1 }),
|
|
4520
|
+
score: Type$2.Union([Type$2.Literal(0), Type$2.Literal(1)]),
|
|
4521
|
+
rationale: Type$2.String({ minLength: 1 })
|
|
4522
|
+
}, {
|
|
4523
|
+
$id: "PrReviewScore",
|
|
4524
|
+
additionalProperties: false
|
|
4525
|
+
});
|
|
4526
|
+
var PrReviewOutput = Type$2.Object({
|
|
4527
|
+
scores: Type$2.Array(PrReviewScore, { minItems: 1 }),
|
|
4528
|
+
composite: Type$2.Number({
|
|
4529
|
+
minimum: 0,
|
|
4530
|
+
maximum: 1
|
|
4531
|
+
}),
|
|
4532
|
+
verdict: Type$2.String({ minLength: 1 })
|
|
4533
|
+
}, {
|
|
4534
|
+
$id: "PrReviewOutput",
|
|
4535
|
+
additionalProperties: false
|
|
4536
|
+
});
|
|
4537
|
+
function requireBooleanRubric(rubric) {
|
|
4538
|
+
for (const criterion of rubric.criteria) if (criterion.scoring !== "boolean") return `pr_review requires boolean scoring for every rubric criterion; criterion "${criterion.id}" uses "${criterion.scoring}"`;
|
|
4539
|
+
return null;
|
|
4540
|
+
}
|
|
4541
|
+
function validatePrReviewInput(input) {
|
|
4542
|
+
const sc = input.successCriteria;
|
|
4543
|
+
if (!sc) return "successCriteria is required for judgment tasks";
|
|
4544
|
+
if (!sc.rubric) return "successCriteria.rubric is required for judgment tasks";
|
|
4545
|
+
return validateRubricWeights(sc.rubric) ?? requireBooleanRubric(sc.rubric);
|
|
4546
|
+
}
|
|
4547
|
+
function validatePrReviewOutput(output, input) {
|
|
4548
|
+
if (!input) return null;
|
|
4549
|
+
const scores = output.scores;
|
|
4550
|
+
const rubric = input.successCriteria.rubric;
|
|
4551
|
+
if (!rubric) return null;
|
|
4552
|
+
if (scores.length !== rubric.criteria.length) return `scores length ${scores.length} does not match rubric criteria length ${rubric.criteria.length}`;
|
|
4553
|
+
let composite = 0;
|
|
4554
|
+
for (let i = 0; i < rubric.criteria.length; i++) {
|
|
4555
|
+
const criterion = rubric.criteria[i];
|
|
4556
|
+
const score = scores[i];
|
|
4557
|
+
if (score.criterionId !== criterion.id) return `scores[${i}] has criterionId "${score.criterionId}" but rubric expects "${criterion.id}" in that position`;
|
|
4558
|
+
composite += criterion.weight * score.score;
|
|
4559
|
+
}
|
|
4560
|
+
const claimed = output.composite;
|
|
4561
|
+
if (Math.abs(claimed - composite) > 1e-6) return `composite ${claimed} does not match weighted sum ${composite.toFixed(6)}`;
|
|
4562
|
+
return null;
|
|
4563
|
+
}
|
|
4564
|
+
//#endregion
|
|
4611
4565
|
//#region ../../libs/tasks/src/task-types/render-pack.ts
|
|
4612
4566
|
/**
|
|
4613
4567
|
* `render_pack` — turn a context pack into a signed rendered artefact.
|
|
@@ -4662,14 +4616,43 @@ async function validateRenderPackInputAsync(input, ctx) {
|
|
|
4662
4616
|
//#region ../../libs/tasks/src/task-types/run-eval.ts
|
|
4663
4617
|
/**
|
|
4664
4618
|
* `run_eval` — execute a scenario prompt under a named variant for
|
|
4665
|
-
* later
|
|
4619
|
+
* later per-attempt grading by `judge_eval_attempt` tasks.
|
|
4666
4620
|
*
|
|
4667
4621
|
* output_kind: artifact
|
|
4668
|
-
* criteria: optional (when set,
|
|
4669
|
-
*
|
|
4622
|
+
* criteria: optional producer-only checks (when set,
|
|
4623
|
+
* output.verification is required — the judge rubric remains hidden
|
|
4624
|
+
* on downstream `judge_eval_attempt` tasks)
|
|
4670
4625
|
* references: not required (scenario lives entirely in input)
|
|
4671
4626
|
*/
|
|
4672
4627
|
var RUN_EVAL_TYPE = "run_eval";
|
|
4628
|
+
var RunEvalMode = Type$2.Union([Type$2.Literal("vitro"), Type$2.Literal("vivo")], { $id: "RunEvalMode" });
|
|
4629
|
+
var RunEvalWorkspace = Type$2.Union([
|
|
4630
|
+
Type$2.Literal("none"),
|
|
4631
|
+
Type$2.Literal("shared_mount"),
|
|
4632
|
+
Type$2.Literal("dedicated_worktree")
|
|
4633
|
+
], { $id: "RunEvalWorkspace" });
|
|
4634
|
+
var RunEvalExecution = Type$2.Object({
|
|
4635
|
+
mode: RunEvalMode,
|
|
4636
|
+
workspace: RunEvalWorkspace
|
|
4637
|
+
}, {
|
|
4638
|
+
$id: "RunEvalExecution",
|
|
4639
|
+
additionalProperties: false
|
|
4640
|
+
});
|
|
4641
|
+
/**
|
|
4642
|
+
* Producer-visible checks for `run_eval`. Deliberately forbids `rubric`
|
|
4643
|
+
* so the variant runner cannot see the downstream judge's answer key.
|
|
4644
|
+
* Keep the rest of the SuccessCriteria envelope available for generic
|
|
4645
|
+
* process / structure checks (`gates`, `assertions`, `sideEffects`).
|
|
4646
|
+
*/
|
|
4647
|
+
var RunEvalSuccessCriteria = Type$2.Object({
|
|
4648
|
+
version: Type$2.Literal(1),
|
|
4649
|
+
gates: Type$2.Optional(SuccessCriteria.properties.gates),
|
|
4650
|
+
assertions: Type$2.Optional(SuccessCriteria.properties.assertions),
|
|
4651
|
+
sideEffects: Type$2.Optional(SuccessCriteria.properties.sideEffects)
|
|
4652
|
+
}, {
|
|
4653
|
+
$id: "RunEvalSuccessCriteria",
|
|
4654
|
+
additionalProperties: false
|
|
4655
|
+
});
|
|
4673
4656
|
var RunEvalInput = Type$2.Object({
|
|
4674
4657
|
scenario: Type$2.Object({
|
|
4675
4658
|
prompt: Type$2.String({ minLength: 1 }),
|
|
@@ -4679,8 +4662,9 @@ var RunEvalInput = Type$2.Object({
|
|
|
4679
4662
|
minLength: 1,
|
|
4680
4663
|
maxLength: 64
|
|
4681
4664
|
}),
|
|
4665
|
+
execution: RunEvalExecution,
|
|
4682
4666
|
context: TaskContext,
|
|
4683
|
-
successCriteria: Type$2.Optional(
|
|
4667
|
+
successCriteria: Type$2.Optional(RunEvalSuccessCriteria)
|
|
4684
4668
|
}, {
|
|
4685
4669
|
$id: "RunEvalInput",
|
|
4686
4670
|
additionalProperties: false
|
|
@@ -4708,8 +4692,8 @@ var RunEvalOutput = Type$2.Object({
|
|
|
4708
4692
|
function validateRunEvalOutput(output, input) {
|
|
4709
4693
|
const hasCriteria = input !== null && input !== void 0 && input.successCriteria !== void 0;
|
|
4710
4694
|
const hasVerification = output !== null && output !== void 0 && output.verification !== void 0;
|
|
4711
|
-
if (hasCriteria && !hasVerification) return "output.verification is required because input.successCriteria is set; the producer LLM must self-assess against the
|
|
4712
|
-
if (!hasCriteria && hasVerification) return "output.verification was supplied but input.successCriteria is unset; omit verification when there are no
|
|
4695
|
+
if (hasCriteria && !hasVerification) return "output.verification is required because input.successCriteria is set; the producer LLM must self-assess against the producer checks";
|
|
4696
|
+
if (!hasCriteria && hasVerification) return "output.verification was supplied but input.successCriteria is unset; omit verification when there are no producer checks to assess against";
|
|
4713
4697
|
return null;
|
|
4714
4698
|
}
|
|
4715
4699
|
//#endregion
|
|
@@ -4775,6 +4759,18 @@ var BUILT_IN_TASK_TYPES = {
|
|
|
4775
4759
|
validateInput: validateJudgmentInput,
|
|
4776
4760
|
validateInputAsync: validateAssessBriefInputAsync
|
|
4777
4761
|
},
|
|
4762
|
+
[PR_REVIEW_TYPE]: {
|
|
4763
|
+
name: PR_REVIEW_TYPE,
|
|
4764
|
+
inputSchema: PrReviewInput,
|
|
4765
|
+
outputSchema: PrReviewOutput,
|
|
4766
|
+
outputKind: "judgment",
|
|
4767
|
+
workspaceMode: "dedicated_worktree",
|
|
4768
|
+
workspaceScope: "attempt",
|
|
4769
|
+
sessionScope: "none",
|
|
4770
|
+
requiresReferences: false,
|
|
4771
|
+
validateInput: validatePrReviewInput,
|
|
4772
|
+
validateOutput: validatePrReviewOutput
|
|
4773
|
+
},
|
|
4778
4774
|
[CURATE_PACK_TYPE]: {
|
|
4779
4775
|
name: CURATE_PACK_TYPE,
|
|
4780
4776
|
inputSchema: CuratePackInput,
|
|
@@ -4813,24 +4809,24 @@ var BUILT_IN_TASK_TYPES = {
|
|
|
4813
4809
|
inputSchema: RunEvalInput,
|
|
4814
4810
|
outputSchema: RunEvalOutput,
|
|
4815
4811
|
outputKind: "artifact",
|
|
4816
|
-
|
|
4812
|
+
resumable: true,
|
|
4813
|
+
workspaceScope: "session",
|
|
4817
4814
|
sessionScope: "custom",
|
|
4818
4815
|
requiresReferences: false,
|
|
4819
4816
|
validateOutput: validateRunEvalOutput
|
|
4820
4817
|
},
|
|
4821
|
-
[
|
|
4822
|
-
name:
|
|
4823
|
-
inputSchema:
|
|
4824
|
-
outputSchema:
|
|
4818
|
+
[JUDGE_EVAL_ATTEMPT_TYPE]: {
|
|
4819
|
+
name: JUDGE_EVAL_ATTEMPT_TYPE,
|
|
4820
|
+
inputSchema: JudgeEvalAttemptInput,
|
|
4821
|
+
outputSchema: JudgeEvalAttemptOutput,
|
|
4825
4822
|
outputKind: "judgment",
|
|
4826
4823
|
workspaceScope: "attempt",
|
|
4827
|
-
sessionScope: "
|
|
4824
|
+
sessionScope: "none",
|
|
4828
4825
|
requiresReferences: false,
|
|
4829
|
-
validateInput:
|
|
4830
|
-
validateOutput:
|
|
4831
|
-
validateInputAsync:
|
|
4832
|
-
onCreate:
|
|
4833
|
-
usesSubagents: true
|
|
4826
|
+
validateInput: validateJudgeEvalAttemptInput,
|
|
4827
|
+
validateOutput: validateJudgeEvalAttemptOutput,
|
|
4828
|
+
validateInputAsync: validateJudgeEvalAttemptInputAsync,
|
|
4829
|
+
onCreate: onCreateJudgeEvalAttempt
|
|
4834
4830
|
}
|
|
4835
4831
|
};
|
|
4836
4832
|
//#endregion
|
|
@@ -6180,6 +6176,11 @@ var PROMPT_SEPARATOR = "\n\n---\n\n";
|
|
|
6180
6176
|
* - `skill` → `deliver.skill({ slug, content })` once per ref.
|
|
6181
6177
|
* Slug collisions on distinct contents are
|
|
6182
6178
|
* refused loudly.
|
|
6179
|
+
* - `context_inline`→ persist raw bytes via `deliver.contextFile(...)`
|
|
6180
|
+
* and inject them into the prompt in an explicit,
|
|
6181
|
+
* named block. Intended for eval/context experiments
|
|
6182
|
+
* where the content must be in the model context
|
|
6183
|
+
* window, not merely discoverable as a skill.
|
|
6183
6184
|
* - `prompt_prefix` → content appended to `systemPromptPrefix` with
|
|
6184
6185
|
* the canonical `\n\n---\n\n` separator (in
|
|
6185
6186
|
* declared order).
|
|
@@ -6212,6 +6213,13 @@ async function resolveTaskContext(args) {
|
|
|
6212
6213
|
slug: ref.slug,
|
|
6213
6214
|
content: ref.content
|
|
6214
6215
|
});
|
|
6216
|
+
} else if (ref.binding === "context_inline") {
|
|
6217
|
+
await args.deliver.contextFile({
|
|
6218
|
+
slug: ref.slug,
|
|
6219
|
+
content: ref.content,
|
|
6220
|
+
suggestedFileName: `${ref.slug}.md`
|
|
6221
|
+
});
|
|
6222
|
+
promptParts.push(formatInlineContextBlock(ref.slug, ref.content));
|
|
6215
6223
|
} else if (ref.binding === "prompt_prefix") promptParts.push(ref.content);
|
|
6216
6224
|
else userParts.push(ref.content);
|
|
6217
6225
|
injected.push(ref);
|
|
@@ -6222,6 +6230,23 @@ async function resolveTaskContext(args) {
|
|
|
6222
6230
|
userInlineSuffix: userParts.join(PROMPT_SEPARATOR)
|
|
6223
6231
|
};
|
|
6224
6232
|
}
|
|
6233
|
+
function formatInlineContextBlock(slug, content) {
|
|
6234
|
+
return [
|
|
6235
|
+
"### Injected Task Context",
|
|
6236
|
+
"",
|
|
6237
|
+
`Context id: \`${slug}\``,
|
|
6238
|
+
"The following raw context was supplied by the task creator. Treat it",
|
|
6239
|
+
"as task-relevant background that may override generic coding instincts",
|
|
6240
|
+
"when it contains repo- or workflow-specific constraints.",
|
|
6241
|
+
"The same content is also materialized in the workspace as",
|
|
6242
|
+
"`/workspace/context-pack.md` and mirrored in `AGENTS.md` for",
|
|
6243
|
+
"repo-context discovery.",
|
|
6244
|
+
"",
|
|
6245
|
+
"<context>",
|
|
6246
|
+
content,
|
|
6247
|
+
"</context>"
|
|
6248
|
+
].join("\n");
|
|
6249
|
+
}
|
|
6225
6250
|
//#endregion
|
|
6226
6251
|
//#region ../../libs/agent-runtime/src/output-tools.ts
|
|
6227
6252
|
/**
|
|
@@ -6287,20 +6312,16 @@ function buildFinalOutputBlock(opts) {
|
|
|
6287
6312
|
"## Final output (read this carefully)",
|
|
6288
6313
|
"",
|
|
6289
6314
|
`Your VERY LAST action in this conversation MUST report the structured`,
|
|
6290
|
-
`output matching \`${outputSchemaName}
|
|
6291
|
-
`preference:`,
|
|
6315
|
+
`output matching \`${outputSchemaName}\`.`,
|
|
6292
6316
|
"",
|
|
6293
|
-
`
|
|
6294
|
-
`
|
|
6295
|
-
`
|
|
6296
|
-
`
|
|
6297
|
-
`
|
|
6298
|
-
` \`${outputSchemaName}\`. No prose before or after. No code fences.`,
|
|
6299
|
-
` No "ok" or "done". The runtime parses the last balanced top-level`,
|
|
6300
|
-
` JSON object as the output.`,
|
|
6317
|
+
`Call \`${submitTool}\` exactly once with the payload.`,
|
|
6318
|
+
`The runtime captures the validated arguments and ends the session.`,
|
|
6319
|
+
`Do NOT emit the output as plain assistant text. Do NOT rely on a`,
|
|
6320
|
+
`JSON-in-message fallback. If you do not call \`${submitTool}\`, the`,
|
|
6321
|
+
`attempt fails even if the underlying work succeeded.`,
|
|
6301
6322
|
"",
|
|
6302
|
-
`
|
|
6303
|
-
`
|
|
6323
|
+
`Your final assistant text before that tool call may explain your work,`,
|
|
6324
|
+
`but the submit-tool call itself must be your VERY LAST action.`,
|
|
6304
6325
|
"",
|
|
6305
6326
|
`Output shape:`,
|
|
6306
6327
|
"",
|
|
@@ -6315,6 +6336,20 @@ function buildFinalOutputBlock(opts) {
|
|
|
6315
6336
|
return lines.join("\n");
|
|
6316
6337
|
}
|
|
6317
6338
|
//#endregion
|
|
6339
|
+
//#region ../../libs/agent-runtime/src/prompts/rubric-common.ts
|
|
6340
|
+
function renderRubricCriteriaList(rubric) {
|
|
6341
|
+
return rubric.criteria.map((c, i) => `${i + 1}. **${c.id}** (weight ${c.weight}, scoring: \`${c.scoring}\`) — ${c.description}`).join("\n");
|
|
6342
|
+
}
|
|
6343
|
+
function renderRubricPreambleSection(rubric) {
|
|
6344
|
+
if (!rubric.preamble) return null;
|
|
6345
|
+
return [
|
|
6346
|
+
"### Rubric preamble",
|
|
6347
|
+
"",
|
|
6348
|
+
rubric.preamble,
|
|
6349
|
+
""
|
|
6350
|
+
].join("\n");
|
|
6351
|
+
}
|
|
6352
|
+
//#endregion
|
|
6318
6353
|
//#region ../../libs/agent-runtime/src/prompts/assess-brief.ts
|
|
6319
6354
|
/**
|
|
6320
6355
|
* Build the first user-message prompt for an `assess_brief` judge attempt.
|
|
@@ -6340,13 +6375,8 @@ function buildFinalOutputBlock(opts) {
|
|
|
6340
6375
|
*/
|
|
6341
6376
|
function buildAssessBriefUserPrompt(input, ctx) {
|
|
6342
6377
|
const rubric = input.successCriteria.rubric;
|
|
6343
|
-
const criteriaList = rubric
|
|
6344
|
-
const preambleSection = rubric
|
|
6345
|
-
"### Rubric preamble",
|
|
6346
|
-
"",
|
|
6347
|
-
rubric.preamble,
|
|
6348
|
-
""
|
|
6349
|
-
].join("\n") : "";
|
|
6378
|
+
const criteriaList = renderRubricCriteriaList(rubric);
|
|
6379
|
+
const preambleSection = renderRubricPreambleSection(rubric) ?? "";
|
|
6350
6380
|
const workspaceSection = ctx.workspace?.mode === "dedicated_worktree" ? [
|
|
6351
6381
|
"### Workspace",
|
|
6352
6382
|
"",
|
|
@@ -6429,21 +6459,30 @@ function buildAssessBriefUserPrompt(input, ctx) {
|
|
|
6429
6459
|
}
|
|
6430
6460
|
//#endregion
|
|
6431
6461
|
//#region ../../libs/agent-runtime/src/prompts/self-verification.ts
|
|
6432
|
-
function buildSelfVerificationBlock(taskId) {
|
|
6462
|
+
function buildSelfVerificationBlock(taskId, criteriaField = "successCriteria") {
|
|
6433
6463
|
return [
|
|
6434
6464
|
"## Self-verification",
|
|
6435
6465
|
"",
|
|
6436
|
-
`
|
|
6466
|
+
`If \`input.${criteriaField}\` is set on this task, your final output MUST`,
|
|
6467
|
+
"include a `verification` block. **The runtime/server rejects task",
|
|
6468
|
+
`submission without \`verification\` when \`${criteriaField}\` is present**`,
|
|
6469
|
+
"— the request fails validation and the attempt is discarded, even if the",
|
|
6470
|
+
"underlying work succeeded. Do not call the submit tool until you have",
|
|
6471
|
+
"computed the verification payload.",
|
|
6472
|
+
"",
|
|
6473
|
+
`Call \`moltnet_get_task\` with task id \`${taskId}\` and read \`input.${criteriaField}\`.`,
|
|
6437
6474
|
"",
|
|
6438
|
-
|
|
6475
|
+
`- If \`input.${criteriaField}\` is **absent**, omit \`verification\` from your`,
|
|
6439
6476
|
" final output entirely.",
|
|
6440
|
-
|
|
6441
|
-
" `verification` block in your final output. Evaluate every applicable",
|
|
6477
|
+
`- If \`input.${criteriaField}\` is **present**, evaluate every applicable`,
|
|
6442
6478
|
" item — `gates`, `assertions`, `rubric` criteria, `sideEffects` — against",
|
|
6443
6479
|
" your produced work and emit one result per id. Be honest: a `fail` with",
|
|
6444
6480
|
" a one-line reason is more useful than a false `pass`. Use `skip` (with a",
|
|
6445
6481
|
" `detail`) when you genuinely could not determine a result. Compute",
|
|
6446
6482
|
" `passed = results.every(r => r.status !== 'fail')`.",
|
|
6483
|
+
"- `verification` MUST be a JSON object. Never send a string, markdown",
|
|
6484
|
+
" block, null, or an empty placeholder. The submit tool expects an object",
|
|
6485
|
+
" with `inputCid`, `results`, and `passed` fields.",
|
|
6447
6486
|
"",
|
|
6448
6487
|
"Verification shape:",
|
|
6449
6488
|
"",
|
|
@@ -6457,6 +6496,23 @@ function buildSelfVerificationBlock(taskId) {
|
|
|
6457
6496
|
" \"passed\": <boolean>",
|
|
6458
6497
|
"}",
|
|
6459
6498
|
"```",
|
|
6499
|
+
"",
|
|
6500
|
+
"Minimal valid example:",
|
|
6501
|
+
"",
|
|
6502
|
+
"```json",
|
|
6503
|
+
"{",
|
|
6504
|
+
" \"inputCid\": \"<task inputCid>\",",
|
|
6505
|
+
" \"results\": [",
|
|
6506
|
+
" {",
|
|
6507
|
+
" \"id\": \"<criterion id>\",",
|
|
6508
|
+
" \"kind\": \"rubric\",",
|
|
6509
|
+
" \"status\": \"pass\",",
|
|
6510
|
+
" \"detail\": \"one-line reason\"",
|
|
6511
|
+
" }",
|
|
6512
|
+
" ],",
|
|
6513
|
+
" \"passed\": true",
|
|
6514
|
+
"}",
|
|
6515
|
+
"```",
|
|
6460
6516
|
""
|
|
6461
6517
|
].join("\n");
|
|
6462
6518
|
}
|
|
@@ -6707,69 +6763,62 @@ function buildFulfillBriefUserPrompt(input, ctx) {
|
|
|
6707
6763
|
].filter(Boolean).join("\n");
|
|
6708
6764
|
}
|
|
6709
6765
|
//#endregion
|
|
6710
|
-
//#region ../../libs/agent-runtime/src/prompts/judge-eval-
|
|
6711
|
-
|
|
6712
|
-
|
|
6713
|
-
|
|
6714
|
-
*
|
|
6715
|
-
* The parent agent's job is **fan-out-and-collect**: for each
|
|
6716
|
-
* `runTaskIds[i]`, spawn an isolated subagent via the `subagent` custom
|
|
6717
|
-
* tool (#1087), have it grade that variant against the shared rubric,
|
|
6718
|
-
* and collect each subagent's structured `judge_eval_variant_result`
|
|
6719
|
-
* payload. The parent does NOT grade itself; it composes the per-
|
|
6720
|
-
* variant results into the final `judge_eval_variant` output (results
|
|
6721
|
-
* array + optional deltas + verdicts).
|
|
6722
|
-
*
|
|
6723
|
-
* Isolation is the point: each variant gets a fresh subagent session
|
|
6724
|
-
* with no carryover context from sibling variants, so per-variant
|
|
6725
|
-
* grading is independent. Cost is bounded by `maxItems: 10` on
|
|
6726
|
-
* runTaskIds.
|
|
6727
|
-
*/
|
|
6728
|
-
function buildJudgeEvalVariantUserPrompt(input, ctx) {
|
|
6729
|
-
const { runTaskIds, successCriteria } = input;
|
|
6730
|
-
const rubric = successCriteria.rubric;
|
|
6731
|
-
if (!rubric) throw new Error("judge_eval_variant requires successCriteria.rubric — none present");
|
|
6766
|
+
//#region ../../libs/agent-runtime/src/prompts/judge-eval-attempt.ts
|
|
6767
|
+
function buildJudgeEvalAttemptUserPrompt(input, ctx) {
|
|
6768
|
+
const rubric = input.successCriteria.rubric;
|
|
6769
|
+
if (!rubric) throw new Error("judge_eval_attempt requires successCriteria.rubric — none present");
|
|
6732
6770
|
const escapeCell = (s) => s.replace(/\\/g, "\\\\").replace(/\|/g, "\\|").replace(/\r?\n/g, " ");
|
|
6733
6771
|
const criteriaTable = rubric.criteria.map((c) => `| \`${c.id}\` | ${c.weight.toFixed(3)} | ${c.scoring} | ${escapeCell(c.description)} |`).join("\n");
|
|
6734
|
-
const targetsBlock = runTaskIds.map((id, i) => `${i + 1}. \`${id}\``).join("\n");
|
|
6735
6772
|
const finalOutputBlock = buildFinalOutputBlock({
|
|
6736
|
-
taskType: "
|
|
6737
|
-
outputSchemaName: "
|
|
6773
|
+
taskType: "judge_eval_attempt",
|
|
6774
|
+
outputSchemaName: "JudgeEvalAttemptOutput",
|
|
6738
6775
|
shapeSketch: [
|
|
6739
6776
|
"{",
|
|
6740
|
-
|
|
6741
|
-
"
|
|
6742
|
-
"
|
|
6743
|
-
"
|
|
6744
|
-
"
|
|
6745
|
-
"
|
|
6746
|
-
" \"verdict\": \"<1-3 sentences>\"",
|
|
6747
|
-
" },",
|
|
6748
|
-
" ...one entry per runTaskIds[i], same order",
|
|
6749
|
-
" ],",
|
|
6750
|
-
" \"deltas\": { \"<labelA> - <labelB>\": <composite(A) - composite(B)> }, // optional",
|
|
6777
|
+
` "targetTaskId": "${input.targetTaskId}",`,
|
|
6778
|
+
` "targetAttemptN": ${input.targetAttemptN},`,
|
|
6779
|
+
" \"variantLabel\": \"<from producer input>\",",
|
|
6780
|
+
" \"scores\": [ { \"criterionId\": \"...\", \"score\": 0..1, \"rationale\": \"...\", \"assertions\": [...]? } ],",
|
|
6781
|
+
" \"composite\": <Σ(weight × score), 0..1>,",
|
|
6782
|
+
" \"verdict\": \"<1-3 sentences>\",",
|
|
6751
6783
|
" \"judgeModel\": \"<id>\", // optional",
|
|
6752
6784
|
" \"traceparent\": \"<from claim>\"",
|
|
6753
6785
|
"}"
|
|
6754
6786
|
].join("\n")
|
|
6755
6787
|
});
|
|
6788
|
+
const workspaceSection = ctx.workspace?.attached === true ? [
|
|
6789
|
+
"### Workspace",
|
|
6790
|
+
"",
|
|
6791
|
+
"Your current workspace is already attached to the producer attempt",
|
|
6792
|
+
"you are judging. Inspect files directly from the current workspace",
|
|
6793
|
+
"root instead of inventing synthetic `artifact_<taskId>` paths.",
|
|
6794
|
+
"If the accepted attempt output lists `artifacts[].path`, treat those",
|
|
6795
|
+
"paths as relative to the current workspace root unless the output",
|
|
6796
|
+
"explicitly says otherwise.",
|
|
6797
|
+
ctx.workspace.mode === "dedicated_worktree" ? `This attachment is a dedicated producer worktree${ctx.workspace.branch ? ` on branch \`${ctx.workspace.branch}\`` : ""}.` : ctx.workspace.mode === "scratch_mount" ? "This attachment is the producer scratch workspace mounted with shadow writes for safe inspection." : "This attachment is the producer shared workspace mounted with shadow writes for safe inspection.",
|
|
6798
|
+
""
|
|
6799
|
+
].join("\n") : "";
|
|
6756
6800
|
return [
|
|
6757
|
-
"# Judge Eval
|
|
6758
|
-
|
|
6759
|
-
"
|
|
6760
|
-
"grade yourself.",
|
|
6801
|
+
"# Judge Eval Attempt\n",
|
|
6802
|
+
"You are grading one accepted `run_eval` producer attempt against a hidden",
|
|
6803
|
+
"judge rubric. Do not delegate to subagents. Grade in this session only.",
|
|
6761
6804
|
"",
|
|
6762
6805
|
`Task id: \`${ctx.taskId}\``,
|
|
6763
6806
|
`Diary: \`${ctx.diaryId}\``,
|
|
6807
|
+
`Producer task: \`${input.targetTaskId}\``,
|
|
6808
|
+
`Producer attempt: \`${input.targetAttemptN}\``,
|
|
6764
6809
|
"",
|
|
6765
|
-
"###
|
|
6766
|
-
"",
|
|
6767
|
-
targetsBlock,
|
|
6810
|
+
"### Evidence gathering",
|
|
6768
6811
|
"",
|
|
6769
|
-
|
|
6770
|
-
|
|
6771
|
-
|
|
6812
|
+
`1. Call \`moltnet_get_task\` with taskId=\`${input.targetTaskId}\`.`,
|
|
6813
|
+
`2. Call \`moltnet_list_task_attempts\` with taskId=\`${input.targetTaskId}\` and inspect the accepted attempt matching \`${input.targetAttemptN}\`.`,
|
|
6814
|
+
`3. Call \`moltnet_list_task_messages\` with taskId=\`${input.targetTaskId}\`, attemptN=\`${input.targetAttemptN}\` to inspect the producer's turn-by-turn behavior.`,
|
|
6815
|
+
"4. Use the accepted attempt output, attempt messages, and any accessible",
|
|
6816
|
+
" artifacts or workspace evidence available in your environment.",
|
|
6817
|
+
" Read artifact files from the mounted producer workspace when present;",
|
|
6818
|
+
" do not assume detached `artifact_<taskId>` directories exist.",
|
|
6819
|
+
"5. Score strictly against the rubric below.",
|
|
6772
6820
|
"",
|
|
6821
|
+
workspaceSection,
|
|
6773
6822
|
"### Rubric",
|
|
6774
6823
|
"",
|
|
6775
6824
|
rubric.preamble ? `${rubric.preamble}\n` : "",
|
|
@@ -6777,34 +6826,10 @@ function buildJudgeEvalVariantUserPrompt(input, ctx) {
|
|
|
6777
6826
|
"| --- | --- | --- | --- |",
|
|
6778
6827
|
criteriaTable,
|
|
6779
6828
|
"",
|
|
6780
|
-
"### How to grade",
|
|
6781
|
-
"",
|
|
6782
|
-
"For EACH `runTaskIds[i]`:",
|
|
6783
|
-
"",
|
|
6784
|
-
"1. Call the `subagent` custom tool with:",
|
|
6785
|
-
" - `task`: a brief instructing the subagent to grade ONLY that variant",
|
|
6786
|
-
" against the rubric above; include the target task id and the rubric",
|
|
6787
|
-
" verbatim. The subagent has the same MoltNet tools and can fetch the",
|
|
6788
|
-
" accepted attempt output independently.",
|
|
6789
|
-
" - `output_schema`: `\"judge_eval_variant_result\"`",
|
|
6790
|
-
"2. Receive the subagent's structured `judge_eval_variant_result` payload.",
|
|
6791
|
-
"3. Append it to your `results[]` array, **in the same order as input.runTaskIds**.",
|
|
6792
|
-
"",
|
|
6793
|
-
"Do NOT score any variant in your own session. The whole point of the",
|
|
6794
|
-
"subagent fan-out is per-variant context isolation — grading two variants",
|
|
6795
|
-
"back-to-back in one session lets the second be biased by the first.",
|
|
6796
|
-
"",
|
|
6797
6829
|
"### Composite arithmetic",
|
|
6798
6830
|
"",
|
|
6799
|
-
"
|
|
6800
|
-
"criteria. Drift > 0.001 is rejected.
|
|
6801
|
-
"themselves; double-check before assembling the final output.",
|
|
6802
|
-
"",
|
|
6803
|
-
"### Deltas (optional)",
|
|
6804
|
-
"",
|
|
6805
|
-
"If useful, populate `deltas` with pairwise composite differences keyed by",
|
|
6806
|
-
"`\"<variantLabel-A> - <variantLabel-B>\"` (single space-hyphen-space). Both",
|
|
6807
|
-
"labels must appear in `results`. Omit `deltas` entirely if not used.",
|
|
6831
|
+
"Your `composite` MUST equal `Σ(criterion.weight × score)` over the rubric",
|
|
6832
|
+
"criteria. Drift > 0.001 is rejected.",
|
|
6808
6833
|
"",
|
|
6809
6834
|
finalOutputBlock
|
|
6810
6835
|
].filter((s) => s !== "").join("\n");
|
|
@@ -6814,13 +6839,8 @@ function buildJudgeEvalVariantUserPrompt(input, ctx) {
|
|
|
6814
6839
|
function buildJudgePackUserPrompt(input, ctx) {
|
|
6815
6840
|
const { renderedPackId, sourcePackId, successCriteria } = input;
|
|
6816
6841
|
const rubric = successCriteria.rubric;
|
|
6817
|
-
const criteriaList = rubric
|
|
6818
|
-
const preambleSection = rubric
|
|
6819
|
-
"### Rubric preamble",
|
|
6820
|
-
"",
|
|
6821
|
-
rubric.preamble,
|
|
6822
|
-
""
|
|
6823
|
-
].join("\n") : null;
|
|
6842
|
+
const criteriaList = renderRubricCriteriaList(rubric);
|
|
6843
|
+
const preambleSection = renderRubricPreambleSection(rubric);
|
|
6824
6844
|
return [
|
|
6825
6845
|
"# Judge Pack Agent",
|
|
6826
6846
|
"",
|
|
@@ -6936,6 +6956,112 @@ function buildJudgePackUserPrompt(input, ctx) {
|
|
|
6936
6956
|
].filter((l) => l !== null).join("\n");
|
|
6937
6957
|
}
|
|
6938
6958
|
//#endregion
|
|
6959
|
+
//#region ../../libs/agent-runtime/src/prompts/pr-review.ts
|
|
6960
|
+
function buildPrReviewUserPrompt(input, ctx) {
|
|
6961
|
+
const rubric = input.successCriteria.rubric;
|
|
6962
|
+
const criteriaList = renderRubricCriteriaList(rubric);
|
|
6963
|
+
const preambleSection = renderRubricPreambleSection(rubric);
|
|
6964
|
+
const taskPromptSection = input.taskPrompt ? [
|
|
6965
|
+
"## Task-specific instructions",
|
|
6966
|
+
"",
|
|
6967
|
+
input.taskPrompt,
|
|
6968
|
+
""
|
|
6969
|
+
].join("\n") : "";
|
|
6970
|
+
const resourceSection = input.subject.resourceUrls && input.subject.resourceUrls.length > 0 ? [
|
|
6971
|
+
"### Resources",
|
|
6972
|
+
"",
|
|
6973
|
+
...input.subject.resourceUrls.map((url) => `- ${url}`),
|
|
6974
|
+
""
|
|
6975
|
+
].join("\n") : "";
|
|
6976
|
+
const hintsSection = input.subject.inspectionHints && input.subject.inspectionHints.length > 0 ? [
|
|
6977
|
+
"### Inspection hints",
|
|
6978
|
+
"",
|
|
6979
|
+
...input.subject.inspectionHints.map((hint) => `- ${hint}`),
|
|
6980
|
+
""
|
|
6981
|
+
].join("\n") : "";
|
|
6982
|
+
const workspaceSection = ctx.workspace?.mode === "dedicated_worktree" ? [
|
|
6983
|
+
"### Workspace",
|
|
6984
|
+
"",
|
|
6985
|
+
"This review attempt is running inside a dedicated disposable git",
|
|
6986
|
+
"worktree. Inspect and reason inside this workspace only.",
|
|
6987
|
+
ctx.workspace.branch ? `The current review branch is \`${ctx.workspace.branch}\`.` : "The current checkout is disposable and will be cleaned up when the task ends.",
|
|
6988
|
+
""
|
|
6989
|
+
].join("\n") : "";
|
|
6990
|
+
return [
|
|
6991
|
+
"# Review Agent",
|
|
6992
|
+
"",
|
|
6993
|
+
"You are an independent judge. You did NOT produce the subject under review.",
|
|
6994
|
+
"Assess it strictly against the rubric below and emit a structured judgment.",
|
|
6995
|
+
"You may inspect the local workspace and the referenced resources, but do NOT modify anything.",
|
|
6996
|
+
"",
|
|
6997
|
+
`Your diary ID is: ${ctx.diaryId}`,
|
|
6998
|
+
`This task's id is: ${ctx.taskId}`,
|
|
6999
|
+
"",
|
|
7000
|
+
"## Subject",
|
|
7001
|
+
"",
|
|
7002
|
+
`**Title:** ${input.subject.title}`,
|
|
7003
|
+
"",
|
|
7004
|
+
input.subject.summary,
|
|
7005
|
+
"",
|
|
7006
|
+
resourceSection,
|
|
7007
|
+
hintsSection,
|
|
7008
|
+
workspaceSection,
|
|
7009
|
+
"### Execution contract",
|
|
7010
|
+
"",
|
|
7011
|
+
"Treat the provided subject, resources, inspection hints, and any",
|
|
7012
|
+
"task-specific instructions as the full",
|
|
7013
|
+
"review contract for this task.",
|
|
7014
|
+
"",
|
|
7015
|
+
"If the task-specific instructions or inspection hints require an outward action tied to the review",
|
|
7016
|
+
"(for example publishing the judgment somewhere), perform that action as",
|
|
7017
|
+
"part of the task before reporting structured output.",
|
|
7018
|
+
"",
|
|
7019
|
+
"## Review workflow",
|
|
7020
|
+
"",
|
|
7021
|
+
"1. Read the subject summary, resources, inspection hints, and any",
|
|
7022
|
+
" task-specific instructions before scoring.",
|
|
7023
|
+
"2. Inspect the target artefact directly using the tools and resources the",
|
|
7024
|
+
" task makes available.",
|
|
7025
|
+
"3. If you are in a dedicated disposable worktree and need the review target",
|
|
7026
|
+
" checked out locally, do that work inside this disposable workspace only.",
|
|
7027
|
+
"4. Apply the rubric strictly. This task is about complexity and",
|
|
7028
|
+
" reviewability, not correctness or feature desirability.",
|
|
7029
|
+
"5. Perform any required outward action before emitting the final",
|
|
7030
|
+
" structured output.",
|
|
7031
|
+
"",
|
|
7032
|
+
taskPromptSection,
|
|
7033
|
+
preambleSection,
|
|
7034
|
+
"## Criteria",
|
|
7035
|
+
"",
|
|
7036
|
+
criteriaList,
|
|
7037
|
+
"",
|
|
7038
|
+
"### Scoring rules",
|
|
7039
|
+
"",
|
|
7040
|
+
"- Every criterion uses binary scoring only.",
|
|
7041
|
+
"- Score `1` when the subject clearly clears the criterion.",
|
|
7042
|
+
"- Score `0` when it does not, or when the evidence is ambiguous.",
|
|
7043
|
+
"- `rationale` is REQUIRED for every score. Keep it concrete and audit-friendly.",
|
|
7044
|
+
"- Compute `composite = Σ(weight_i × score_i)` exactly; the runtime rejects mismatches.",
|
|
7045
|
+
"",
|
|
7046
|
+
"Write a signed diary entry (tags: `judgment`, `pr_review`) capturing the rationale before reporting structured output.",
|
|
7047
|
+
"",
|
|
7048
|
+
buildFinalOutputBlock({
|
|
7049
|
+
taskType: "pr_review",
|
|
7050
|
+
outputSchemaName: "PrReviewOutput",
|
|
7051
|
+
shapeSketch: [
|
|
7052
|
+
"{",
|
|
7053
|
+
" \"scores\": [",
|
|
7054
|
+
" { \"criterionId\": \"...\", \"score\": 0, \"rationale\": \"...\" }",
|
|
7055
|
+
" ],",
|
|
7056
|
+
" \"composite\": <sum-of-weighted-binary-scores>,",
|
|
7057
|
+
" \"verdict\": \"<1-3 sentence overall>\"",
|
|
7058
|
+
"}"
|
|
7059
|
+
].join("\n"),
|
|
7060
|
+
extraNotes: ["`scores` MUST stay in the same order as the rubric criteria.", "`score` MUST be exactly `0` or `1` for every criterion."]
|
|
7061
|
+
})
|
|
7062
|
+
].filter(Boolean).join("\n");
|
|
7063
|
+
}
|
|
7064
|
+
//#endregion
|
|
6939
7065
|
//#region ../../libs/agent-runtime/src/prompts/render-pack.ts
|
|
6940
7066
|
/**
|
|
6941
7067
|
* Build the first user-message prompt for a `render_pack` task. Almost mechanical:
|
|
@@ -7000,8 +7126,9 @@ function buildRenderPackUserPrompt(input, ctx) {
|
|
|
7000
7126
|
* Build the first user-message prompt for a `run_eval` task.
|
|
7001
7127
|
*
|
|
7002
7128
|
* Free-form: no git workflow, no commit ceremony. The executor produces
|
|
7003
|
-
* a textual response (and optional file artifacts) that
|
|
7004
|
-
* `
|
|
7129
|
+
* a textual response (and optional file artifacts) that later
|
|
7130
|
+
* `judge_eval_attempt` task(s) grade against their own hidden
|
|
7131
|
+
* rubric.
|
|
7005
7132
|
*
|
|
7006
7133
|
* Context delivery is handled by `resolveTaskContext` (see
|
|
7007
7134
|
* libs/agent-runtime/src/context-bindings.ts) and runs BEFORE this
|
|
@@ -7011,7 +7138,9 @@ function buildRenderPackUserPrompt(input, ctx) {
|
|
|
7011
7138
|
* builder does NOT inline `input.context[]` itself.
|
|
7012
7139
|
*/
|
|
7013
7140
|
function buildRunEvalUserPrompt(input, ctx) {
|
|
7014
|
-
const { scenario, variantLabel, successCriteria } = input;
|
|
7141
|
+
const { scenario, variantLabel, execution, successCriteria } = input;
|
|
7142
|
+
const hasContext = input.context.length > 0;
|
|
7143
|
+
const hasInlineContext = input.context.some((entry) => entry.binding === "context_inline");
|
|
7015
7144
|
const inputFilesSection = scenario.inputFiles?.length ? [
|
|
7016
7145
|
"### Input files",
|
|
7017
7146
|
"",
|
|
@@ -7024,9 +7153,30 @@ function buildRunEvalUserPrompt(input, ctx) {
|
|
|
7024
7153
|
"",
|
|
7025
7154
|
`This task carries correlationId \`${ctx.correlationId}\`. It joins`,
|
|
7026
7155
|
"this variant to its sibling `run_eval` tasks (other variants of the",
|
|
7027
|
-
"same scenario
|
|
7028
|
-
"
|
|
7029
|
-
"
|
|
7156
|
+
"same scenario and to any later `judge_eval_attempt` tasks created",
|
|
7157
|
+
"against those variants. You do not need to act on it directly — it",
|
|
7158
|
+
"is recorded for cross-variant aggregation at query time.",
|
|
7159
|
+
""
|
|
7160
|
+
].join("\n") : "";
|
|
7161
|
+
const executionSection = [
|
|
7162
|
+
"### Execution mode",
|
|
7163
|
+
"",
|
|
7164
|
+
`Mode: \`${execution.mode}\``,
|
|
7165
|
+
`Workspace: \`${execution.workspace}\``,
|
|
7166
|
+
execution.workspace === "none" ? "You are running in a scratch workspace with no repository checkout mounted. Do not assume git history or repo files are present unless the scenario provided them explicitly." : execution.workspace === "shared_mount" ? "You are running against the daemon shared mount. Treat any repository mutations as affecting the mounted checkout directly." : "You are running in a dedicated disposable git worktree isolated from the daemon shared checkout.",
|
|
7167
|
+
""
|
|
7168
|
+
].join("\n");
|
|
7169
|
+
const contextDisciplineSection = hasContext ? [
|
|
7170
|
+
"### Injected context discipline",
|
|
7171
|
+
"",
|
|
7172
|
+
"This task includes extra injected context from the task creator.",
|
|
7173
|
+
"You MUST inspect and use that context BEFORE you write solution",
|
|
7174
|
+
"files or draft your final answer.",
|
|
7175
|
+
"Do not solve first and only review the context afterward.",
|
|
7176
|
+
hasInlineContext ? "For `context_inline`, your FIRST content-inspection step should be a `read` of `/workspace/context-pack.md` before your first `write` call. The same content is also mirrored in `/workspace/AGENTS.md` and may be referenced from `/workspace/.claude/CLAUDE.md`." : "If injected context was provided as a skill, inspect that task-injected context before solving.",
|
|
7177
|
+
hasInlineContext ? "If `/workspace/context-pack.md` exists and you skip reading it before writing solution files, you are not following the task instructions." : "Do not rely on memory alone when task-injected context is available; inspect it first.",
|
|
7178
|
+
"If the injected context contains repo- or workflow-specific rules,",
|
|
7179
|
+
"those rules override your generic instincts.",
|
|
7030
7180
|
""
|
|
7031
7181
|
].join("\n") : "";
|
|
7032
7182
|
const finalOutputBlock = buildFinalOutputBlock({
|
|
@@ -7039,7 +7189,13 @@ function buildRunEvalUserPrompt(input, ctx) {
|
|
|
7039
7189
|
" \"totalTokens\": <int>,",
|
|
7040
7190
|
" \"durationMs\": <int>,",
|
|
7041
7191
|
" \"traceparent\": \"<from claim>\",",
|
|
7042
|
-
" \"verification\":
|
|
7192
|
+
" \"verification\": {",
|
|
7193
|
+
" \"inputCid\": \"<task inputCid>\",",
|
|
7194
|
+
" \"results\": [",
|
|
7195
|
+
" { \"id\": \"<criterion id>\", \"kind\": \"rubric\", \"status\": \"pass|fail|skip\", \"detail\": \"<optional one-liner>\" }",
|
|
7196
|
+
" ],",
|
|
7197
|
+
" \"passed\": <boolean>",
|
|
7198
|
+
" } // required iff input.successCriteria; must be an object, never a string",
|
|
7043
7199
|
"}"
|
|
7044
7200
|
].join("\n")
|
|
7045
7201
|
});
|
|
@@ -7047,6 +7203,8 @@ function buildRunEvalUserPrompt(input, ctx) {
|
|
|
7047
7203
|
"# Run Eval Agent\n",
|
|
7048
7204
|
`You are running an evaluation scenario as variant \`${variantLabel}\`.\nTask id: \`${ctx.taskId}\`\n`,
|
|
7049
7205
|
correlationSection,
|
|
7206
|
+
executionSection,
|
|
7207
|
+
contextDisciplineSection,
|
|
7050
7208
|
`### Scenario\n\n${scenario.prompt}\n`,
|
|
7051
7209
|
inputFilesSection,
|
|
7052
7210
|
verificationSection,
|
|
@@ -7118,14 +7276,25 @@ function buildTaskUserPrompt(task, ctx) {
|
|
|
7118
7276
|
diaryId: ctx.diaryId,
|
|
7119
7277
|
taskId: ctx.taskId
|
|
7120
7278
|
});
|
|
7121
|
-
case
|
|
7122
|
-
if (!Check(
|
|
7123
|
-
const errors = [...Errors(
|
|
7124
|
-
throw new Error(`
|
|
7279
|
+
case JUDGE_EVAL_ATTEMPT_TYPE:
|
|
7280
|
+
if (!Check(JudgeEvalAttemptInput, task.input)) {
|
|
7281
|
+
const errors = [...Errors(JudgeEvalAttemptInput, task.input)];
|
|
7282
|
+
throw new Error(`judge_eval_attempt input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
|
|
7125
7283
|
}
|
|
7126
|
-
return
|
|
7284
|
+
return buildJudgeEvalAttemptUserPrompt(task.input, {
|
|
7127
7285
|
diaryId: ctx.diaryId,
|
|
7128
|
-
taskId: ctx.taskId
|
|
7286
|
+
taskId: ctx.taskId,
|
|
7287
|
+
workspace: ctx.workspace
|
|
7288
|
+
});
|
|
7289
|
+
case PR_REVIEW_TYPE:
|
|
7290
|
+
if (!Check(PrReviewInput, task.input)) {
|
|
7291
|
+
const errors = [...Errors(PrReviewInput, task.input)];
|
|
7292
|
+
throw new Error(`pr_review input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
|
|
7293
|
+
}
|
|
7294
|
+
return buildPrReviewUserPrompt(task.input, {
|
|
7295
|
+
diaryId: ctx.diaryId,
|
|
7296
|
+
taskId: ctx.taskId,
|
|
7297
|
+
workspace: ctx.workspace
|
|
7129
7298
|
});
|
|
7130
7299
|
case RUN_EVAL_TYPE:
|
|
7131
7300
|
if (!Check(RunEvalInput, task.input)) {
|
|
@@ -9879,12 +10048,20 @@ var MoltNetError = class extends Error {
|
|
|
9879
10048
|
code;
|
|
9880
10049
|
statusCode;
|
|
9881
10050
|
detail;
|
|
10051
|
+
/**
|
|
10052
|
+
* Populated when the server returned a `VALIDATION_FAILED` problem
|
|
10053
|
+
* (status 400) with field-level errors. Empty / undefined for every
|
|
10054
|
+
* other problem kind. Imposer scripts surface these to operators so
|
|
10055
|
+
* they don't have to re-run with curl to see what was rejected.
|
|
10056
|
+
*/
|
|
10057
|
+
validationErrors;
|
|
9882
10058
|
constructor(message, options) {
|
|
9883
10059
|
super(message);
|
|
9884
10060
|
this.name = "MoltNetError";
|
|
9885
10061
|
this.code = options.code;
|
|
9886
10062
|
this.statusCode = options.statusCode;
|
|
9887
10063
|
this.detail = options.detail;
|
|
10064
|
+
this.validationErrors = options.validationErrors;
|
|
9888
10065
|
}
|
|
9889
10066
|
};
|
|
9890
10067
|
var NetworkError = class extends MoltNetError {
|
|
@@ -9908,10 +10085,14 @@ var AuthenticationError = class extends MoltNetError {
|
|
|
9908
10085
|
};
|
|
9909
10086
|
function problemToError(problem, statusCode) {
|
|
9910
10087
|
const title = problem.title ?? "Request failed";
|
|
9911
|
-
|
|
10088
|
+
const message = problem.detail ? `${title}: ${problem.detail}` : title;
|
|
10089
|
+
const rawErrors = problem.errors;
|
|
10090
|
+
const validationErrors = Array.isArray(rawErrors) ? rawErrors.filter((e) => typeof e === "object" && e !== null && typeof e.field === "string" && typeof e.message === "string") : void 0;
|
|
10091
|
+
return new MoltNetError(message, {
|
|
9912
10092
|
code: problem.type ?? problem.code ?? "UNKNOWN",
|
|
9913
10093
|
statusCode,
|
|
9914
|
-
detail: problem.detail
|
|
10094
|
+
detail: problem.detail,
|
|
10095
|
+
validationErrors
|
|
9915
10096
|
});
|
|
9916
10097
|
}
|
|
9917
10098
|
//#endregion
|
|
@@ -13836,31 +14017,6 @@ function abortableSleep(ms, signal) {
|
|
|
13836
14017
|
});
|
|
13837
14018
|
}
|
|
13838
14019
|
//#endregion
|
|
13839
|
-
//#region ../../libs/agent-runtime/src/subagent-output-contracts.ts
|
|
13840
|
-
/**
|
|
13841
|
-
* Construct an immutable contract registry from a static list.
|
|
13842
|
-
*
|
|
13843
|
-
* The resulting registry is safe to share across sessions and
|
|
13844
|
-
* invocations — no mutation is possible after construction.
|
|
13845
|
-
*/
|
|
13846
|
-
function createSubagentContractRegistry(contracts) {
|
|
13847
|
-
const lookup = /* @__PURE__ */ new Map();
|
|
13848
|
-
for (const c of contracts) {
|
|
13849
|
-
if (!c.name || c.name.trim().length === 0) throw new Error("subagent output contract name is required");
|
|
13850
|
-
if (!/^[a-z][a-z0-9_]*$/.test(c.name)) throw new Error(`subagent output contract name '${c.name}' must be lower_snake_case (starts with a letter, then [a-z0-9_]+)`);
|
|
13851
|
-
if (lookup.has(c.name)) throw new Error(`duplicate subagent output contract name '${c.name}' in constructor args`);
|
|
13852
|
-
lookup.set(c.name, c);
|
|
13853
|
-
}
|
|
13854
|
-
return {
|
|
13855
|
-
get(name) {
|
|
13856
|
-
return lookup.get(name) ?? null;
|
|
13857
|
-
},
|
|
13858
|
-
list() {
|
|
13859
|
-
return [...lookup.values()];
|
|
13860
|
-
}
|
|
13861
|
-
};
|
|
13862
|
-
}
|
|
13863
|
-
//#endregion
|
|
13864
14020
|
//#region ../../libs/pi-extension/src/moltnet/render-phase6.ts
|
|
13865
14021
|
function slugToTitle(value) {
|
|
13866
14022
|
return value.split(/[:/_-]+/).filter(Boolean).map((part) => part[0]?.toUpperCase() + part.slice(1)).join(" ");
|
|
@@ -14471,6 +14627,41 @@ function createMoltNetTools(config) {
|
|
|
14471
14627
|
};
|
|
14472
14628
|
}
|
|
14473
14629
|
});
|
|
14630
|
+
const listTaskMessages = defineTool({
|
|
14631
|
+
name: "moltnet_list_task_messages",
|
|
14632
|
+
label: "List MoltNet Task Attempt Messages",
|
|
14633
|
+
description: "List messages for a specific task attempt. Use this when you need the turn-by-turn execution record behind an accepted attempt — tool calls, text deltas, and error/info events that do not appear in the attempt output alone.",
|
|
14634
|
+
parameters: Type.Object({
|
|
14635
|
+
taskId: Type.String({ description: "Task ID (UUID)." }),
|
|
14636
|
+
attemptN: Type.Integer({
|
|
14637
|
+
minimum: 1,
|
|
14638
|
+
description: "Attempt number to inspect."
|
|
14639
|
+
}),
|
|
14640
|
+
afterSeq: Type.Optional(Type.Integer({
|
|
14641
|
+
minimum: 0,
|
|
14642
|
+
description: "Optional cursor: only return messages with seq > afterSeq."
|
|
14643
|
+
})),
|
|
14644
|
+
limit: Type.Optional(Type.Integer({
|
|
14645
|
+
minimum: 1,
|
|
14646
|
+
maximum: 500,
|
|
14647
|
+
description: "Optional maximum messages to return. Defaults to the API value."
|
|
14648
|
+
}))
|
|
14649
|
+
}),
|
|
14650
|
+
async execute(_id, params) {
|
|
14651
|
+
const { agent } = ensureConnected(config);
|
|
14652
|
+
const messages = await agent.tasks.listMessages(params.taskId, params.attemptN, {
|
|
14653
|
+
afterSeq: params.afterSeq,
|
|
14654
|
+
limit: params.limit
|
|
14655
|
+
});
|
|
14656
|
+
return {
|
|
14657
|
+
content: [{
|
|
14658
|
+
type: "text",
|
|
14659
|
+
text: JSON.stringify(messages, null, 2)
|
|
14660
|
+
}],
|
|
14661
|
+
details: {}
|
|
14662
|
+
};
|
|
14663
|
+
}
|
|
14664
|
+
});
|
|
14474
14665
|
const reviewSessionErrors = defineTool({
|
|
14475
14666
|
name: "moltnet_review_session_errors",
|
|
14476
14667
|
label: "Review Session Tool Errors",
|
|
@@ -14519,6 +14710,7 @@ function createMoltNetTools(config) {
|
|
|
14519
14710
|
createEntry,
|
|
14520
14711
|
getTask,
|
|
14521
14712
|
listTaskAttempts,
|
|
14713
|
+
listTaskMessages,
|
|
14522
14714
|
reviewSessionErrors,
|
|
14523
14715
|
defineTool({
|
|
14524
14716
|
name: "moltnet_host_exec",
|
|
@@ -14817,6 +15009,12 @@ var GUEST_WORKSPACE$1 = "/workspace";
|
|
|
14817
15009
|
* investigation and the alternatives we rejected.
|
|
14818
15010
|
*/
|
|
14819
15011
|
var GUEST_TASK_SKILLS_MOUNT = "/moltnet-task-skills";
|
|
15012
|
+
function shouldRunResumeCommand(entry, ctx) {
|
|
15013
|
+
if (typeof entry === "string") return true;
|
|
15014
|
+
const workspaceModes = entry.when?.workspaceMode;
|
|
15015
|
+
if (workspaceModes && !workspaceModes.includes(ctx.workspaceMode)) return false;
|
|
15016
|
+
return true;
|
|
15017
|
+
}
|
|
14820
15018
|
/**
|
|
14821
15019
|
* Resolve the main worktree root (where .moltnet/ lives — it's untracked,
|
|
14822
15020
|
* only exists in the main worktree, not in git worktrees).
|
|
@@ -14962,6 +15160,7 @@ async function resumeVm(config) {
|
|
|
14962
15160
|
...envOverrides
|
|
14963
15161
|
};
|
|
14964
15162
|
const resources = config.sandboxConfig?.resources;
|
|
15163
|
+
const workspaceMode = config.workspaceMode ?? "shared_mount";
|
|
14965
15164
|
const vm = await VmCheckpoint.load(config.checkpointPath).resume({
|
|
14966
15165
|
httpHooks,
|
|
14967
15166
|
env: vmEnv,
|
|
@@ -14980,7 +15179,32 @@ async function resumeVm(config) {
|
|
|
14980
15179
|
'`);
|
|
14981
15180
|
await vmRun(vm, "DNS resolvers", `printf 'nameserver 8.8.8.8\\nnameserver 1.1.1.1\\n' > /etc/resolv.conf`);
|
|
14982
15181
|
await vmRun(vm, "git safe.directory", `git config --system --add safe.directory '*'`);
|
|
14983
|
-
for (const [i,
|
|
15182
|
+
for (const [i, entry] of (config.sandboxConfig?.resumeCommands ?? []).entries()) {
|
|
15183
|
+
if (!shouldRunResumeCommand(entry, { workspaceMode })) continue;
|
|
15184
|
+
const { run, retries, backoffMs } = typeof entry === "string" ? {
|
|
15185
|
+
run: entry,
|
|
15186
|
+
retries: 0,
|
|
15187
|
+
backoffMs: 2e3
|
|
15188
|
+
} : {
|
|
15189
|
+
run: entry.run,
|
|
15190
|
+
retries: entry.retries ?? 0,
|
|
15191
|
+
backoffMs: entry.retryBackoffMs ?? 2e3
|
|
15192
|
+
};
|
|
15193
|
+
const label = `resumeCommands[${i}]`;
|
|
15194
|
+
let lastErr;
|
|
15195
|
+
for (let attempt = 0; attempt <= retries; attempt++) try {
|
|
15196
|
+
await vmRun(vm, label, run);
|
|
15197
|
+
lastErr = void 0;
|
|
15198
|
+
break;
|
|
15199
|
+
} catch (err) {
|
|
15200
|
+
lastErr = err;
|
|
15201
|
+
if (attempt === retries) break;
|
|
15202
|
+
await new Promise((resolve) => {
|
|
15203
|
+
setTimeout(resolve, (attempt + 1) * backoffMs);
|
|
15204
|
+
});
|
|
15205
|
+
}
|
|
15206
|
+
if (lastErr) throw lastErr instanceof Error ? lastErr : new Error(String(lastErr));
|
|
15207
|
+
}
|
|
14984
15208
|
const vmSshDir = `${vmAgentDir}/ssh`;
|
|
14985
15209
|
await vm.exec(`mkdir -p ${vmAgentDir}/ssh /home/agent/.pi/agent`);
|
|
14986
15210
|
if (creds.piAuthJson !== null) await vm.fs.writeFile("/home/agent/.pi/agent/auth.json", creds.piAuthJson, { mode: 384 });
|
|
@@ -15359,7 +15583,8 @@ async function buildAgentSession(args) {
|
|
|
15359
15583
|
await resourceLoader.reload();
|
|
15360
15584
|
const sessionManager = args.sessionPersistence ? await resolvePersistentSessionManager({
|
|
15361
15585
|
cwd: args.cwdPath,
|
|
15362
|
-
sessionDir: args.sessionPersistence.sessionDir
|
|
15586
|
+
sessionDir: args.sessionPersistence.sessionDir,
|
|
15587
|
+
forkFromSessionPath: args.sessionPersistence.forkFromSessionPath
|
|
15363
15588
|
}) : SessionManager.inMemory(args.cwdPath);
|
|
15364
15589
|
return (await createAgentSession({
|
|
15365
15590
|
agentDir: args.piAuthDir,
|
|
@@ -15371,6 +15596,7 @@ async function buildAgentSession(args) {
|
|
|
15371
15596
|
})).session;
|
|
15372
15597
|
}
|
|
15373
15598
|
async function resolvePersistentSessionManager(args) {
|
|
15599
|
+
if (args.forkFromSessionPath) return SessionManager.forkFrom(args.forkFromSessionPath, args.cwd, args.sessionDir);
|
|
15374
15600
|
await SessionManager.list(args.cwd, args.sessionDir);
|
|
15375
15601
|
return SessionManager.continueRecent(args.cwd, args.sessionDir);
|
|
15376
15602
|
}
|
|
@@ -15411,6 +15637,11 @@ async function resolvePersistentSessionManager(args) {
|
|
|
15411
15637
|
* paths under this mount via `toGuestPath` in `tool-operations.ts`.
|
|
15412
15638
|
*/
|
|
15413
15639
|
var SKILL_ROOT_IN_VM = GUEST_TASK_SKILLS_MOUNT;
|
|
15640
|
+
var INLINE_CONTEXT_ROOT_IN_VM = "/workspace/.moltnet/context";
|
|
15641
|
+
var WORKSPACE_CONTEXT_PACK = "/workspace/context-pack.md";
|
|
15642
|
+
var WORKSPACE_AGENTS_MD = "/workspace/AGENTS.md";
|
|
15643
|
+
var WORKSPACE_CLAUDE_DIR = "/workspace/.claude";
|
|
15644
|
+
var WORKSPACE_CLAUDE_MD = "/workspace/.claude/CLAUDE.md";
|
|
15414
15645
|
/** Bounds borrowed from pi's skill validation; conservative caps so a
|
|
15415
15646
|
* malformed SKILL.md doesn't bloat the system prompt. */
|
|
15416
15647
|
var MAX_SKILL_NAME = 64;
|
|
@@ -15421,21 +15652,40 @@ var MAX_SKILL_DESCRIPTION = 1024;
|
|
|
15421
15652
|
*/
|
|
15422
15653
|
async function injectTaskContext(args) {
|
|
15423
15654
|
const skills = [];
|
|
15655
|
+
const inlineContexts = [];
|
|
15424
15656
|
const resolved = await resolveTaskContext({
|
|
15425
15657
|
context: args.context,
|
|
15426
|
-
deliver: {
|
|
15427
|
-
|
|
15428
|
-
|
|
15429
|
-
|
|
15430
|
-
|
|
15431
|
-
|
|
15432
|
-
|
|
15433
|
-
|
|
15434
|
-
|
|
15435
|
-
|
|
15436
|
-
|
|
15437
|
-
|
|
15658
|
+
deliver: {
|
|
15659
|
+
skill: async ({ slug, content }) => {
|
|
15660
|
+
const dir = `${SKILL_ROOT_IN_VM}/${slug}`;
|
|
15661
|
+
const filePath = `${dir}/SKILL.md`;
|
|
15662
|
+
await args.fs.mkdir(dir, { recursive: true });
|
|
15663
|
+
await args.fs.writeFile(filePath, content, { mode: 420 });
|
|
15664
|
+
skills.push(buildSyntheticSkill({
|
|
15665
|
+
slug,
|
|
15666
|
+
content,
|
|
15667
|
+
filePath,
|
|
15668
|
+
dir
|
|
15669
|
+
}));
|
|
15670
|
+
},
|
|
15671
|
+
contextFile: async ({ suggestedFileName, content }) => {
|
|
15672
|
+
await args.fs.mkdir(INLINE_CONTEXT_ROOT_IN_VM, { recursive: true });
|
|
15673
|
+
const filePath = `${INLINE_CONTEXT_ROOT_IN_VM}/${suggestedFileName}`;
|
|
15674
|
+
await args.fs.writeFile(filePath, content, { mode: 420 });
|
|
15675
|
+
inlineContexts.push({
|
|
15676
|
+
slug: suggestedFileName.replace(/\.md$/u, ""),
|
|
15677
|
+
content
|
|
15678
|
+
});
|
|
15679
|
+
}
|
|
15680
|
+
}
|
|
15438
15681
|
});
|
|
15682
|
+
if (inlineContexts.length > 0) {
|
|
15683
|
+
const packContent = buildWorkspaceContextPack(inlineContexts);
|
|
15684
|
+
await args.fs.writeFile(WORKSPACE_CONTEXT_PACK, packContent, { mode: 420 });
|
|
15685
|
+
await args.fs.writeFile(WORKSPACE_AGENTS_MD, packContent, { mode: 420 });
|
|
15686
|
+
await args.fs.mkdir(WORKSPACE_CLAUDE_DIR, { recursive: true });
|
|
15687
|
+
await args.fs.writeFile(WORKSPACE_CLAUDE_MD, "@../context-pack.md\n", { mode: 420 });
|
|
15688
|
+
}
|
|
15439
15689
|
return {
|
|
15440
15690
|
injected: resolved.injected,
|
|
15441
15691
|
skills,
|
|
@@ -15443,6 +15693,17 @@ async function injectTaskContext(args) {
|
|
|
15443
15693
|
userInlineSuffix: resolved.userInlineSuffix
|
|
15444
15694
|
};
|
|
15445
15695
|
}
|
|
15696
|
+
function buildWorkspaceContextPack(contexts) {
|
|
15697
|
+
return [
|
|
15698
|
+
"# Context Pack",
|
|
15699
|
+
"",
|
|
15700
|
+
...contexts.map(({ slug, content }) => [
|
|
15701
|
+
`## ${slug}`,
|
|
15702
|
+
"",
|
|
15703
|
+
content.trimEnd()
|
|
15704
|
+
].join("\n"))
|
|
15705
|
+
].join("\n\n").trimEnd() + "\n";
|
|
15706
|
+
}
|
|
15446
15707
|
/**
|
|
15447
15708
|
* Build a `Skill` object pi will faithfully render in
|
|
15448
15709
|
* `<available_skills>`. We extract `name` and `description` from the
|
|
@@ -15806,7 +16067,7 @@ async function parseStructuredTaskOutput(assistantText, taskType, opts = {}) {
|
|
|
15806
16067
|
}
|
|
15807
16068
|
};
|
|
15808
16069
|
}
|
|
15809
|
-
const errors = validateTaskOutput(taskType, extracted);
|
|
16070
|
+
const errors = validateTaskOutput(taskType, extracted, opts.input);
|
|
15810
16071
|
if (errors.length > 0) {
|
|
15811
16072
|
const details = errors.slice(0, 3).map((error) => `${error.field}: ${error.message}`);
|
|
15812
16073
|
const [firstError] = errors;
|
|
@@ -15920,7 +16181,7 @@ function createSubmitOutputTool(taskType, opts = {}) {
|
|
|
15920
16181
|
description: contract.description,
|
|
15921
16182
|
parameters: schema,
|
|
15922
16183
|
async execute(_id, params) {
|
|
15923
|
-
const errors = validateTaskOutput(taskType, params);
|
|
16184
|
+
const errors = validateTaskOutput(taskType, params, opts.input);
|
|
15924
16185
|
if (errors.length > 0) {
|
|
15925
16186
|
const detailMsg = errors.slice(0, 3).map((err) => `${err.field}: ${err.message}`).join("; ");
|
|
15926
16187
|
const details = {
|
|
@@ -15989,6 +16250,39 @@ function resolveSubmitTools(taskType, opts = {}) {
|
|
|
15989
16250
|
//#region ../../libs/pi-extension/src/runtime/task-workspace.ts
|
|
15990
16251
|
function prepareTaskWorkspace(task, requestedMountPath, executionPlan) {
|
|
15991
16252
|
const branch = executionPlan?.worktreeBranch ?? null;
|
|
16253
|
+
const workspaceMode = executionPlan?.workspaceMode ?? "shared_mount";
|
|
16254
|
+
const attachedWorkspace = executionPlan?.workspaceAttachment ?? null;
|
|
16255
|
+
if (attachedWorkspace) return {
|
|
16256
|
+
mountPath: attachedWorkspace.mountPath,
|
|
16257
|
+
cwdPath: attachedWorkspace.cwdPath,
|
|
16258
|
+
mode: workspaceMode,
|
|
16259
|
+
branch,
|
|
16260
|
+
cleanup: () => {}
|
|
16261
|
+
};
|
|
16262
|
+
if (workspaceMode === "scratch_mount") {
|
|
16263
|
+
const scratchDir = resolveTaskScratchPath(findMainWorktree(), executionPlan?.workspaceId ?? `task-${task.id}`);
|
|
16264
|
+
const keepWorkspace = executionPlan?.workspaceScope === "session" && executionPlan.sessionKey !== null;
|
|
16265
|
+
if (keepWorkspace) mkdirSync(scratchDir, { recursive: true });
|
|
16266
|
+
else {
|
|
16267
|
+
rmSync(scratchDir, {
|
|
16268
|
+
recursive: true,
|
|
16269
|
+
force: true
|
|
16270
|
+
});
|
|
16271
|
+
mkdirSync(scratchDir, { recursive: true });
|
|
16272
|
+
}
|
|
16273
|
+
return {
|
|
16274
|
+
mountPath: scratchDir,
|
|
16275
|
+
cwdPath: scratchDir,
|
|
16276
|
+
mode: "scratch_mount",
|
|
16277
|
+
branch: null,
|
|
16278
|
+
cleanup: keepWorkspace ? () => {} : () => {
|
|
16279
|
+
rmSync(scratchDir, {
|
|
16280
|
+
recursive: true,
|
|
16281
|
+
force: true
|
|
16282
|
+
});
|
|
16283
|
+
}
|
|
16284
|
+
};
|
|
16285
|
+
}
|
|
15992
16286
|
if (!branch) return {
|
|
15993
16287
|
mountPath: requestedMountPath,
|
|
15994
16288
|
cwdPath: requestedMountPath,
|
|
@@ -16026,6 +16320,9 @@ function prepareTaskWorkspace(task, requestedMountPath, executionPlan) {
|
|
|
16026
16320
|
function resolveTaskWorktreePath(mainRepo, workspaceId) {
|
|
16027
16321
|
return join(mainRepo, ".worktrees", workspaceId);
|
|
16028
16322
|
}
|
|
16323
|
+
function resolveTaskScratchPath(mainRepo, workspaceId) {
|
|
16324
|
+
return join(mainRepo, ".moltnet", "d", "task-workspaces", workspaceId);
|
|
16325
|
+
}
|
|
16029
16326
|
function ensureReusableTaskWorktree(mainRepo, worktreeDir, branch) {
|
|
16030
16327
|
if (isRegisteredWorktree$1(mainRepo, worktreeDir)) return;
|
|
16031
16328
|
if (existsSync(worktreeDir)) throw new Error(`Expected reusable worktree ${worktreeDir} to be git-managed, but it exists outside git worktree metadata.`);
|
|
@@ -16262,12 +16559,14 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
16262
16559
|
return makeFailedOutput("worktree_setup_failed", message);
|
|
16263
16560
|
}
|
|
16264
16561
|
try {
|
|
16562
|
+
const sandboxConfig = applyExecutionPlanSandboxOverrides(opts.sandboxConfig, executionPlan);
|
|
16265
16563
|
managed = await resumeVm({
|
|
16266
16564
|
checkpointPath,
|
|
16267
16565
|
agentName: opts.agentName,
|
|
16268
16566
|
mountPath,
|
|
16567
|
+
workspaceMode: workspace.mode,
|
|
16269
16568
|
extraAllowedHosts: opts.extraAllowedHosts,
|
|
16270
|
-
sandboxConfig
|
|
16569
|
+
sandboxConfig
|
|
16271
16570
|
});
|
|
16272
16571
|
} catch (err) {
|
|
16273
16572
|
const message = err instanceof Error ? err.message : String(err);
|
|
@@ -16296,7 +16595,8 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
16296
16595
|
taskId: task.id,
|
|
16297
16596
|
workspace: {
|
|
16298
16597
|
mode: activeWorkspace.mode,
|
|
16299
|
-
branch: activeWorkspace.branch
|
|
16598
|
+
branch: activeWorkspace.branch,
|
|
16599
|
+
attached: executionPlan?.workspaceAttachment !== void 0
|
|
16300
16600
|
},
|
|
16301
16601
|
extras: opts.promptExtras
|
|
16302
16602
|
});
|
|
@@ -16338,7 +16638,10 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
16338
16638
|
createEditToolDefinition(mountPath, { operations: createGondolinEditOps(managed.vm, mountPath) }),
|
|
16339
16639
|
createBashToolDefinition(mountPath, { operations: createGondolinBashOps(managed.vm, mountPath) })
|
|
16340
16640
|
];
|
|
16341
|
-
const { handle: submitToolHandle, tools: submitToolDefs } = resolveSubmitTools(task.taskType, {
|
|
16641
|
+
const { handle: submitToolHandle, tools: submitToolDefs } = resolveSubmitTools(task.taskType, {
|
|
16642
|
+
model: opts.model,
|
|
16643
|
+
input: task.input
|
|
16644
|
+
});
|
|
16342
16645
|
const submitTools = submitToolDefs;
|
|
16343
16646
|
try {
|
|
16344
16647
|
const moltnetAgent = await connect({ configDir: managed.agentDir });
|
|
@@ -16557,8 +16860,20 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
16557
16860
|
phase: "output_validation"
|
|
16558
16861
|
});
|
|
16559
16862
|
}
|
|
16560
|
-
else {
|
|
16561
|
-
|
|
16863
|
+
else if (submitToolHandle) {
|
|
16864
|
+
parseError = {
|
|
16865
|
+
code: "output_missing",
|
|
16866
|
+
message: "Agent did not submit output through the task submit tool. A valid submit tool call is required to complete this task type."
|
|
16867
|
+
};
|
|
16868
|
+
await emit("error", {
|
|
16869
|
+
message: parseError.message,
|
|
16870
|
+
phase: "output_validation"
|
|
16871
|
+
});
|
|
16872
|
+
} else {
|
|
16873
|
+
const parsed = await parseStructuredTaskOutput(assistantText, task.taskType, {
|
|
16874
|
+
model: opts.model,
|
|
16875
|
+
input: task.input
|
|
16876
|
+
});
|
|
16562
16877
|
parsedOutput = parsed.output;
|
|
16563
16878
|
parsedOutputCid = parsed.outputCid;
|
|
16564
16879
|
parseError = parsed.error;
|
|
@@ -16644,6 +16959,18 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
16644
16959
|
}
|
|
16645
16960
|
}
|
|
16646
16961
|
}
|
|
16962
|
+
function applyExecutionPlanSandboxOverrides(sandboxConfig, executionPlan) {
|
|
16963
|
+
const shadowWrites = executionPlan?.workspaceAttachment?.shadowWrites;
|
|
16964
|
+
if (!shadowWrites) return sandboxConfig;
|
|
16965
|
+
return {
|
|
16966
|
+
...sandboxConfig,
|
|
16967
|
+
vfs: {
|
|
16968
|
+
...sandboxConfig?.vfs,
|
|
16969
|
+
shadow: ["**"],
|
|
16970
|
+
shadowMode: shadowWrites
|
|
16971
|
+
}
|
|
16972
|
+
};
|
|
16973
|
+
}
|
|
16647
16974
|
function emptyUsage(provider, model) {
|
|
16648
16975
|
return {
|
|
16649
16976
|
inputTokens: 0,
|
|
@@ -16868,11 +17195,12 @@ var DaemonSlotRegistryError = class extends Error {
|
|
|
16868
17195
|
this.name = "DaemonSlotRegistryError";
|
|
16869
17196
|
}
|
|
16870
17197
|
};
|
|
17198
|
+
var SqliteDatabaseSync = DatabaseSync;
|
|
16871
17199
|
var DaemonSlotRegistry = class {
|
|
16872
17200
|
db;
|
|
16873
17201
|
constructor(dbPath) {
|
|
16874
17202
|
try {
|
|
16875
|
-
this.db = new
|
|
17203
|
+
this.db = new SqliteDatabaseSync(dbPath);
|
|
16876
17204
|
this.withDb("initialize schema", () => {
|
|
16877
17205
|
this.db.exec(`
|
|
16878
17206
|
PRAGMA journal_mode = WAL;
|
|
@@ -16921,6 +17249,9 @@ var DaemonSlotRegistry = class {
|
|
|
16921
17249
|
|
|
16922
17250
|
CREATE INDEX IF NOT EXISTS daemon_slots_expires_idx
|
|
16923
17251
|
ON daemon_slots (expires_at_ms);
|
|
17252
|
+
|
|
17253
|
+
CREATE INDEX IF NOT EXISTS daemon_slots_task_attempt_idx
|
|
17254
|
+
ON daemon_slots (last_task_id, last_attempt_n, last_used_at_ms DESC);
|
|
16924
17255
|
`);
|
|
16925
17256
|
});
|
|
16926
17257
|
} catch (error) {
|
|
@@ -17010,6 +17341,30 @@ var DaemonSlotRegistry = class {
|
|
|
17010
17341
|
SET session_path = ?
|
|
17011
17342
|
WHERE agent_name = ? AND provider = ? AND model = ? AND slot_key = ?`).run(sessionPath, identity.agentName, identity.provider, identity.model, slotKey));
|
|
17012
17343
|
}
|
|
17344
|
+
findLatestProducerSlotByTaskAttempt(taskId, attemptN) {
|
|
17345
|
+
const slot = this.withDb("find producer slot by task attempt", () => this.db.prepare(`SELECT
|
|
17346
|
+
agent_name as agentName,
|
|
17347
|
+
provider,
|
|
17348
|
+
model,
|
|
17349
|
+
slot_key as slotKey,
|
|
17350
|
+
task_type as taskType,
|
|
17351
|
+
state,
|
|
17352
|
+
last_task_id as lastTaskId,
|
|
17353
|
+
last_attempt_n as lastAttemptN,
|
|
17354
|
+
created_at_ms as createdAtMs,
|
|
17355
|
+
last_used_at_ms as lastUsedAtMs,
|
|
17356
|
+
expires_at_ms as expiresAtMs
|
|
17357
|
+
FROM daemon_slots
|
|
17358
|
+
WHERE last_task_id = ? AND last_attempt_n = ?
|
|
17359
|
+
ORDER BY last_used_at_ms DESC
|
|
17360
|
+
LIMIT 1`).get(taskId, attemptN) ?? null);
|
|
17361
|
+
if (!slot) return null;
|
|
17362
|
+
return {
|
|
17363
|
+
slot,
|
|
17364
|
+
session: this.lookupSession(slot),
|
|
17365
|
+
workspace: this.lookupWorkspace(slot)
|
|
17366
|
+
};
|
|
17367
|
+
}
|
|
17013
17368
|
reapExpiredSlots(now = Date.now()) {
|
|
17014
17369
|
this.withDb("begin reap transaction", () => this.db.exec("BEGIN IMMEDIATE"));
|
|
17015
17370
|
try {
|
|
@@ -17053,8 +17408,8 @@ var DaemonSlotRegistry = class {
|
|
|
17053
17408
|
const deleteStmt = this.withDb("prepare expired slot delete", () => this.db.prepare("DELETE FROM daemon_slots WHERE agent_name = ? AND provider = ? AND model = ? AND slot_key = ?"));
|
|
17054
17409
|
const out = [];
|
|
17055
17410
|
for (const slot of slots) {
|
|
17056
|
-
const session = this.
|
|
17057
|
-
const workspace = this.
|
|
17411
|
+
const session = this.lookupSession(slot, selectSession);
|
|
17412
|
+
const workspace = this.lookupWorkspace(slot, selectWorkspace);
|
|
17058
17413
|
out.push({
|
|
17059
17414
|
slot,
|
|
17060
17415
|
session,
|
|
@@ -17082,6 +17437,29 @@ var DaemonSlotRegistry = class {
|
|
|
17082
17437
|
throw new DaemonSlotRegistryError(operation, error);
|
|
17083
17438
|
}
|
|
17084
17439
|
}
|
|
17440
|
+
lookupSession(slot, stmt = this.withDb("prepare slot session lookup", () => this.db.prepare(`SELECT
|
|
17441
|
+
agent_name as agentName,
|
|
17442
|
+
provider,
|
|
17443
|
+
model,
|
|
17444
|
+
slot_key as slotKey,
|
|
17445
|
+
session_dir as sessionDir,
|
|
17446
|
+
session_path as sessionPath
|
|
17447
|
+
FROM daemon_slot_sessions
|
|
17448
|
+
WHERE agent_name = ? AND provider = ? AND model = ? AND slot_key = ?`))) {
|
|
17449
|
+
return this.withDb("select slot session", () => stmt.get(slot.agentName, slot.provider, slot.model, slot.slotKey) ?? null);
|
|
17450
|
+
}
|
|
17451
|
+
lookupWorkspace(slot, stmt = this.withDb("prepare slot workspace lookup", () => this.db.prepare(`SELECT
|
|
17452
|
+
agent_name as agentName,
|
|
17453
|
+
provider,
|
|
17454
|
+
model,
|
|
17455
|
+
slot_key as slotKey,
|
|
17456
|
+
workspace_id as workspaceId,
|
|
17457
|
+
worktree_path as worktreePath,
|
|
17458
|
+
worktree_branch as worktreeBranch
|
|
17459
|
+
FROM daemon_slot_workspaces
|
|
17460
|
+
WHERE agent_name = ? AND provider = ? AND model = ? AND slot_key = ?`))) {
|
|
17461
|
+
return this.withDb("select slot workspace", () => stmt.get(slot.agentName, slot.provider, slot.model, slot.slotKey) ?? null);
|
|
17462
|
+
}
|
|
17085
17463
|
};
|
|
17086
17464
|
function resolveLatestPiSessionPath(sessionDir) {
|
|
17087
17465
|
try {
|
|
@@ -17182,11 +17560,6 @@ function buildCustomSessionKey(task) {
|
|
|
17182
17560
|
if (!task.correlationId || !variantLabel) return null;
|
|
17183
17561
|
return `run_eval:correlation:${task.correlationId}:variant:${slugifySessionComponent(variantLabel)}`;
|
|
17184
17562
|
}
|
|
17185
|
-
case "judge_eval_variant": {
|
|
17186
|
-
const runTaskIds = Array.isArray(task.input.runTaskIds) ? task.input.runTaskIds.filter((value) => typeof value === "string") : [];
|
|
17187
|
-
if (runTaskIds.length < 1) return null;
|
|
17188
|
-
return `judge_eval_variant:run_tasks:${[...runTaskIds].sort().join(",")}`;
|
|
17189
|
-
}
|
|
17190
17563
|
default: return null;
|
|
17191
17564
|
}
|
|
17192
17565
|
}
|
|
@@ -17197,18 +17570,20 @@ function slugifySessionComponent(input) {
|
|
|
17197
17570
|
//#region src/lib/task-execution-plan.ts
|
|
17198
17571
|
function buildDaemonTaskExecutionPlan(task, stateDirs, identity, warmSessionTtlSec) {
|
|
17199
17572
|
const descriptor = deriveTaskSessionDescriptor(task);
|
|
17573
|
+
const workspaceMode = resolveTaskWorkspaceMode(task, descriptor.policy);
|
|
17200
17574
|
const slotKey = warmSessionTtlSec > 0 ? descriptor.sessionKey : null;
|
|
17201
17575
|
const workspaceScope = slotKey !== null ? descriptor.policy.workspaceScope : "attempt";
|
|
17202
17576
|
const slotId = slotKey ? buildDaemonSlotId(identity, slotKey) : null;
|
|
17203
17577
|
const sessionDir = slotId ? `${stateDirs.piSessionsDir}/${encodeURIComponent(slotId)}` : null;
|
|
17204
|
-
const worktreeBranch = resolveTaskWorktreeBranch(task,
|
|
17205
|
-
const workspaceId =
|
|
17578
|
+
const worktreeBranch = resolveTaskWorktreeBranch(task, workspaceMode);
|
|
17579
|
+
const workspaceId = workspaceMode !== "shared_mount" ? resolveTaskWorkspaceId(task, {
|
|
17206
17580
|
sessionKey: slotId,
|
|
17207
17581
|
workspaceScope,
|
|
17208
17582
|
sessionPersistence: sessionDir ? { sessionDir } : null
|
|
17209
17583
|
}) : null;
|
|
17210
17584
|
return {
|
|
17211
17585
|
descriptor,
|
|
17586
|
+
workspaceMode,
|
|
17212
17587
|
sessionKey: slotId,
|
|
17213
17588
|
slotKey,
|
|
17214
17589
|
slotId,
|
|
@@ -17237,8 +17612,8 @@ function slugSlotIdentityComponent(input) {
|
|
|
17237
17612
|
"-"
|
|
17238
17613
|
]);
|
|
17239
17614
|
}
|
|
17240
|
-
function resolveTaskWorktreeBranch(task,
|
|
17241
|
-
if (
|
|
17615
|
+
function resolveTaskWorktreeBranch(task, workspaceMode) {
|
|
17616
|
+
if (workspaceMode !== "dedicated_worktree") return null;
|
|
17242
17617
|
if (task.taskType === "fulfill_brief") {
|
|
17243
17618
|
const input = task.input;
|
|
17244
17619
|
const slug = slugifyAsciiLower(typeof input.title === "string" && input.title.trim().length > 0 ? input.title : typeof input.brief === "string" && input.brief.trim().length > 0 ? input.brief : task.taskType, 60) || "task";
|
|
@@ -17247,12 +17622,27 @@ function resolveTaskWorktreeBranch(task, policy) {
|
|
|
17247
17622
|
}
|
|
17248
17623
|
return `task/${slugifyAsciiLower(task.taskType, 60) || "task"}-${task.id.slice(0, 8)}`;
|
|
17249
17624
|
}
|
|
17625
|
+
function resolveTaskWorkspaceMode(task, policy) {
|
|
17626
|
+
if (task.taskType !== "run_eval") return policy.workspaceMode;
|
|
17627
|
+
switch (typeof task.input.execution?.workspace === "string" ? task.input.execution.workspace : null) {
|
|
17628
|
+
case "none": return "scratch_mount";
|
|
17629
|
+
case "shared_mount": return "shared_mount";
|
|
17630
|
+
case "dedicated_worktree": return "dedicated_worktree";
|
|
17631
|
+
default: return policy.workspaceMode;
|
|
17632
|
+
}
|
|
17633
|
+
}
|
|
17250
17634
|
function resolveTaskWorkspaceId(task, executionPlan) {
|
|
17251
17635
|
if (executionPlan.workspaceScope === "session" && executionPlan.sessionKey !== null) return `session-${encodeURIComponent(executionPlan.sessionKey)}`;
|
|
17252
17636
|
return `task-${task.id}`;
|
|
17253
17637
|
}
|
|
17254
17638
|
//#endregion
|
|
17255
17639
|
//#region src/lib/execution-plan-cache.ts
|
|
17640
|
+
var ProducerContextResolutionError = class extends Error {
|
|
17641
|
+
constructor(message) {
|
|
17642
|
+
super(message);
|
|
17643
|
+
this.name = "ProducerContextResolutionError";
|
|
17644
|
+
}
|
|
17645
|
+
};
|
|
17256
17646
|
function createExecutionPlanCache(args) {
|
|
17257
17647
|
const cache = /* @__PURE__ */ new Map();
|
|
17258
17648
|
return {
|
|
@@ -17260,7 +17650,7 @@ function createExecutionPlanCache(args) {
|
|
|
17260
17650
|
const key = buildClaimedTaskKey(claimedTask);
|
|
17261
17651
|
const existing = cache.get(key);
|
|
17262
17652
|
if (existing) return existing;
|
|
17263
|
-
const plan = buildDaemonTaskExecutionPlan(claimedTask.task, args.stateDirs, args.slotIdentity, args.warmSessionTtlSec);
|
|
17653
|
+
const plan = maybeAttachProducerContext(claimedTask, buildDaemonTaskExecutionPlan(claimedTask.task, args.stateDirs, args.slotIdentity, args.warmSessionTtlSec), args.stateDirs, args.slotRegistry);
|
|
17264
17654
|
cache.set(key, plan);
|
|
17265
17655
|
return plan;
|
|
17266
17656
|
},
|
|
@@ -17272,17 +17662,90 @@ function createExecutionPlanCache(args) {
|
|
|
17272
17662
|
function buildClaimedTaskKey(task) {
|
|
17273
17663
|
return `${task.task.id}:${task.attemptN}`;
|
|
17274
17664
|
}
|
|
17665
|
+
function maybeAttachProducerContext(claimedTask, basePlan, stateDirs, slotRegistry) {
|
|
17666
|
+
if (claimedTask.task.taskType !== "judge_eval_attempt") return basePlan;
|
|
17667
|
+
const targetTaskId = typeof claimedTask.task.input.targetTaskId === "string" ? claimedTask.task.input.targetTaskId : null;
|
|
17668
|
+
const targetAttemptN = typeof claimedTask.task.input.targetAttemptN === "number" ? claimedTask.task.input.targetAttemptN : null;
|
|
17669
|
+
if (!targetTaskId || !targetAttemptN) throw new ProducerContextResolutionError("judge_eval_attempt is missing targetTaskId/targetAttemptN");
|
|
17670
|
+
const producer = slotRegistry.findLatestProducerSlotByTaskAttempt(targetTaskId, targetAttemptN);
|
|
17671
|
+
if (!producer) throw new ProducerContextResolutionError(`No persisted producer daemon slot found for task ${targetTaskId} attempt ${targetAttemptN}`);
|
|
17672
|
+
const sourceSessionPath = resolveProducerSessionPath(producer);
|
|
17673
|
+
if (!sourceSessionPath) throw new ProducerContextResolutionError(`Producer task ${targetTaskId} attempt ${targetAttemptN} has no persisted Pi session path`);
|
|
17674
|
+
const attachedWorkspace = resolveProducerWorkspaceAttachment(producer, stateDirs);
|
|
17675
|
+
return {
|
|
17676
|
+
...basePlan,
|
|
17677
|
+
workspaceMode: attachedWorkspace.mode,
|
|
17678
|
+
worktreeBranch: attachedWorkspace.branch,
|
|
17679
|
+
workspaceAttachment: {
|
|
17680
|
+
mountPath: attachedWorkspace.mountPath,
|
|
17681
|
+
cwdPath: attachedWorkspace.cwdPath,
|
|
17682
|
+
shadowWrites: "tmpfs"
|
|
17683
|
+
},
|
|
17684
|
+
sessionPersistence: {
|
|
17685
|
+
sessionDir: `${stateDirs.piSessionsDir}/judge-${claimedTask.task.id}-attempt-${claimedTask.attemptN}`,
|
|
17686
|
+
forkFromSessionPath: sourceSessionPath
|
|
17687
|
+
}
|
|
17688
|
+
};
|
|
17689
|
+
}
|
|
17690
|
+
function resolveProducerSessionPath(producer) {
|
|
17691
|
+
const explicit = producer.session?.sessionPath ?? null;
|
|
17692
|
+
if (explicit && existsSync(explicit)) return explicit;
|
|
17693
|
+
const sessionDir = producer.session?.sessionDir ?? null;
|
|
17694
|
+
if (!sessionDir || !existsSync(sessionDir)) return null;
|
|
17695
|
+
const latest = resolveLatestPiSessionPath(sessionDir);
|
|
17696
|
+
return latest && existsSync(latest) ? latest : null;
|
|
17697
|
+
}
|
|
17698
|
+
function resolveProducerWorkspaceAttachment(producer, stateDirs) {
|
|
17699
|
+
const workspacePath = producer.workspace?.worktreePath ?? null;
|
|
17700
|
+
if (workspacePath) {
|
|
17701
|
+
if (existsSync(workspacePath)) return {
|
|
17702
|
+
mountPath: workspacePath,
|
|
17703
|
+
cwdPath: workspacePath,
|
|
17704
|
+
mode: producer.workspace?.worktreeBranch ? "dedicated_worktree" : "scratch_mount",
|
|
17705
|
+
branch: producer.workspace?.worktreeBranch ?? null
|
|
17706
|
+
};
|
|
17707
|
+
const recoveredPath = recoverScratchWorkspacePath(producer, stateDirs);
|
|
17708
|
+
if (recoveredPath) return {
|
|
17709
|
+
mountPath: recoveredPath,
|
|
17710
|
+
cwdPath: recoveredPath,
|
|
17711
|
+
mode: "scratch_mount",
|
|
17712
|
+
branch: null
|
|
17713
|
+
};
|
|
17714
|
+
throw new ProducerContextResolutionError(`Producer workspace path is missing on disk: ${workspacePath}`);
|
|
17715
|
+
}
|
|
17716
|
+
const sharedMountRoot = dirname(dirname(stateDirs.rootDir));
|
|
17717
|
+
if (!existsSync(sharedMountRoot)) throw new ProducerContextResolutionError(`Shared producer mount root is missing on disk: ${sharedMountRoot}`);
|
|
17718
|
+
return {
|
|
17719
|
+
mountPath: sharedMountRoot,
|
|
17720
|
+
cwdPath: sharedMountRoot,
|
|
17721
|
+
mode: "shared_mount",
|
|
17722
|
+
branch: null
|
|
17723
|
+
};
|
|
17724
|
+
}
|
|
17725
|
+
function recoverScratchWorkspacePath(producer, stateDirs) {
|
|
17726
|
+
if (producer.workspace?.worktreeBranch) return null;
|
|
17727
|
+
if (!producer.workspace?.workspaceId) return null;
|
|
17728
|
+
const fallback = join(stateDirs.rootDir, "task-workspaces", producer.workspace.workspaceId);
|
|
17729
|
+
return existsSync(fallback) ? fallback : null;
|
|
17730
|
+
}
|
|
17275
17731
|
//#endregion
|
|
17276
17732
|
//#region src/lib/finalize.ts
|
|
17277
17733
|
async function finalizeTask(agent, output, ctx = {}) {
|
|
17278
17734
|
if (output.status === "cancelled") return;
|
|
17279
17735
|
if (output.status === "completed" && output.output && output.outputCid) {
|
|
17280
|
-
|
|
17281
|
-
output
|
|
17282
|
-
|
|
17283
|
-
|
|
17284
|
-
|
|
17285
|
-
|
|
17736
|
+
try {
|
|
17737
|
+
await agent.tasks.complete(output.taskId, output.attemptN, {
|
|
17738
|
+
output: output.output,
|
|
17739
|
+
outputCid: output.outputCid,
|
|
17740
|
+
usage: output.usage,
|
|
17741
|
+
...output.contentSignature ? { contentSignature: output.contentSignature } : {}
|
|
17742
|
+
});
|
|
17743
|
+
} catch (err) {
|
|
17744
|
+
const reason = errorToFailReason(err);
|
|
17745
|
+
ctx.log?.("complete-rejected-falling-back-to-fail", err);
|
|
17746
|
+
await agent.tasks.fail(output.taskId, output.attemptN, { error: reason });
|
|
17747
|
+
return;
|
|
17748
|
+
}
|
|
17286
17749
|
await maybeWriteAnchors(output, ctx);
|
|
17287
17750
|
return;
|
|
17288
17751
|
}
|
|
@@ -17294,6 +17757,21 @@ async function finalizeTask(agent, output, ctx = {}) {
|
|
|
17294
17757
|
if ((await agent.tasks.heartbeat(output.taskId, output.attemptN, {})).cancelled) return;
|
|
17295
17758
|
await agent.tasks.fail(output.taskId, output.attemptN, { error });
|
|
17296
17759
|
}
|
|
17760
|
+
function errorToFailReason(err) {
|
|
17761
|
+
if (err instanceof MoltNetError) {
|
|
17762
|
+
const fields = err.validationErrors?.length ? "; " + err.validationErrors.map((e) => `${e.field}: ${e.message}`).join(" | ") : "";
|
|
17763
|
+
return {
|
|
17764
|
+
code: "output_rejected_by_server",
|
|
17765
|
+
message: `Server rejected tasks.complete (${err.code}, status ${err.statusCode ?? "?"}): ${err.detail ?? err.message}${fields}`,
|
|
17766
|
+
retryable: false
|
|
17767
|
+
};
|
|
17768
|
+
}
|
|
17769
|
+
return {
|
|
17770
|
+
code: "complete_call_failed",
|
|
17771
|
+
message: err instanceof Error ? err.message : String(err),
|
|
17772
|
+
retryable: false
|
|
17773
|
+
};
|
|
17774
|
+
}
|
|
17297
17775
|
async function maybeWriteAnchors(output, ctx) {
|
|
17298
17776
|
const { task, writeCorrelationAnchors, log } = ctx;
|
|
17299
17777
|
if (!task || task.taskType !== "fulfill_brief") return;
|
|
@@ -17595,7 +18073,8 @@ async function runPolling(opts) {
|
|
|
17595
18073
|
const executionPlans = createExecutionPlanCache({
|
|
17596
18074
|
stateDirs,
|
|
17597
18075
|
slotIdentity,
|
|
17598
|
-
warmSessionTtlSec: common.warmSessionTtlSec
|
|
18076
|
+
warmSessionTtlSec: common.warmSessionTtlSec,
|
|
18077
|
+
slotRegistry
|
|
17599
18078
|
});
|
|
17600
18079
|
const ctx = await resolveAgentContext(common.agent);
|
|
17601
18080
|
const cfg = loadConfig();
|
|
@@ -17640,15 +18119,9 @@ async function runPolling(opts) {
|
|
|
17640
18119
|
maxPollIntervalMs
|
|
17641
18120
|
}, "agent-daemon.starting");
|
|
17642
18121
|
const outputs = [];
|
|
17643
|
-
const subagentContractRegistry = createSubagentContractRegistry([{
|
|
17644
|
-
name: "judge_eval_variant_result",
|
|
17645
|
-
description: "Per-variant grading result produced by a subagent of judge_eval_variant: scores against the shared rubric, composite, and a 1-3 sentence verdict for a single variant.",
|
|
17646
|
-
parametersSchema: JudgeEvalVariantResult
|
|
17647
|
-
}]);
|
|
17648
18122
|
try {
|
|
17649
18123
|
const executeTask = createPiTaskExecutor({
|
|
17650
18124
|
agentName: common.agent,
|
|
17651
|
-
subagentContractRegistry,
|
|
17652
18125
|
mountPath: sandbox.rootDir,
|
|
17653
18126
|
provider: common.provider,
|
|
17654
18127
|
model: common.model,
|
|
@@ -17692,7 +18165,34 @@ async function runPolling(opts) {
|
|
|
17692
18165
|
log: (msg, err) => rootLogger.warn({ err }, msg)
|
|
17693
18166
|
}),
|
|
17694
18167
|
executeTask: async (claimedTask, reporter) => {
|
|
17695
|
-
|
|
18168
|
+
let executionPlan;
|
|
18169
|
+
try {
|
|
18170
|
+
executionPlan = executionPlans.getOrCreate(claimedTask);
|
|
18171
|
+
} catch (err) {
|
|
18172
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
18173
|
+
rootLogger.warn({
|
|
18174
|
+
taskId: claimedTask.task.id,
|
|
18175
|
+
attemptN: claimedTask.attemptN,
|
|
18176
|
+
err: message
|
|
18177
|
+
}, "agent-daemon.execution_plan_failed");
|
|
18178
|
+
return {
|
|
18179
|
+
taskId: claimedTask.task.id,
|
|
18180
|
+
attemptN: claimedTask.attemptN,
|
|
18181
|
+
status: "failed",
|
|
18182
|
+
output: null,
|
|
18183
|
+
outputCid: null,
|
|
18184
|
+
usage: {
|
|
18185
|
+
inputTokens: 0,
|
|
18186
|
+
outputTokens: 0
|
|
18187
|
+
},
|
|
18188
|
+
durationMs: 0,
|
|
18189
|
+
error: {
|
|
18190
|
+
code: err instanceof ProducerContextResolutionError ? "producer_context_missing" : "execution_plan_failed",
|
|
18191
|
+
message,
|
|
18192
|
+
retryable: false
|
|
18193
|
+
}
|
|
18194
|
+
};
|
|
18195
|
+
}
|
|
17696
18196
|
const sessionDescriptor = executionPlan.descriptor;
|
|
17697
18197
|
let expired;
|
|
17698
18198
|
try {
|
|
@@ -17711,7 +18211,7 @@ async function runPolling(opts) {
|
|
|
17711
18211
|
taskId: claimedTask.task.id,
|
|
17712
18212
|
taskType: claimedTask.task.taskType,
|
|
17713
18213
|
resumable: sessionDescriptor.policy.resumable,
|
|
17714
|
-
workspaceMode:
|
|
18214
|
+
workspaceMode: executionPlan.workspaceMode,
|
|
17715
18215
|
workspaceScope: sessionDescriptor.policy.workspaceScope,
|
|
17716
18216
|
sessionScope: sessionDescriptor.policy.sessionScope,
|
|
17717
18217
|
slotKey: executionPlan.slotKey,
|
|
@@ -17761,7 +18261,7 @@ async function runPolling(opts) {
|
|
|
17761
18261
|
sessionDir: executionPlan.sessionPersistence.sessionDir,
|
|
17762
18262
|
sessionPath: resolveLatestPiSessionPath(executionPlan.sessionPersistence.sessionDir),
|
|
17763
18263
|
workspaceId: executionPlan.workspaceId,
|
|
17764
|
-
worktreePath:
|
|
18264
|
+
worktreePath: resolveRecordedWorkspacePath$1(mainRepo, stateDirs.rootDir, executionPlan),
|
|
17765
18265
|
worktreeBranch: executionPlan.worktreeBranch,
|
|
17766
18266
|
lastTaskId: claimedTask.task.id,
|
|
17767
18267
|
lastAttemptN: claimedTask.attemptN,
|
|
@@ -17785,6 +18285,10 @@ async function runPolling(opts) {
|
|
|
17785
18285
|
await shutdownLogger();
|
|
17786
18286
|
}
|
|
17787
18287
|
}
|
|
18288
|
+
function resolveRecordedWorkspacePath$1(mainRepo, stateRootDir, executionPlan) {
|
|
18289
|
+
if (!executionPlan.workspaceId) return null;
|
|
18290
|
+
return executionPlan.workspaceMode === "scratch_mount" ? join(stateRootDir, "task-workspaces", executionPlan.workspaceId) : join(mainRepo, ".worktrees", executionPlan.workspaceId);
|
|
18291
|
+
}
|
|
17788
18292
|
function parseCsv(raw) {
|
|
17789
18293
|
return (raw ?? "").split(",").map((s) => s.trim()).filter((s) => s.length > 0);
|
|
17790
18294
|
}
|
|
@@ -17852,7 +18356,8 @@ async function runOnce(argv) {
|
|
|
17852
18356
|
const executionPlans = createExecutionPlanCache({
|
|
17853
18357
|
stateDirs,
|
|
17854
18358
|
slotIdentity,
|
|
17855
|
-
warmSessionTtlSec: opts.warmSessionTtlSec
|
|
18359
|
+
warmSessionTtlSec: opts.warmSessionTtlSec,
|
|
18360
|
+
slotRegistry
|
|
17856
18361
|
});
|
|
17857
18362
|
const ctx = await resolveAgentContext(opts.agent);
|
|
17858
18363
|
const cfg = loadConfig();
|
|
@@ -17901,15 +18406,9 @@ async function runOnce(argv) {
|
|
|
17901
18406
|
process.on("SIGTERM", () => {
|
|
17902
18407
|
onSignal("SIGTERM");
|
|
17903
18408
|
});
|
|
17904
|
-
const subagentContractRegistry = createSubagentContractRegistry([{
|
|
17905
|
-
name: "judge_eval_variant_result",
|
|
17906
|
-
description: "Per-variant grading result produced by a subagent of judge_eval_variant: scores against the shared rubric, composite, and a 1-3 sentence verdict for a single variant.",
|
|
17907
|
-
parametersSchema: JudgeEvalVariantResult
|
|
17908
|
-
}]);
|
|
17909
18409
|
try {
|
|
17910
18410
|
const rawExecuteTask = createPiTaskExecutor({
|
|
17911
18411
|
agentName: opts.agent,
|
|
17912
|
-
subagentContractRegistry,
|
|
17913
18412
|
mountPath: sandbox.rootDir,
|
|
17914
18413
|
provider: opts.provider,
|
|
17915
18414
|
model: opts.model,
|
|
@@ -17932,7 +18431,34 @@ async function runOnce(argv) {
|
|
|
17932
18431
|
err: err instanceof Error ? err.message : String(err)
|
|
17933
18432
|
}, "agent-daemon.daemon_slot_reap_failed");
|
|
17934
18433
|
}
|
|
17935
|
-
|
|
18434
|
+
let executionPlan;
|
|
18435
|
+
try {
|
|
18436
|
+
executionPlan = executionPlans.getOrCreate(claimedTask);
|
|
18437
|
+
} catch (err) {
|
|
18438
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
18439
|
+
rootLogger.warn({
|
|
18440
|
+
taskId: claimedTask.task.id,
|
|
18441
|
+
attemptN: claimedTask.attemptN,
|
|
18442
|
+
err: message
|
|
18443
|
+
}, "agent-daemon.execution_plan_failed");
|
|
18444
|
+
return {
|
|
18445
|
+
taskId: claimedTask.task.id,
|
|
18446
|
+
attemptN: claimedTask.attemptN,
|
|
18447
|
+
status: "failed",
|
|
18448
|
+
output: null,
|
|
18449
|
+
outputCid: null,
|
|
18450
|
+
usage: {
|
|
18451
|
+
inputTokens: 0,
|
|
18452
|
+
outputTokens: 0
|
|
18453
|
+
},
|
|
18454
|
+
durationMs: 0,
|
|
18455
|
+
error: {
|
|
18456
|
+
code: err instanceof ProducerContextResolutionError ? "producer_context_missing" : "execution_plan_failed",
|
|
18457
|
+
message,
|
|
18458
|
+
retryable: false
|
|
18459
|
+
}
|
|
18460
|
+
};
|
|
18461
|
+
}
|
|
17936
18462
|
if (executionPlan.slotKey && executionPlan.sessionPersistence) slotRegistry.beginSlot({
|
|
17937
18463
|
...slotIdentity,
|
|
17938
18464
|
slotKey: executionPlan.slotKey,
|
|
@@ -17940,7 +18466,7 @@ async function runOnce(argv) {
|
|
|
17940
18466
|
sessionDir: executionPlan.sessionPersistence.sessionDir,
|
|
17941
18467
|
sessionPath: resolveLatestPiSessionPath(executionPlan.sessionPersistence.sessionDir),
|
|
17942
18468
|
workspaceId: executionPlan.workspaceId,
|
|
17943
|
-
worktreePath:
|
|
18469
|
+
worktreePath: resolveRecordedWorkspacePath(mainRepo, stateDirs.rootDir, executionPlan),
|
|
17944
18470
|
worktreeBranch: executionPlan.worktreeBranch,
|
|
17945
18471
|
lastTaskId: claimedTask.task.id,
|
|
17946
18472
|
lastAttemptN: claimedTask.attemptN,
|
|
@@ -17992,6 +18518,10 @@ async function runOnce(argv) {
|
|
|
17992
18518
|
await shutdownLogger();
|
|
17993
18519
|
}
|
|
17994
18520
|
}
|
|
18521
|
+
function resolveRecordedWorkspacePath(mainRepo, stateRootDir, executionPlan) {
|
|
18522
|
+
if (!executionPlan.workspaceId) return null;
|
|
18523
|
+
return executionPlan.workspaceMode === "scratch_mount" ? join(stateRootDir, "task-workspaces", executionPlan.workspaceId) : join(mainRepo, ".worktrees", executionPlan.workspaceId);
|
|
18524
|
+
}
|
|
17995
18525
|
//#endregion
|
|
17996
18526
|
//#region src/cli/poll.ts
|
|
17997
18527
|
function runPoll(argv) {
|