@themoltnet/agent-daemon 0.10.4 → 0.10.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/main.js +747 -476
- package/package.json +6 -6
package/dist/main.js
CHANGED
|
@@ -3060,7 +3060,7 @@ function validateRubricWeights(rubric) {
|
|
|
3060
3060
|
* attaches to any task type. It has four orthogonal sections — pick
|
|
3061
3061
|
* whichever apply per task type:
|
|
3062
3062
|
*
|
|
3063
|
-
* - `gates`
|
|
3063
|
+
* - `gates` Promise-level structural/process checks
|
|
3064
3064
|
* - `assertions` Declarative claims about output JSON
|
|
3065
3065
|
* - `rubric` Weighted-criteria scoring instrument, reused
|
|
3066
3066
|
* verbatim from `./rubric.ts`.
|
|
@@ -3105,17 +3105,27 @@ var CidEqualsSpec = Type$2.Object({
|
|
|
3105
3105
|
path: Type$2.String({ minLength: 1 }),
|
|
3106
3106
|
expected: Type$2.String({ minLength: 1 })
|
|
3107
3107
|
}, { additionalProperties: false });
|
|
3108
|
-
var
|
|
3108
|
+
var SubmitToolCallGate = Type$2.Object({
|
|
3109
3109
|
id: Type$2.String({ minLength: 1 }),
|
|
3110
|
-
kind: Type$2.Literal("
|
|
3111
|
-
|
|
3112
|
-
required: Type$2.Boolean()
|
|
3113
|
-
}, { additionalProperties: false }), Type$2.Object({
|
|
3114
|
-
id: Type$2.String({ minLength: 1 }),
|
|
3115
|
-
kind: Type$2.Literal("cid-equals"),
|
|
3116
|
-
spec: CidEqualsSpec,
|
|
3110
|
+
kind: Type$2.Literal("submit-tool-call"),
|
|
3111
|
+
description: Type$2.String({ minLength: 1 }),
|
|
3117
3112
|
required: Type$2.Boolean()
|
|
3118
|
-
}, { additionalProperties: false })
|
|
3113
|
+
}, { additionalProperties: false });
|
|
3114
|
+
var Gate = Type$2.Union([
|
|
3115
|
+
SubmitToolCallGate,
|
|
3116
|
+
Type$2.Object({
|
|
3117
|
+
id: Type$2.String({ minLength: 1 }),
|
|
3118
|
+
kind: Type$2.Literal("schema-check"),
|
|
3119
|
+
spec: SchemaCheckSpec,
|
|
3120
|
+
required: Type$2.Boolean()
|
|
3121
|
+
}, { additionalProperties: false }),
|
|
3122
|
+
Type$2.Object({
|
|
3123
|
+
id: Type$2.String({ minLength: 1 }),
|
|
3124
|
+
kind: Type$2.Literal("cid-equals"),
|
|
3125
|
+
spec: CidEqualsSpec,
|
|
3126
|
+
required: Type$2.Boolean()
|
|
3127
|
+
}, { additionalProperties: false })
|
|
3128
|
+
], { $id: "Gate" });
|
|
3119
3129
|
var AssertionOp = Type$2.Union([
|
|
3120
3130
|
Type$2.Literal("exists"),
|
|
3121
3131
|
Type$2.Literal("equals"),
|
|
@@ -6305,6 +6315,32 @@ function submitOutputToolName(taskType) {
|
|
|
6305
6315
|
return `submit_${taskType}_output`;
|
|
6306
6316
|
}
|
|
6307
6317
|
//#endregion
|
|
6318
|
+
//#region ../../libs/agent-runtime/src/prompts/assemble.ts
|
|
6319
|
+
/**
|
|
6320
|
+
* Render a `PromptSection[]` into final text + structured trace.
|
|
6321
|
+
* Single source of truth for inter-section spacing and header
|
|
6322
|
+
* rendering across all task types.
|
|
6323
|
+
*/
|
|
6324
|
+
function assembleTaskPrompt(taskType, sections) {
|
|
6325
|
+
const trace = [];
|
|
6326
|
+
const rendered = [];
|
|
6327
|
+
for (const section of sections) {
|
|
6328
|
+
trace.push({
|
|
6329
|
+
id: section.id,
|
|
6330
|
+
source: section.source,
|
|
6331
|
+
header: section.header,
|
|
6332
|
+
char_count: section.body.length
|
|
6333
|
+
});
|
|
6334
|
+
if (section.body === "") continue;
|
|
6335
|
+
rendered.push(section.header ? `## ${section.header}\n\n${section.body}` : section.body);
|
|
6336
|
+
}
|
|
6337
|
+
return {
|
|
6338
|
+
text: rendered.join("\n\n"),
|
|
6339
|
+
trace,
|
|
6340
|
+
taskType
|
|
6341
|
+
};
|
|
6342
|
+
}
|
|
6343
|
+
//#endregion
|
|
6308
6344
|
//#region ../../libs/agent-runtime/src/prompts/final-output.ts
|
|
6309
6345
|
function buildFinalOutputBlock(opts) {
|
|
6310
6346
|
const { taskType, outputSchemaName, shapeSketch, extraNotes } = opts;
|
|
@@ -6319,7 +6355,8 @@ function buildFinalOutputBlock(opts) {
|
|
|
6319
6355
|
`The runtime captures the validated arguments and ends the session.`,
|
|
6320
6356
|
`Do NOT emit the output as plain assistant text. Do NOT rely on a`,
|
|
6321
6357
|
`JSON-in-message fallback. If you do not call \`${submitTool}\`, the`,
|
|
6322
|
-
`attempt
|
|
6358
|
+
`attempt is recorded as failing the promised submit-output criterion`,
|
|
6359
|
+
`even if the underlying work succeeded.`,
|
|
6323
6360
|
"",
|
|
6324
6361
|
`Your final assistant text before that tool call may explain your work,`,
|
|
6325
6362
|
`but the submit-tool call itself must be your VERY LAST action.`,
|
|
@@ -6357,37 +6394,17 @@ function renderRubricPreambleSection(rubric) {
|
|
|
6357
6394
|
*
|
|
6358
6395
|
* Design note — no pre-resolved `target` projection
|
|
6359
6396
|
* --------------------------------------------------
|
|
6360
|
-
* Earlier drafts hand-wired a `target` bundle (branch, PR url,
|
|
6361
|
-
*
|
|
6362
|
-
*
|
|
6363
|
-
*
|
|
6364
|
-
*
|
|
6365
|
-
*
|
|
6366
|
-
*
|
|
6367
|
-
* fetching their own data.
|
|
6368
|
-
*
|
|
6369
|
-
* Now: the prompt tells the judge the `targetTaskId` and instructs
|
|
6370
|
-
* it to call `moltnet_get_task` + `moltnet_list_task_attempts`
|
|
6371
|
-
* itself. The judge sees whatever the producer's accepted attempt
|
|
6372
|
-
* actually wrote — no projection, no lossiness, no daemon-side
|
|
6373
|
-
* type knowledge required. Different producers (fulfill_brief,
|
|
6374
|
-
* future task types whose products are docs / configs / changes /
|
|
6375
|
-
* anything) work without any code path here.
|
|
6397
|
+
* Earlier drafts hand-wired a `target` bundle (branch, PR url, commits,
|
|
6398
|
+
* summary, diary entry ids) into the prompt before the judge started.
|
|
6399
|
+
* That coupled the daemon to one specific producer shape, forced every
|
|
6400
|
+
* executor to know how to project it, and went stale every time a
|
|
6401
|
+
* producer task type grew a field. Now: the prompt tells the judge
|
|
6402
|
+
* the `targetTaskId` and instructs it to call `moltnet_get_task` +
|
|
6403
|
+
* `moltnet_list_task_attempts` itself.
|
|
6376
6404
|
*/
|
|
6377
6405
|
function buildAssessBriefUserPrompt(input, ctx) {
|
|
6378
6406
|
const rubric = input.successCriteria.rubric;
|
|
6379
|
-
const
|
|
6380
|
-
const preambleSection = renderRubricPreambleSection(rubric) ?? "";
|
|
6381
|
-
const workspaceSection = ctx.workspace?.mode === "dedicated_worktree" ? [
|
|
6382
|
-
"### Workspace",
|
|
6383
|
-
"",
|
|
6384
|
-
"This review attempt is running inside a dedicated disposable git",
|
|
6385
|
-
"worktree created for this task. If you need to check out the target",
|
|
6386
|
-
"branch or inspect refs locally, do it only inside this worktree.",
|
|
6387
|
-
ctx.workspace.branch ? `The current review branch is \`${ctx.workspace.branch}\`. You may replace it with the target branch locally if that helps your inspection.` : "The current checkout is disposable and will be cleaned up when the task ends.",
|
|
6388
|
-
""
|
|
6389
|
-
].join("\n") : "";
|
|
6390
|
-
return [
|
|
6407
|
+
const header = [
|
|
6391
6408
|
"# Assess Brief Judge",
|
|
6392
6409
|
"",
|
|
6393
6410
|
"You are an independent judge. You did NOT produce the work under review.",
|
|
@@ -6395,10 +6412,9 @@ function buildAssessBriefUserPrompt(input, ctx) {
|
|
|
6395
6412
|
"You may read code, commits, and diary entries — but do NOT modify anything.",
|
|
6396
6413
|
"",
|
|
6397
6414
|
`Your diary ID is: ${ctx.diaryId}`,
|
|
6398
|
-
`This task's id is: ${ctx.taskId}
|
|
6399
|
-
|
|
6400
|
-
|
|
6401
|
-
"",
|
|
6415
|
+
`This task's id is: ${ctx.taskId}`
|
|
6416
|
+
].join("\n");
|
|
6417
|
+
const target = [
|
|
6402
6418
|
`**Producer task id:** \`${input.targetTaskId}\``,
|
|
6403
6419
|
"",
|
|
6404
6420
|
"Investigate the producer task before scoring:",
|
|
@@ -6411,10 +6427,9 @@ function buildAssessBriefUserPrompt(input, ctx) {
|
|
|
6411
6427
|
" - `commits[].sha` listed → use `git show <sha>` for individual commits.",
|
|
6412
6428
|
" - `diaryEntryIds[]` listed → fetch each via `moltnet_get_entry` to read the producer's reasoning.",
|
|
6413
6429
|
" - `summary` set → use as orientation, not as ground truth.",
|
|
6414
|
-
"Adapt your investigation to whatever the output actually contains. Score conservatively when the producer's output is opaque or thin."
|
|
6415
|
-
|
|
6416
|
-
|
|
6417
|
-
"",
|
|
6430
|
+
"Adapt your investigation to whatever the output actually contains. Score conservatively when the producer's output is opaque or thin."
|
|
6431
|
+
].join("\n");
|
|
6432
|
+
const diaryQuery = [
|
|
6418
6433
|
`Beyond the explicit \`diaryEntryIds[]\` from step 3, the producer's`,
|
|
6419
6434
|
"attempts auto-tag every entry with the `task:*` provenance namespace.",
|
|
6420
6435
|
"You can pull the full set without enumerating ids by passing the",
|
|
@@ -6425,38 +6440,84 @@ function buildAssessBriefUserPrompt(input, ctx) {
|
|
|
6425
6440
|
"- Just the accepted attempt: add `attemptN: <acceptedAttemptN>`.",
|
|
6426
6441
|
"- The producer plus any prior chain (when a correlationId was set):",
|
|
6427
6442
|
" read it from the task you fetched in step 1 and pass",
|
|
6428
|
-
" `taskFilter: { correlationId: \"<id>\" }`."
|
|
6429
|
-
|
|
6430
|
-
|
|
6431
|
-
|
|
6432
|
-
"
|
|
6433
|
-
"",
|
|
6434
|
-
|
|
6435
|
-
|
|
6436
|
-
|
|
6437
|
-
|
|
6443
|
+
" `taskFilter: { correlationId: \"<id>\" }`."
|
|
6444
|
+
].join("\n");
|
|
6445
|
+
const workspace = ctx.workspace?.mode === "dedicated_worktree" ? [
|
|
6446
|
+
"This review attempt is running inside a dedicated disposable git",
|
|
6447
|
+
"worktree created for this task. If you need to check out the target",
|
|
6448
|
+
"branch or inspect refs locally, do it only inside this worktree.",
|
|
6449
|
+
ctx.workspace.branch ? `The current review branch is \`${ctx.workspace.branch}\`. You may replace it with the target branch locally if that helps your inspection.` : "The current checkout is disposable and will be cleaned up when the task ends."
|
|
6450
|
+
].join("\n") : "";
|
|
6451
|
+
const preamble = renderRubricPreambleSection(rubric) ?? "";
|
|
6452
|
+
const criteria = renderRubricCriteriaList(rubric);
|
|
6453
|
+
const scoring = [
|
|
6438
6454
|
"- `llm_score`: score 0..1 continuous. `rationale` REQUIRED (2–4 sentences).",
|
|
6439
6455
|
"- `boolean`: score exactly 0 or 1. `rationale` optional.",
|
|
6440
6456
|
"- `deterministic_signature_check`: run `moltnet entry verify` on every diary entry returned by step 3 above AND `git verify-commit` on every commit. Score 1 iff ALL signatures are valid; otherwise 0. Populate `evidence.commitsVerified`, `evidence.commitsTotal`, `evidence.signatureFailures`.",
|
|
6441
6457
|
"",
|
|
6442
|
-
"Write a signed diary entry (tags: \"judgment\", \"assess_brief\") capturing the rationale before reporting structured output."
|
|
6443
|
-
|
|
6444
|
-
|
|
6445
|
-
|
|
6446
|
-
|
|
6447
|
-
|
|
6448
|
-
|
|
6449
|
-
|
|
6450
|
-
|
|
6451
|
-
|
|
6452
|
-
|
|
6453
|
-
|
|
6454
|
-
|
|
6455
|
-
|
|
6456
|
-
|
|
6457
|
-
|
|
6458
|
-
|
|
6459
|
-
|
|
6458
|
+
"Write a signed diary entry (tags: \"judgment\", \"assess_brief\") capturing the rationale before reporting structured output."
|
|
6459
|
+
].join("\n");
|
|
6460
|
+
return assembleTaskPrompt("assess_brief", [
|
|
6461
|
+
{
|
|
6462
|
+
id: "assess_brief.header",
|
|
6463
|
+
source: "header",
|
|
6464
|
+
body: header
|
|
6465
|
+
},
|
|
6466
|
+
{
|
|
6467
|
+
id: "assess_brief.target",
|
|
6468
|
+
source: "task_input",
|
|
6469
|
+
header: "Target of assessment",
|
|
6470
|
+
body: target
|
|
6471
|
+
},
|
|
6472
|
+
{
|
|
6473
|
+
id: "assess_brief.diary_query",
|
|
6474
|
+
source: "static",
|
|
6475
|
+
header: "Querying the producer's diary entries",
|
|
6476
|
+
body: diaryQuery
|
|
6477
|
+
},
|
|
6478
|
+
{
|
|
6479
|
+
id: "assess_brief.workspace",
|
|
6480
|
+
source: "workspace",
|
|
6481
|
+
header: "Workspace",
|
|
6482
|
+
body: workspace
|
|
6483
|
+
},
|
|
6484
|
+
{
|
|
6485
|
+
id: "assess_brief.preamble",
|
|
6486
|
+
source: "rubric_judge",
|
|
6487
|
+
body: preamble
|
|
6488
|
+
},
|
|
6489
|
+
{
|
|
6490
|
+
id: "assess_brief.criteria",
|
|
6491
|
+
source: "rubric_judge",
|
|
6492
|
+
header: "Criteria",
|
|
6493
|
+
body: criteria
|
|
6494
|
+
},
|
|
6495
|
+
{
|
|
6496
|
+
id: "assess_brief.scoring",
|
|
6497
|
+
source: "rubric_judge",
|
|
6498
|
+
header: "Scoring rules",
|
|
6499
|
+
body: scoring
|
|
6500
|
+
},
|
|
6501
|
+
{
|
|
6502
|
+
id: "assess_brief.final_output",
|
|
6503
|
+
source: "final_output",
|
|
6504
|
+
body: buildFinalOutputBlock({
|
|
6505
|
+
taskType: "assess_brief",
|
|
6506
|
+
outputSchemaName: "AssessBriefOutput",
|
|
6507
|
+
shapeSketch: [
|
|
6508
|
+
"{",
|
|
6509
|
+
" \"scores\": [",
|
|
6510
|
+
" { \"criterionId\": \"...\", \"score\": 0.0, \"rationale\": \"...\", \"evidence\": {} }",
|
|
6511
|
+
" ],",
|
|
6512
|
+
" \"composite\": <sum>,",
|
|
6513
|
+
" \"verdict\": \"<1-3 sentence overall>\",",
|
|
6514
|
+
" \"judgeModel\": \"<provider:model>\"",
|
|
6515
|
+
"}"
|
|
6516
|
+
].join("\n"),
|
|
6517
|
+
extraNotes: ["`composite` = Σ(weight_i × score_i) recomputed. The runtime rejects a mismatch."]
|
|
6518
|
+
})
|
|
6519
|
+
}
|
|
6520
|
+
]);
|
|
6460
6521
|
}
|
|
6461
6522
|
//#endregion
|
|
6462
6523
|
//#region ../../libs/agent-runtime/src/prompts/self-verification.ts
|
|
@@ -6465,11 +6526,11 @@ function buildSelfVerificationBlock(taskId, criteriaField = "successCriteria") {
|
|
|
6465
6526
|
"## Self-verification",
|
|
6466
6527
|
"",
|
|
6467
6528
|
`If \`input.${criteriaField}\` is set on this task, your final output MUST`,
|
|
6468
|
-
"include a `verification` block.
|
|
6469
|
-
|
|
6470
|
-
"
|
|
6471
|
-
"
|
|
6472
|
-
"
|
|
6529
|
+
"include a `verification` block. Treat every item in those criteria as",
|
|
6530
|
+
"part of the promise you made when you claimed the task. That includes",
|
|
6531
|
+
"the built-in submit-output gate when present. Do not call the submit",
|
|
6532
|
+
"tool until you have computed the verification payload you can honestly",
|
|
6533
|
+
"stand behind.",
|
|
6473
6534
|
"",
|
|
6474
6535
|
`Call \`moltnet_get_task\` with task id \`${taskId}\` and read \`input.${criteriaField}\`.`,
|
|
6475
6536
|
"",
|
|
@@ -6533,22 +6594,13 @@ function buildSelfVerificationBlock(taskId, criteriaField = "successCriteria") {
|
|
|
6533
6594
|
* TODO(#885): add a `moltnet_parallel_explore` custom tool that spawns
|
|
6534
6595
|
* N isolated `createAgentSession` children (one per tag cluster or
|
|
6535
6596
|
* entry_type axis the curator picks after recon), each with a narrow
|
|
6536
|
-
* tool subset and a turn cap, and returns compressed summaries.
|
|
6537
|
-
* curator keeps a warm context and only sees {candidateIds, notes}
|
|
6538
|
-
* per probe — mirrors the fan-out pattern pi-mono SDK example #13
|
|
6539
|
-
* (session runtime) + #05 (custom tools) makes possible. Until that
|
|
6540
|
-
* lands, the `checkpoints[]` output field is the fallback: curator
|
|
6541
|
-
* emits pruned state at phase boundaries so a follow-up session can
|
|
6542
|
-
* resume without replaying the tool history.
|
|
6597
|
+
* tool subset and a turn cap, and returns compressed summaries.
|
|
6543
6598
|
*/
|
|
6544
6599
|
function buildCuratePackUserPrompt(input, ctx) {
|
|
6545
6600
|
const { diaryId, taskPrompt, entryTypes, tagFilters, tokenBudget, recipe } = input;
|
|
6546
6601
|
const entryTypesPinned = Boolean(entryTypes);
|
|
6547
6602
|
const resolvedRecipe = recipe ?? "topic-focused-v1";
|
|
6548
|
-
const
|
|
6549
|
-
const excludeLine = tagFilters?.exclude?.length ? `- Hard exclude (drop if ANY present): ${tagFilters.exclude.map((t) => `\`${t}\``).join(", ")}` : null;
|
|
6550
|
-
const prefixLine = tagFilters?.prefix ? `- Tag prefix hint when inventorying: \`${tagFilters.prefix}\`` : null;
|
|
6551
|
-
return [
|
|
6603
|
+
const header = [
|
|
6552
6604
|
"# Curate Pack Agent",
|
|
6553
6605
|
"",
|
|
6554
6606
|
"You are the curator. Step 1 of the three-session attribution loop:",
|
|
@@ -6556,40 +6608,29 @@ function buildCuratePackUserPrompt(input, ctx) {
|
|
|
6556
6608
|
"will judge. Your output IS the pack — nobody downstream will re-rank.",
|
|
6557
6609
|
"",
|
|
6558
6610
|
`Your agent-session diary ID is: ${ctx.diaryId}`,
|
|
6559
|
-
`This task's id is: ${ctx.taskId}
|
|
6560
|
-
|
|
6561
|
-
|
|
6562
|
-
"",
|
|
6611
|
+
`This task's id is: ${ctx.taskId}`
|
|
6612
|
+
].join("\n");
|
|
6613
|
+
const goal = [
|
|
6563
6614
|
`Build a pack from diary \`${diaryId}\` that faithfully serves this`,
|
|
6564
|
-
|
|
6615
|
+
"prompt:",
|
|
6565
6616
|
"",
|
|
6566
6617
|
`> ${taskPrompt}`,
|
|
6567
6618
|
"",
|
|
6568
6619
|
"What \"faithfully\" means is your call. A broad prompt may warrant 20",
|
|
6569
6620
|
"entries spanning clusters; a sharp one may resolve to 4 high-signal",
|
|
6570
6621
|
"entries. Trust your own judgment on breadth vs. depth — but be able",
|
|
6571
|
-
"to defend it in the summary."
|
|
6572
|
-
|
|
6573
|
-
|
|
6574
|
-
|
|
6575
|
-
|
|
6576
|
-
|
|
6577
|
-
|
|
6578
|
-
|
|
6579
|
-
|
|
6580
|
-
|
|
6581
|
-
|
|
6582
|
-
|
|
6583
|
-
entryTypesPinned ? null : " style content (e.g., \"what shipped this week\"). State your choice",
|
|
6584
|
-
entryTypesPinned ? null : " briefly in the final `summary`.",
|
|
6585
|
-
`- Recipe tag: \`${resolvedRecipe}\` (recorded on pack params)`,
|
|
6586
|
-
tokenBudget ? `- Token budget (soft cap on final pack): ${tokenBudget}. Pick entry count so the pack fits — estimate ~300 tok/entry as a starting heuristic, tighten after inspecting actual content lengths.` : "- No token budget — size the pack to match the prompt, not an arbitrary target.",
|
|
6587
|
-
includeLine,
|
|
6588
|
-
excludeLine,
|
|
6589
|
-
prefixLine,
|
|
6590
|
-
"",
|
|
6591
|
-
"## Tools available (not a recipe — use what the situation calls for)",
|
|
6592
|
-
"",
|
|
6622
|
+
"to defend it in the summary."
|
|
6623
|
+
].join("\n");
|
|
6624
|
+
const constraintsLines = [];
|
|
6625
|
+
if (entryTypesPinned) constraintsLines.push(`- Entry types pinned by imposer (do not widen): ${entryTypes.map((t) => `\`${t}\``).join(", ")}`);
|
|
6626
|
+
else constraintsLines.push("- Entry types: **you choose**. The diary contains three kinds:", " - `episodic` — incident reports, \"what happened and how we fixed it\" narratives.", " - `semantic` — durable decisions, patterns, design rationale.", " - `procedural` — commit audit trails / changelog-style provenance.", " Pick the subset that fits the prompt. For \"failures and workarounds\"", " or \"decisions we made\" you generally do NOT want `procedural` — those", " entries are append-only commit logs and produce changelog-shaped packs.", " Include `procedural` only when the prompt explicitly asks for changelog-", " style content (e.g., \"what shipped this week\"). State your choice", " briefly in the final `summary`.");
|
|
6627
|
+
constraintsLines.push(`- Recipe tag: \`${resolvedRecipe}\` (recorded on pack params)`);
|
|
6628
|
+
constraintsLines.push(tokenBudget ? `- Token budget (soft cap on final pack): ${tokenBudget}. Pick entry count so the pack fits — estimate ~300 tok/entry as a starting heuristic, tighten after inspecting actual content lengths.` : "- No token budget — size the pack to match the prompt, not an arbitrary target.");
|
|
6629
|
+
if (tagFilters?.include?.length) constraintsLines.push(`- Hard include (ALL must be present on an entry): ${tagFilters.include.map((t) => `\`${t}\``).join(", ")}`);
|
|
6630
|
+
if (tagFilters?.exclude?.length) constraintsLines.push(`- Hard exclude (drop if ANY present): ${tagFilters.exclude.map((t) => `\`${t}\``).join(", ")}`);
|
|
6631
|
+
if (tagFilters?.prefix) constraintsLines.push(`- Tag prefix hint when inventorying: \`${tagFilters.prefix}\``);
|
|
6632
|
+
const constraints = constraintsLines.join("\n");
|
|
6633
|
+
const tools = [
|
|
6593
6634
|
"- `moltnet_diary_tags` — tag inventory with counts. Cheap reconnaissance",
|
|
6594
6635
|
" when the prompt implies a scope but not a tag. Pass",
|
|
6595
6636
|
" `prefix: \"task:\"` to enumerate task-provenance tags only",
|
|
@@ -6602,10 +6643,9 @@ function buildCuratePackUserPrompt(input, ctx) {
|
|
|
6602
6643
|
"- `moltnet_list_entries` — multi-tag (AND) listing with optional",
|
|
6603
6644
|
" `excludeTags`, `entryType`, and the same `taskFilter` shorthand.",
|
|
6604
6645
|
"- `moltnet_get_entry` — full entry read, for disambiguation.",
|
|
6605
|
-
"- `moltnet_pack_create` — terminal call that persists the pack."
|
|
6606
|
-
|
|
6607
|
-
|
|
6608
|
-
"",
|
|
6646
|
+
"- `moltnet_pack_create` — terminal call that persists the pack."
|
|
6647
|
+
].join("\n");
|
|
6648
|
+
const exploration = [
|
|
6609
6649
|
"Context is finite. Treat every tool call as buying information against",
|
|
6610
6650
|
"a budget. Some heuristics that tend to work:",
|
|
6611
6651
|
"",
|
|
@@ -6622,57 +6662,110 @@ function buildCuratePackUserPrompt(input, ctx) {
|
|
|
6622
6662
|
"- **Emit a checkpoint if your working set exceeds ~30 candidates.**",
|
|
6623
6663
|
" Write one to the `checkpoints` array (see Output) listing the ids",
|
|
6624
6664
|
" you're keeping and dropping, plus a note explaining the cut. This",
|
|
6625
|
-
" lets a follow-up session resume without replaying your tool history."
|
|
6626
|
-
|
|
6627
|
-
|
|
6628
|
-
"",
|
|
6665
|
+
" lets a follow-up session resume without replaying your tool history."
|
|
6666
|
+
].join("\n");
|
|
6667
|
+
const ranking = [
|
|
6629
6668
|
"Assign integer ranks 1..N, lower = more prominent. Rank reflects",
|
|
6630
6669
|
"relevance to the prompt, NOT recency or entry popularity. Each entry",
|
|
6631
6670
|
"in the output must carry a short `rationale` — one sentence pointing",
|
|
6632
|
-
"at what in its content earned the rank."
|
|
6633
|
-
|
|
6634
|
-
|
|
6635
|
-
"",
|
|
6671
|
+
"at what in its content earned the rank."
|
|
6672
|
+
].join("\n");
|
|
6673
|
+
const persisting = [
|
|
6636
6674
|
"Call `moltnet_pack_create` with:",
|
|
6637
6675
|
"- `entries`: `[{ entryId, rank }]` for each selected entry.",
|
|
6638
|
-
|
|
6676
|
+
`- \`params\`: \`{ recipe: "${resolvedRecipe}", prompt: <the task prompt>, selection_rationale: "<2-sentence summary>" }\`.`,
|
|
6639
6677
|
tokenBudget ? `- \`tokenBudget\`: ${tokenBudget}.` : "- `tokenBudget`: omit.",
|
|
6640
6678
|
"- `pinned: false` (packs in this pipeline are ephemeral by design).",
|
|
6641
6679
|
"",
|
|
6642
6680
|
"The tool returns a JSON payload whose top-level fields are `packId` and",
|
|
6643
6681
|
"`packCid` (NOT `id`). Copy those exact UUID/CID strings verbatim into",
|
|
6644
6682
|
"`packId` and `packCid` in your final output — do not substitute an",
|
|
6645
|
-
"entry id, do not reformat, do not fabricate a UUID."
|
|
6646
|
-
|
|
6647
|
-
|
|
6648
|
-
"",
|
|
6683
|
+
"entry id, do not reformat, do not fabricate a UUID."
|
|
6684
|
+
].join("\n");
|
|
6685
|
+
const hardConstraints = [
|
|
6649
6686
|
"- Do NOT call `moltnet_pack_render` — that belongs to the next session.",
|
|
6650
6687
|
"- Do NOT write diary entries unless curation surfaces a genuine",
|
|
6651
6688
|
" incident worth recording. The curation reasoning lives in the task",
|
|
6652
6689
|
" output, not in the diary.",
|
|
6653
|
-
"- Respect hard include/exclude filters literally."
|
|
6654
|
-
|
|
6655
|
-
|
|
6656
|
-
|
|
6657
|
-
|
|
6658
|
-
|
|
6659
|
-
|
|
6660
|
-
|
|
6661
|
-
|
|
6662
|
-
|
|
6663
|
-
|
|
6664
|
-
|
|
6665
|
-
|
|
6666
|
-
|
|
6667
|
-
|
|
6668
|
-
|
|
6669
|
-
|
|
6670
|
-
|
|
6671
|
-
|
|
6672
|
-
|
|
6673
|
-
|
|
6674
|
-
|
|
6675
|
-
|
|
6690
|
+
"- Respect hard include/exclude filters literally."
|
|
6691
|
+
].join("\n");
|
|
6692
|
+
return assembleTaskPrompt("curate_pack", [
|
|
6693
|
+
{
|
|
6694
|
+
id: "curate_pack.header",
|
|
6695
|
+
source: "header",
|
|
6696
|
+
body: header
|
|
6697
|
+
},
|
|
6698
|
+
{
|
|
6699
|
+
id: "curate_pack.goal",
|
|
6700
|
+
source: "task_input",
|
|
6701
|
+
header: "Goal",
|
|
6702
|
+
body: goal
|
|
6703
|
+
},
|
|
6704
|
+
{
|
|
6705
|
+
id: "curate_pack.constraints",
|
|
6706
|
+
source: "task_input",
|
|
6707
|
+
header: "Constraints",
|
|
6708
|
+
body: constraints
|
|
6709
|
+
},
|
|
6710
|
+
{
|
|
6711
|
+
id: "curate_pack.tools",
|
|
6712
|
+
source: "static",
|
|
6713
|
+
header: "Tools available (not a recipe — use what the situation calls for)",
|
|
6714
|
+
body: tools
|
|
6715
|
+
},
|
|
6716
|
+
{
|
|
6717
|
+
id: "curate_pack.exploration",
|
|
6718
|
+
source: "static",
|
|
6719
|
+
header: "Exploration discipline",
|
|
6720
|
+
body: exploration
|
|
6721
|
+
},
|
|
6722
|
+
{
|
|
6723
|
+
id: "curate_pack.ranking",
|
|
6724
|
+
source: "static",
|
|
6725
|
+
header: "Ranking",
|
|
6726
|
+
body: ranking
|
|
6727
|
+
},
|
|
6728
|
+
{
|
|
6729
|
+
id: "curate_pack.persisting",
|
|
6730
|
+
source: "static",
|
|
6731
|
+
header: "Persisting the pack",
|
|
6732
|
+
body: persisting
|
|
6733
|
+
},
|
|
6734
|
+
{
|
|
6735
|
+
id: "curate_pack.hard_constraints",
|
|
6736
|
+
source: "static",
|
|
6737
|
+
header: "Hard constraints",
|
|
6738
|
+
body: hardConstraints
|
|
6739
|
+
},
|
|
6740
|
+
{
|
|
6741
|
+
id: "curate_pack.verification",
|
|
6742
|
+
source: "verification",
|
|
6743
|
+
body: buildSelfVerificationBlock(ctx.taskId)
|
|
6744
|
+
},
|
|
6745
|
+
{
|
|
6746
|
+
id: "curate_pack.final_output",
|
|
6747
|
+
source: "final_output",
|
|
6748
|
+
body: buildFinalOutputBlock({
|
|
6749
|
+
taskType: "curate_pack",
|
|
6750
|
+
outputSchemaName: "CuratePackOutput",
|
|
6751
|
+
shapeSketch: [
|
|
6752
|
+
"{",
|
|
6753
|
+
" \"packId\": \"<uuid>\",",
|
|
6754
|
+
" \"packCid\": \"<cid>\",",
|
|
6755
|
+
" \"entries\": [",
|
|
6756
|
+
" { \"entryId\": \"<uuid>\", \"rank\": 1, \"rationale\": \"<why>\" }",
|
|
6757
|
+
" ],",
|
|
6758
|
+
" \"recipeParams\": { \"recipe\": \"...\", \"prompt\": \"...\", ... },",
|
|
6759
|
+
" \"checkpoints\": [",
|
|
6760
|
+
" { \"phase\": \"recon\", \"candidateIds\": [...], \"droppedIds\": [...], \"notes\": \"...\" }",
|
|
6761
|
+
" ],",
|
|
6762
|
+
" \"summary\": \"<2-4 sentences: what you looked for, how you narrowed, what defines the final set>\",",
|
|
6763
|
+
" \"verification\": <required iff input.successCriteria; see Self-verification>",
|
|
6764
|
+
"}"
|
|
6765
|
+
].join("\n")
|
|
6766
|
+
})
|
|
6767
|
+
}
|
|
6768
|
+
]);
|
|
6676
6769
|
}
|
|
6677
6770
|
//#endregion
|
|
6678
6771
|
//#region ../../libs/agent-runtime/src/prompts/fulfill-brief.ts
|
|
@@ -6685,17 +6778,22 @@ function buildCuratePackUserPrompt(input, ctx) {
|
|
|
6685
6778
|
*/
|
|
6686
6779
|
function buildFulfillBriefUserPrompt(input, ctx) {
|
|
6687
6780
|
const { brief, title, seedFiles, scopeHint } = input;
|
|
6688
|
-
const
|
|
6689
|
-
"
|
|
6781
|
+
const header = [
|
|
6782
|
+
"# Fulfill Brief Agent",
|
|
6690
6783
|
"",
|
|
6691
|
-
"
|
|
6692
|
-
|
|
6693
|
-
""
|
|
6694
|
-
|
|
6695
|
-
|
|
6696
|
-
|
|
6697
|
-
"
|
|
6784
|
+
"You are a software engineering agent working in a sandboxed environment.",
|
|
6785
|
+
"Your workspace is at /workspace (mounted from the host repository).",
|
|
6786
|
+
"The MoltNet runtime instructor (above, in this system prompt) defines the",
|
|
6787
|
+
"invariants for this task: identity, gh authentication, diary discipline,",
|
|
6788
|
+
"and the accountable-commit shape. Follow it for every commit.",
|
|
6789
|
+
"",
|
|
6790
|
+
`## Task: ${title ?? "Fulfill brief"}`,
|
|
6698
6791
|
"",
|
|
6792
|
+
`Task id: \`${ctx.taskId}\``
|
|
6793
|
+
].join("\n");
|
|
6794
|
+
const seedFilesBody = seedFiles?.length ? ["Start by reading these files to ground yourself:", ...seedFiles.map((f) => `- \`${f}\``)].join("\n") : "";
|
|
6795
|
+
const branchSlug = ctx.correlationId ? `moltnet/${ctx.correlationId}/` : scopeHint ? `feat/${scopeHint}-` : "feat/";
|
|
6796
|
+
const correlation = ctx.correlationId ? [
|
|
6699
6797
|
`This task carries correlationId \`${ctx.correlationId}\`. You MUST:`,
|
|
6700
6798
|
"",
|
|
6701
6799
|
`1. Name your branch \`moltnet/${ctx.correlationId}/<short-slug>\` — use a`,
|
|
@@ -6704,39 +6802,14 @@ function buildFulfillBriefUserPrompt(input, ctx) {
|
|
|
6704
6802
|
" your **first** commit on that branch (subsequent commits do not need it).",
|
|
6705
6803
|
"",
|
|
6706
6804
|
"These are recovery anchors for the MoltNet mention-bot. Do not deviate",
|
|
6707
|
-
"from this branch naming scheme when correlationId is set."
|
|
6708
|
-
""
|
|
6805
|
+
"from this branch naming scheme when correlationId is set."
|
|
6709
6806
|
].join("\n") : "";
|
|
6710
|
-
const
|
|
6711
|
-
"### Workspace",
|
|
6712
|
-
"",
|
|
6807
|
+
const workspace = ctx.workspace?.mode === "dedicated_worktree" ? [
|
|
6713
6808
|
"This attempt is running inside a dedicated git worktree created",
|
|
6714
6809
|
"for this task. Do not repurpose or switch the primary checkout.",
|
|
6715
|
-
ctx.workspace.branch ? `The current branch is \`${ctx.workspace.branch}\`. Stay on this branch unless the runtime instructor explicitly tells you otherwise.` : "Stay on the branch that was pre-provisioned for this task."
|
|
6716
|
-
""
|
|
6810
|
+
ctx.workspace.branch ? `The current branch is \`${ctx.workspace.branch}\`. Stay on this branch unless the runtime instructor explicitly tells you otherwise.` : "Stay on the branch that was pre-provisioned for this task."
|
|
6717
6811
|
].join("\n") : "";
|
|
6718
|
-
|
|
6719
|
-
"# Fulfill Brief Agent",
|
|
6720
|
-
"",
|
|
6721
|
-
"You are a software engineering agent working in a sandboxed environment.",
|
|
6722
|
-
"Your workspace is at /workspace (mounted from the host repository).",
|
|
6723
|
-
"The MoltNet runtime instructor (above, in this system prompt) defines the",
|
|
6724
|
-
"invariants for this task: identity, gh authentication, diary discipline,",
|
|
6725
|
-
"and the accountable-commit shape. Follow it for every commit.",
|
|
6726
|
-
"",
|
|
6727
|
-
`## Task: ${title ?? "Fulfill brief"}`,
|
|
6728
|
-
"",
|
|
6729
|
-
`Task id: \`${ctx.taskId}\``,
|
|
6730
|
-
"",
|
|
6731
|
-
"### Brief",
|
|
6732
|
-
"",
|
|
6733
|
-
brief,
|
|
6734
|
-
"",
|
|
6735
|
-
seedSection,
|
|
6736
|
-
correlationSection,
|
|
6737
|
-
workspaceSection,
|
|
6738
|
-
"### Workflow",
|
|
6739
|
-
"",
|
|
6812
|
+
const workflow = [
|
|
6740
6813
|
ctx.workspace?.mode === "dedicated_worktree" ? `1. Use the already-provisioned dedicated worktree branch${ctx.workspace.branch ? ` (\`${ctx.workspace.branch}\`)` : ""}; do not create or switch the primary checkout.` : `1. Create a feature branch (starting prefix suggestion: \`${branchSlug}<short-slug>\`).`,
|
|
6741
6814
|
"2. Understand the problem — read relevant code; do not speculate.",
|
|
6742
6815
|
"3. Implement the change. Keep commits small and coherent.",
|
|
@@ -6744,24 +6817,68 @@ function buildFulfillBriefUserPrompt(input, ctx) {
|
|
|
6744
6817
|
"5. For every commit, create a signed diary entry first via",
|
|
6745
6818
|
" `moltnet_create_entry` and embed its id in the commit trailer",
|
|
6746
6819
|
" `MoltNet-Diary: <id>` (per the runtime instructor).",
|
|
6747
|
-
"6. Push the branch and open a PR."
|
|
6748
|
-
|
|
6749
|
-
|
|
6750
|
-
|
|
6751
|
-
|
|
6752
|
-
|
|
6753
|
-
|
|
6754
|
-
|
|
6755
|
-
|
|
6756
|
-
|
|
6757
|
-
|
|
6758
|
-
|
|
6759
|
-
|
|
6760
|
-
|
|
6761
|
-
|
|
6762
|
-
|
|
6763
|
-
|
|
6764
|
-
|
|
6820
|
+
"6. Push the branch and open a PR."
|
|
6821
|
+
].join("\n");
|
|
6822
|
+
return assembleTaskPrompt("fulfill_brief", [
|
|
6823
|
+
{
|
|
6824
|
+
id: "fulfill_brief.header",
|
|
6825
|
+
source: "header",
|
|
6826
|
+
body: header
|
|
6827
|
+
},
|
|
6828
|
+
{
|
|
6829
|
+
id: "fulfill_brief.brief",
|
|
6830
|
+
source: "task_input",
|
|
6831
|
+
header: "Brief",
|
|
6832
|
+
body: brief
|
|
6833
|
+
},
|
|
6834
|
+
{
|
|
6835
|
+
id: "fulfill_brief.seed_files",
|
|
6836
|
+
source: "task_input",
|
|
6837
|
+
header: "Seed files",
|
|
6838
|
+
body: seedFilesBody
|
|
6839
|
+
},
|
|
6840
|
+
{
|
|
6841
|
+
id: "fulfill_brief.correlation",
|
|
6842
|
+
source: "task_input",
|
|
6843
|
+
header: "Correlation",
|
|
6844
|
+
body: correlation
|
|
6845
|
+
},
|
|
6846
|
+
{
|
|
6847
|
+
id: "fulfill_brief.workspace",
|
|
6848
|
+
source: "workspace",
|
|
6849
|
+
header: "Workspace",
|
|
6850
|
+
body: workspace
|
|
6851
|
+
},
|
|
6852
|
+
{
|
|
6853
|
+
id: "fulfill_brief.workflow",
|
|
6854
|
+
source: "static",
|
|
6855
|
+
header: "Workflow",
|
|
6856
|
+
body: workflow
|
|
6857
|
+
},
|
|
6858
|
+
{
|
|
6859
|
+
id: "fulfill_brief.verification",
|
|
6860
|
+
source: "verification",
|
|
6861
|
+
body: buildSelfVerificationBlock(ctx.taskId)
|
|
6862
|
+
},
|
|
6863
|
+
{
|
|
6864
|
+
id: "fulfill_brief.final_output",
|
|
6865
|
+
source: "final_output",
|
|
6866
|
+
body: buildFinalOutputBlock({
|
|
6867
|
+
taskType: "fulfill_brief",
|
|
6868
|
+
outputSchemaName: "FulfillBriefOutput",
|
|
6869
|
+
shapeSketch: [
|
|
6870
|
+
"{",
|
|
6871
|
+
" \"branch\": \"<branch-name>\",",
|
|
6872
|
+
" \"commits\": [{ \"sha\": \"...\", \"message\": \"...\", \"diaryEntryId\": \"...\" }],",
|
|
6873
|
+
" \"pullRequestUrl\": \"<url-or-null>\",",
|
|
6874
|
+
" \"diaryEntryIds\": [\"...\"],",
|
|
6875
|
+
" \"summary\": \"<1-3 sentence recap>\",",
|
|
6876
|
+
" \"verification\": <required iff input.successCriteria; see Self-verification>",
|
|
6877
|
+
"}"
|
|
6878
|
+
].join("\n")
|
|
6879
|
+
})
|
|
6880
|
+
}
|
|
6881
|
+
]);
|
|
6765
6882
|
}
|
|
6766
6883
|
//#endregion
|
|
6767
6884
|
//#region ../../libs/agent-runtime/src/prompts/judge-eval-attempt.ts
|
|
@@ -6770,46 +6887,18 @@ function buildJudgeEvalAttemptUserPrompt(input, ctx) {
|
|
|
6770
6887
|
if (!rubric) throw new Error("judge_eval_attempt requires successCriteria.rubric — none present");
|
|
6771
6888
|
const escapeCell = (s) => s.replace(/\\/g, "\\\\").replace(/\|/g, "\\|").replace(/\r?\n/g, " ");
|
|
6772
6889
|
const criteriaTable = rubric.criteria.map((c) => `| \`${c.id}\` | ${c.weight.toFixed(3)} | ${c.scoring} | ${escapeCell(c.description)} |`).join("\n");
|
|
6773
|
-
const
|
|
6774
|
-
|
|
6775
|
-
outputSchemaName: "JudgeEvalAttemptOutput",
|
|
6776
|
-
shapeSketch: [
|
|
6777
|
-
"{",
|
|
6778
|
-
` "targetTaskId": "${input.targetTaskId}",`,
|
|
6779
|
-
` "targetAttemptN": ${input.targetAttemptN},`,
|
|
6780
|
-
" \"variantLabel\": \"<from producer input>\",",
|
|
6781
|
-
" \"scores\": [ { \"criterionId\": \"...\", \"score\": 0..1, \"rationale\": \"...\", \"assertions\": [...]? } ],",
|
|
6782
|
-
" \"composite\": <Σ(weight × score), 0..1>,",
|
|
6783
|
-
" \"verdict\": \"<1-3 sentences>\",",
|
|
6784
|
-
" \"judgeModel\": \"<id>\", // optional",
|
|
6785
|
-
" \"traceparent\": \"<from claim>\"",
|
|
6786
|
-
"}"
|
|
6787
|
-
].join("\n")
|
|
6788
|
-
});
|
|
6789
|
-
const workspaceSection = ctx.workspace?.attached === true ? [
|
|
6790
|
-
"### Workspace",
|
|
6890
|
+
const header = [
|
|
6891
|
+
"# Judge Eval Attempt",
|
|
6791
6892
|
"",
|
|
6792
|
-
"Your current workspace is already attached to the producer attempt",
|
|
6793
|
-
"you are judging. Inspect files directly from the current workspace",
|
|
6794
|
-
"root instead of inventing synthetic `artifact_<taskId>` paths.",
|
|
6795
|
-
"If the accepted attempt output lists `artifacts[].path`, treat those",
|
|
6796
|
-
"paths as relative to the current workspace root unless the output",
|
|
6797
|
-
"explicitly says otherwise.",
|
|
6798
|
-
ctx.workspace.mode === "dedicated_worktree" ? `This attachment is a dedicated producer worktree${ctx.workspace.branch ? ` on branch \`${ctx.workspace.branch}\`` : ""}.` : ctx.workspace.mode === "scratch_mount" ? "This workspace is a fresh judge-owned scratch copy of the producer workspace." : "This attachment is the producer shared workspace mounted with shadow writes for safe inspection.",
|
|
6799
|
-
""
|
|
6800
|
-
].join("\n") : "";
|
|
6801
|
-
return [
|
|
6802
|
-
"# Judge Eval Attempt\n",
|
|
6803
6893
|
"You are grading one accepted `run_eval` producer attempt against a hidden",
|
|
6804
6894
|
"judge rubric. Do not delegate to subagents. Grade in this session only.",
|
|
6805
6895
|
"",
|
|
6806
6896
|
`Task id: \`${ctx.taskId}\``,
|
|
6807
6897
|
`Diary: \`${ctx.diaryId}\``,
|
|
6808
6898
|
`Producer task: \`${input.targetTaskId}\``,
|
|
6809
|
-
`Producer attempt: \`${input.targetAttemptN}
|
|
6810
|
-
|
|
6811
|
-
|
|
6812
|
-
"",
|
|
6899
|
+
`Producer attempt: \`${input.targetAttemptN}\``
|
|
6900
|
+
].join("\n");
|
|
6901
|
+
const evidence = [
|
|
6813
6902
|
`1. Call \`moltnet_get_task\` with taskId=\`${input.targetTaskId}\`.`,
|
|
6814
6903
|
`2. Call \`moltnet_list_task_attempts\` with taskId=\`${input.targetTaskId}\` and inspect the accepted attempt matching \`${input.targetAttemptN}\`.`,
|
|
6815
6904
|
`3. Call \`moltnet_list_task_messages\` with taskId=\`${input.targetTaskId}\`, attemptN=\`${input.targetAttemptN}\` to inspect the producer's turn-by-turn behavior.`,
|
|
@@ -6817,32 +6906,82 @@ function buildJudgeEvalAttemptUserPrompt(input, ctx) {
|
|
|
6817
6906
|
" artifacts or workspace evidence available in your environment.",
|
|
6818
6907
|
" Read artifact files from the mounted producer workspace when present;",
|
|
6819
6908
|
" do not assume detached `artifact_<taskId>` directories exist.",
|
|
6820
|
-
"5. Score strictly against the rubric below."
|
|
6821
|
-
|
|
6822
|
-
|
|
6823
|
-
"
|
|
6824
|
-
"",
|
|
6825
|
-
|
|
6909
|
+
"5. Score strictly against the rubric below."
|
|
6910
|
+
].join("\n");
|
|
6911
|
+
const workspace = ctx.workspace?.attached === true ? [
|
|
6912
|
+
"Your current workspace is already attached to the producer attempt",
|
|
6913
|
+
"you are judging. Inspect files directly from the current workspace",
|
|
6914
|
+
"root instead of inventing synthetic `artifact_<taskId>` paths.",
|
|
6915
|
+
"If the accepted attempt output lists `artifacts[].path`, treat those",
|
|
6916
|
+
"paths as relative to the current workspace root unless the output",
|
|
6917
|
+
"explicitly says otherwise.",
|
|
6918
|
+
ctx.workspace.mode === "dedicated_worktree" ? `This attachment is a dedicated producer worktree${ctx.workspace.branch ? ` on branch \`${ctx.workspace.branch}\`` : ""}.` : ctx.workspace.mode === "scratch_mount" ? "This workspace is a fresh judge-owned scratch copy of the producer workspace." : "This attachment is the producer shared workspace mounted with shadow writes for safe inspection."
|
|
6919
|
+
].join("\n") : "";
|
|
6920
|
+
const rubricBody = [
|
|
6921
|
+
rubric.preamble ?? "",
|
|
6826
6922
|
"| Criterion | Weight | Scoring | Description |",
|
|
6827
6923
|
"| --- | --- | --- | --- |",
|
|
6828
|
-
criteriaTable
|
|
6829
|
-
"",
|
|
6830
|
-
"### Composite arithmetic",
|
|
6831
|
-
"",
|
|
6832
|
-
"Your `composite` MUST equal `Σ(criterion.weight × score)` over the rubric",
|
|
6833
|
-
"criteria. Drift > 0.001 is rejected.",
|
|
6834
|
-
"",
|
|
6835
|
-
finalOutputBlock
|
|
6924
|
+
criteriaTable
|
|
6836
6925
|
].filter((s) => s !== "").join("\n");
|
|
6926
|
+
const composite = ["Your `composite` MUST equal `Σ(criterion.weight × score)` over the rubric", "criteria. Drift > 0.001 is rejected."].join("\n");
|
|
6927
|
+
return assembleTaskPrompt("judge_eval_attempt", [
|
|
6928
|
+
{
|
|
6929
|
+
id: "judge_eval_attempt.header",
|
|
6930
|
+
source: "header",
|
|
6931
|
+
body: header
|
|
6932
|
+
},
|
|
6933
|
+
{
|
|
6934
|
+
id: "judge_eval_attempt.evidence",
|
|
6935
|
+
source: "evidence",
|
|
6936
|
+
header: "Evidence gathering",
|
|
6937
|
+
body: evidence
|
|
6938
|
+
},
|
|
6939
|
+
{
|
|
6940
|
+
id: "judge_eval_attempt.workspace",
|
|
6941
|
+
source: "workspace",
|
|
6942
|
+
header: "Workspace",
|
|
6943
|
+
body: workspace
|
|
6944
|
+
},
|
|
6945
|
+
{
|
|
6946
|
+
id: "judge_eval_attempt.rubric",
|
|
6947
|
+
source: "rubric_judge",
|
|
6948
|
+
header: "Rubric",
|
|
6949
|
+
body: rubricBody
|
|
6950
|
+
},
|
|
6951
|
+
{
|
|
6952
|
+
id: "judge_eval_attempt.composite",
|
|
6953
|
+
source: "rubric_judge",
|
|
6954
|
+
header: "Composite arithmetic",
|
|
6955
|
+
body: composite
|
|
6956
|
+
},
|
|
6957
|
+
{
|
|
6958
|
+
id: "judge_eval_attempt.final_output",
|
|
6959
|
+
source: "final_output",
|
|
6960
|
+
body: buildFinalOutputBlock({
|
|
6961
|
+
taskType: "judge_eval_attempt",
|
|
6962
|
+
outputSchemaName: "JudgeEvalAttemptOutput",
|
|
6963
|
+
shapeSketch: [
|
|
6964
|
+
"{",
|
|
6965
|
+
` "targetTaskId": "${input.targetTaskId}",`,
|
|
6966
|
+
` "targetAttemptN": ${input.targetAttemptN},`,
|
|
6967
|
+
" \"variantLabel\": \"<from producer input>\",",
|
|
6968
|
+
" \"scores\": [ { \"criterionId\": \"...\", \"score\": 0..1, \"rationale\": \"...\", \"assertions\": [...]? } ],",
|
|
6969
|
+
" \"composite\": <Σ(weight × score), 0..1>,",
|
|
6970
|
+
" \"verdict\": \"<1-3 sentences>\",",
|
|
6971
|
+
" \"judgeModel\": \"<id>\", // optional",
|
|
6972
|
+
" \"traceparent\": \"<from claim>\"",
|
|
6973
|
+
"}"
|
|
6974
|
+
].join("\n")
|
|
6975
|
+
})
|
|
6976
|
+
}
|
|
6977
|
+
]);
|
|
6837
6978
|
}
|
|
6838
6979
|
//#endregion
|
|
6839
6980
|
//#region ../../libs/agent-runtime/src/prompts/judge-pack.ts
|
|
6840
6981
|
function buildJudgePackUserPrompt(input, ctx) {
|
|
6841
6982
|
const { renderedPackId, sourcePackId, successCriteria } = input;
|
|
6842
6983
|
const rubric = successCriteria.rubric;
|
|
6843
|
-
const
|
|
6844
|
-
const preambleSection = renderRubricPreambleSection(rubric);
|
|
6845
|
-
return [
|
|
6984
|
+
const header = [
|
|
6846
6985
|
"# Judge Pack Agent",
|
|
6847
6986
|
"",
|
|
6848
6987
|
"You are an independent judge. You did NOT curate or render the pack",
|
|
@@ -6851,17 +6990,15 @@ function buildJudgePackUserPrompt(input, ctx) {
|
|
|
6851
6990
|
"referenced entries — but do NOT modify anything.",
|
|
6852
6991
|
"",
|
|
6853
6992
|
`Your diary ID is: ${ctx.diaryId}`,
|
|
6854
|
-
`This task's id is: ${ctx.taskId}
|
|
6855
|
-
|
|
6856
|
-
|
|
6857
|
-
"",
|
|
6993
|
+
`This task's id is: ${ctx.taskId}`
|
|
6994
|
+
].join("\n");
|
|
6995
|
+
const target = [
|
|
6858
6996
|
`- **Rendered pack**: \`${renderedPackId}\``,
|
|
6859
6997
|
`- **Source pack**: \`${sourcePackId}\``,
|
|
6860
|
-
`- **Rubric**: \`${rubric.rubricId}\` v${rubric.version}
|
|
6861
|
-
|
|
6862
|
-
|
|
6863
|
-
|
|
6864
|
-
"",
|
|
6998
|
+
`- **Rubric**: \`${rubric.rubricId}\` v${rubric.version}`
|
|
6999
|
+
].join("\n");
|
|
7000
|
+
const preamble = renderRubricPreambleSection(rubric) ?? "";
|
|
7001
|
+
const workflow = [
|
|
6865
7002
|
"1. Call `moltnet_rendered_pack_get` for the rendered pack. Keep the",
|
|
6866
7003
|
" `content` string — you will score it.",
|
|
6867
7004
|
"2. Call `moltnet_pack_get` with `expandEntries: true` for the source",
|
|
@@ -6869,14 +7006,10 @@ function buildJudgePackUserPrompt(input, ctx) {
|
|
|
6869
7006
|
"3. For each criterion, score according to its `scoring` mode (see",
|
|
6870
7007
|
" Scoring rules below). Produce rationales where required.",
|
|
6871
7008
|
"4. Compute `composite = Σ(weight_i × score_i)` and sanity-check it",
|
|
6872
|
-
" equals the sum you will emit — the runtime rejects mismatches."
|
|
6873
|
-
|
|
6874
|
-
|
|
6875
|
-
|
|
6876
|
-
criteriaList,
|
|
6877
|
-
"",
|
|
6878
|
-
"### Scoring rules",
|
|
6879
|
-
"",
|
|
7009
|
+
" equals the sum you will emit — the runtime rejects mismatches."
|
|
7010
|
+
].join("\n");
|
|
7011
|
+
const criteria = renderRubricCriteriaList(rubric);
|
|
7012
|
+
const scoring = [
|
|
6880
7013
|
"- `llm_score`: score 0..1 continuous. `rationale` REQUIRED (2–4",
|
|
6881
7014
|
" sentences pointing at specific evidence in the rendered content or",
|
|
6882
7015
|
" the source entries). NOTE: this mode smooths individual failures",
|
|
@@ -6915,80 +7048,95 @@ function buildJudgePackUserPrompt(input, ctx) {
|
|
|
6915
7048
|
"- `deterministic_coverage_check`: for every source entry, check",
|
|
6916
7049
|
" whether its `entryId` (or a stable reference like title + CID",
|
|
6917
7050
|
" prefix) appears in the rendered `content`. Score 1 iff coverage is",
|
|
6918
|
-
" complete; otherwise 0. Populate `evidence` with `{ covered, total, missing: [entryIds] }`."
|
|
6919
|
-
|
|
6920
|
-
|
|
6921
|
-
"",
|
|
7051
|
+
" complete; otherwise 0. Populate `evidence` with `{ covered, total, missing: [entryIds] }`."
|
|
7052
|
+
].join("\n");
|
|
7053
|
+
const constraints = [
|
|
6922
7054
|
"- Do NOT call `moltnet_pack_create` or `moltnet_pack_render`.",
|
|
6923
7055
|
"- Do NOT fetch the curator's or renderer's task output directly — they",
|
|
6924
7056
|
" may leak guidance that biases judgment.",
|
|
6925
7057
|
"- Keep the session focused on scoring; no speculative exploration.",
|
|
6926
7058
|
"",
|
|
6927
|
-
`Write a signed diary entry (tags: \`judgment\`, \`judge_pack\`, \`rubric:${rubric.rubricId}\`) capturing the rationale before
|
|
6928
|
-
|
|
6929
|
-
|
|
6930
|
-
|
|
6931
|
-
|
|
6932
|
-
|
|
6933
|
-
|
|
6934
|
-
|
|
6935
|
-
|
|
6936
|
-
|
|
6937
|
-
|
|
6938
|
-
|
|
6939
|
-
|
|
6940
|
-
|
|
6941
|
-
|
|
6942
|
-
|
|
6943
|
-
|
|
6944
|
-
|
|
6945
|
-
|
|
6946
|
-
|
|
6947
|
-
|
|
6948
|
-
|
|
6949
|
-
|
|
6950
|
-
|
|
6951
|
-
|
|
6952
|
-
|
|
6953
|
-
|
|
6954
|
-
|
|
6955
|
-
|
|
6956
|
-
|
|
6957
|
-
|
|
7059
|
+
`Write a signed diary entry (tags: \`judgment\`, \`judge_pack\`, \`rubric:${rubric.rubricId}\`) capturing the rationale before reporting structured output.`
|
|
7060
|
+
].join("\n");
|
|
7061
|
+
return assembleTaskPrompt("judge_pack", [
|
|
7062
|
+
{
|
|
7063
|
+
id: "judge_pack.header",
|
|
7064
|
+
source: "header",
|
|
7065
|
+
body: header
|
|
7066
|
+
},
|
|
7067
|
+
{
|
|
7068
|
+
id: "judge_pack.target",
|
|
7069
|
+
source: "task_input",
|
|
7070
|
+
header: "Target",
|
|
7071
|
+
body: target
|
|
7072
|
+
},
|
|
7073
|
+
{
|
|
7074
|
+
id: "judge_pack.preamble",
|
|
7075
|
+
source: "rubric_judge",
|
|
7076
|
+
body: preamble
|
|
7077
|
+
},
|
|
7078
|
+
{
|
|
7079
|
+
id: "judge_pack.workflow",
|
|
7080
|
+
source: "static",
|
|
7081
|
+
header: "Workflow",
|
|
7082
|
+
body: workflow
|
|
7083
|
+
},
|
|
7084
|
+
{
|
|
7085
|
+
id: "judge_pack.criteria",
|
|
7086
|
+
source: "rubric_judge",
|
|
7087
|
+
header: "Criteria",
|
|
7088
|
+
body: criteria
|
|
7089
|
+
},
|
|
7090
|
+
{
|
|
7091
|
+
id: "judge_pack.scoring",
|
|
7092
|
+
source: "rubric_judge",
|
|
7093
|
+
header: "Scoring rules",
|
|
7094
|
+
body: scoring
|
|
7095
|
+
},
|
|
7096
|
+
{
|
|
7097
|
+
id: "judge_pack.constraints",
|
|
7098
|
+
source: "static",
|
|
7099
|
+
header: "Constraints",
|
|
7100
|
+
body: constraints
|
|
7101
|
+
},
|
|
7102
|
+
{
|
|
7103
|
+
id: "judge_pack.final_output",
|
|
7104
|
+
source: "final_output",
|
|
7105
|
+
body: buildFinalOutputBlock({
|
|
7106
|
+
taskType: "judge_pack",
|
|
7107
|
+
outputSchemaName: "JudgePackOutput",
|
|
7108
|
+
shapeSketch: [
|
|
7109
|
+
"{",
|
|
7110
|
+
" \"scores\": [",
|
|
7111
|
+
" { \"criterionId\": \"...\", \"score\": 0.0, \"rationale\": \"...\", \"evidence\": {} },",
|
|
7112
|
+
" {",
|
|
7113
|
+
" \"criterionId\": \"<llm_checklist criterion>\",",
|
|
7114
|
+
" \"score\": 0, // 1 iff every assertion passed",
|
|
7115
|
+
" \"assertions\": [",
|
|
7116
|
+
" { \"id\": \"claim-1\", \"text\": \"...\", \"passed\": false, \"evidence\": \"...\" }",
|
|
7117
|
+
" ]",
|
|
7118
|
+
" }",
|
|
7119
|
+
" ],",
|
|
7120
|
+
" \"composite\": <sum-of-weighted-scores>,",
|
|
7121
|
+
" \"verdict\": \"<1-3 sentence overall>\",",
|
|
7122
|
+
" \"judgeModel\": \"<provider:model>\",",
|
|
7123
|
+
" \"rendererBinaryCid\": \"<cid-string-only-if-available>\"",
|
|
7124
|
+
"}"
|
|
7125
|
+
].join("\n"),
|
|
7126
|
+
extraNotes: [
|
|
7127
|
+
"Omit `rendererBinaryCid` entirely when no binary CID is exposed by",
|
|
7128
|
+
"`moltnet_rendered_pack_get`. Do NOT emit `null` — the field is",
|
|
7129
|
+
"optional and absence is the correct representation when unavailable."
|
|
7130
|
+
]
|
|
7131
|
+
})
|
|
7132
|
+
}
|
|
7133
|
+
]);
|
|
6958
7134
|
}
|
|
6959
7135
|
//#endregion
|
|
6960
7136
|
//#region ../../libs/agent-runtime/src/prompts/pr-review.ts
|
|
6961
7137
|
function buildPrReviewUserPrompt(input, ctx) {
|
|
6962
7138
|
const rubric = input.successCriteria.rubric;
|
|
6963
|
-
const
|
|
6964
|
-
const preambleSection = renderRubricPreambleSection(rubric);
|
|
6965
|
-
const taskPromptSection = input.taskPrompt ? [
|
|
6966
|
-
"## Task-specific instructions",
|
|
6967
|
-
"",
|
|
6968
|
-
input.taskPrompt,
|
|
6969
|
-
""
|
|
6970
|
-
].join("\n") : "";
|
|
6971
|
-
const resourceSection = input.subject.resourceUrls && input.subject.resourceUrls.length > 0 ? [
|
|
6972
|
-
"### Resources",
|
|
6973
|
-
"",
|
|
6974
|
-
...input.subject.resourceUrls.map((url) => `- ${url}`),
|
|
6975
|
-
""
|
|
6976
|
-
].join("\n") : "";
|
|
6977
|
-
const hintsSection = input.subject.inspectionHints && input.subject.inspectionHints.length > 0 ? [
|
|
6978
|
-
"### Inspection hints",
|
|
6979
|
-
"",
|
|
6980
|
-
...input.subject.inspectionHints.map((hint) => `- ${hint}`),
|
|
6981
|
-
""
|
|
6982
|
-
].join("\n") : "";
|
|
6983
|
-
const workspaceSection = ctx.workspace?.mode === "dedicated_worktree" ? [
|
|
6984
|
-
"### Workspace",
|
|
6985
|
-
"",
|
|
6986
|
-
"This review attempt is running inside a dedicated disposable git",
|
|
6987
|
-
"worktree. Inspect and reason inside this workspace only.",
|
|
6988
|
-
ctx.workspace.branch ? `The current review branch is \`${ctx.workspace.branch}\`.` : "The current checkout is disposable and will be cleaned up when the task ends.",
|
|
6989
|
-
""
|
|
6990
|
-
].join("\n") : "";
|
|
6991
|
-
return [
|
|
7139
|
+
const header = [
|
|
6992
7140
|
"# Review Agent",
|
|
6993
7141
|
"",
|
|
6994
7142
|
"You are an independent judge. You did NOT produce the subject under review.",
|
|
@@ -6996,29 +7144,30 @@ function buildPrReviewUserPrompt(input, ctx) {
|
|
|
6996
7144
|
"You may inspect the local workspace and the referenced resources, but do NOT modify anything.",
|
|
6997
7145
|
"",
|
|
6998
7146
|
`Your diary ID is: ${ctx.diaryId}`,
|
|
6999
|
-
`This task's id is: ${ctx.taskId}
|
|
7000
|
-
|
|
7001
|
-
|
|
7002
|
-
"",
|
|
7147
|
+
`This task's id is: ${ctx.taskId}`
|
|
7148
|
+
].join("\n");
|
|
7149
|
+
const subject = [
|
|
7003
7150
|
`**Title:** ${input.subject.title}`,
|
|
7004
7151
|
"",
|
|
7005
|
-
input.subject.summary
|
|
7006
|
-
|
|
7007
|
-
|
|
7008
|
-
|
|
7009
|
-
|
|
7010
|
-
"
|
|
7011
|
-
"",
|
|
7152
|
+
input.subject.summary
|
|
7153
|
+
].join("\n");
|
|
7154
|
+
const resources = input.subject.resourceUrls && input.subject.resourceUrls.length > 0 ? input.subject.resourceUrls.map((url) => `- ${url}`).join("\n") : "";
|
|
7155
|
+
const hints = input.subject.inspectionHints && input.subject.inspectionHints.length > 0 ? input.subject.inspectionHints.map((hint) => `- ${hint}`).join("\n") : "";
|
|
7156
|
+
const workspace = ctx.workspace?.mode === "dedicated_worktree" ? [
|
|
7157
|
+
"This review attempt is running inside a dedicated disposable git",
|
|
7158
|
+
"worktree. Inspect and reason inside this workspace only.",
|
|
7159
|
+
ctx.workspace.branch ? `The current review branch is \`${ctx.workspace.branch}\`.` : "The current checkout is disposable and will be cleaned up when the task ends."
|
|
7160
|
+
].join("\n") : "";
|
|
7161
|
+
const executionContract = [
|
|
7012
7162
|
"Treat the provided subject, resources, inspection hints, and any",
|
|
7013
7163
|
"task-specific instructions as the full",
|
|
7014
7164
|
"review contract for this task.",
|
|
7015
7165
|
"",
|
|
7016
7166
|
"If the task-specific instructions or inspection hints require an outward action tied to the review",
|
|
7017
7167
|
"(for example publishing the judgment somewhere), perform that action as",
|
|
7018
|
-
"part of the task before reporting structured output."
|
|
7019
|
-
|
|
7020
|
-
|
|
7021
|
-
"",
|
|
7168
|
+
"part of the task before reporting structured output."
|
|
7169
|
+
].join("\n");
|
|
7170
|
+
const workflow = [
|
|
7022
7171
|
"1. Read the subject summary, resources, inspection hints, and any",
|
|
7023
7172
|
" task-specific instructions before scoring.",
|
|
7024
7173
|
"2. Inspect the target artefact directly using the tools and resources the",
|
|
@@ -7028,39 +7177,104 @@ function buildPrReviewUserPrompt(input, ctx) {
|
|
|
7028
7177
|
"4. Apply the rubric strictly. This task is about complexity and",
|
|
7029
7178
|
" reviewability, not correctness or feature desirability.",
|
|
7030
7179
|
"5. Perform any required outward action before emitting the final",
|
|
7031
|
-
" structured output."
|
|
7032
|
-
|
|
7033
|
-
|
|
7034
|
-
|
|
7035
|
-
|
|
7036
|
-
|
|
7037
|
-
criteriaList,
|
|
7038
|
-
"",
|
|
7039
|
-
"### Scoring rules",
|
|
7040
|
-
"",
|
|
7180
|
+
" structured output."
|
|
7181
|
+
].join("\n");
|
|
7182
|
+
const taskPromptSection = input.taskPrompt ?? "";
|
|
7183
|
+
const preamble = renderRubricPreambleSection(rubric) ?? "";
|
|
7184
|
+
const criteria = renderRubricCriteriaList(rubric);
|
|
7185
|
+
const scoring = [
|
|
7041
7186
|
"- Every criterion uses binary scoring only.",
|
|
7042
7187
|
"- Score `1` when the subject clearly clears the criterion.",
|
|
7043
7188
|
"- Score `0` when it does not, or when the evidence is ambiguous.",
|
|
7044
7189
|
"- `rationale` is REQUIRED for every score. Keep it concrete and audit-friendly.",
|
|
7045
7190
|
"- Compute `composite = Σ(weight_i × score_i)` exactly; the runtime rejects mismatches.",
|
|
7046
7191
|
"",
|
|
7047
|
-
"Write a signed diary entry (tags: `judgment`, `pr_review`) capturing the rationale before reporting structured output."
|
|
7048
|
-
|
|
7049
|
-
|
|
7050
|
-
|
|
7051
|
-
|
|
7052
|
-
|
|
7053
|
-
|
|
7054
|
-
|
|
7055
|
-
|
|
7056
|
-
|
|
7057
|
-
|
|
7058
|
-
|
|
7059
|
-
|
|
7060
|
-
|
|
7061
|
-
|
|
7062
|
-
|
|
7063
|
-
|
|
7192
|
+
"Write a signed diary entry (tags: `judgment`, `pr_review`) capturing the rationale before reporting structured output."
|
|
7193
|
+
].join("\n");
|
|
7194
|
+
return assembleTaskPrompt("pr_review", [
|
|
7195
|
+
{
|
|
7196
|
+
id: "pr_review.header",
|
|
7197
|
+
source: "header",
|
|
7198
|
+
body: header
|
|
7199
|
+
},
|
|
7200
|
+
{
|
|
7201
|
+
id: "pr_review.subject",
|
|
7202
|
+
source: "task_input",
|
|
7203
|
+
header: "Subject",
|
|
7204
|
+
body: subject
|
|
7205
|
+
},
|
|
7206
|
+
{
|
|
7207
|
+
id: "pr_review.resources",
|
|
7208
|
+
source: "task_input",
|
|
7209
|
+
header: "Resources",
|
|
7210
|
+
body: resources
|
|
7211
|
+
},
|
|
7212
|
+
{
|
|
7213
|
+
id: "pr_review.hints",
|
|
7214
|
+
source: "task_input",
|
|
7215
|
+
header: "Inspection hints",
|
|
7216
|
+
body: hints
|
|
7217
|
+
},
|
|
7218
|
+
{
|
|
7219
|
+
id: "pr_review.workspace",
|
|
7220
|
+
source: "workspace",
|
|
7221
|
+
header: "Workspace",
|
|
7222
|
+
body: workspace
|
|
7223
|
+
},
|
|
7224
|
+
{
|
|
7225
|
+
id: "pr_review.execution_contract",
|
|
7226
|
+
source: "static",
|
|
7227
|
+
header: "Execution contract",
|
|
7228
|
+
body: executionContract
|
|
7229
|
+
},
|
|
7230
|
+
{
|
|
7231
|
+
id: "pr_review.workflow",
|
|
7232
|
+
source: "static",
|
|
7233
|
+
header: "Review workflow",
|
|
7234
|
+
body: workflow
|
|
7235
|
+
},
|
|
7236
|
+
{
|
|
7237
|
+
id: "pr_review.task_prompt",
|
|
7238
|
+
source: "task_input",
|
|
7239
|
+
header: "Task-specific instructions",
|
|
7240
|
+
body: taskPromptSection
|
|
7241
|
+
},
|
|
7242
|
+
{
|
|
7243
|
+
id: "pr_review.preamble",
|
|
7244
|
+
source: "rubric_judge",
|
|
7245
|
+
body: preamble
|
|
7246
|
+
},
|
|
7247
|
+
{
|
|
7248
|
+
id: "pr_review.criteria",
|
|
7249
|
+
source: "rubric_judge",
|
|
7250
|
+
header: "Criteria",
|
|
7251
|
+
body: criteria
|
|
7252
|
+
},
|
|
7253
|
+
{
|
|
7254
|
+
id: "pr_review.scoring",
|
|
7255
|
+
source: "rubric_judge",
|
|
7256
|
+
header: "Scoring rules",
|
|
7257
|
+
body: scoring
|
|
7258
|
+
},
|
|
7259
|
+
{
|
|
7260
|
+
id: "pr_review.final_output",
|
|
7261
|
+
source: "final_output",
|
|
7262
|
+
body: buildFinalOutputBlock({
|
|
7263
|
+
taskType: "pr_review",
|
|
7264
|
+
outputSchemaName: "PrReviewOutput",
|
|
7265
|
+
shapeSketch: [
|
|
7266
|
+
"{",
|
|
7267
|
+
" \"scores\": [",
|
|
7268
|
+
" { \"criterionId\": \"...\", \"score\": 0, \"rationale\": \"...\" }",
|
|
7269
|
+
" ],",
|
|
7270
|
+
" \"composite\": <sum-of-weighted-binary-scores>,",
|
|
7271
|
+
" \"verdict\": \"<1-3 sentence overall>\"",
|
|
7272
|
+
"}"
|
|
7273
|
+
].join("\n"),
|
|
7274
|
+
extraNotes: ["`scores` MUST stay in the same order as the rubric criteria.", "`score` MUST be exactly `0` or `1` for every criterion."]
|
|
7275
|
+
})
|
|
7276
|
+
}
|
|
7277
|
+
]);
|
|
7064
7278
|
}
|
|
7065
7279
|
//#endregion
|
|
7066
7280
|
//#region ../../libs/agent-runtime/src/prompts/render-pack.ts
|
|
@@ -7070,7 +7284,7 @@ function buildPrReviewUserPrompt(input, ctx) {
|
|
|
7070
7284
|
*/
|
|
7071
7285
|
function buildRenderPackUserPrompt(input, ctx) {
|
|
7072
7286
|
const { packId, persist = true, pinned = false } = input;
|
|
7073
|
-
|
|
7287
|
+
const header = [
|
|
7074
7288
|
"# Render Pack Agent",
|
|
7075
7289
|
"",
|
|
7076
7290
|
"You are rendering a context pack to markdown. Step 2 of the",
|
|
@@ -7078,16 +7292,14 @@ function buildRenderPackUserPrompt(input, ctx) {
|
|
|
7078
7292
|
"a third will judge the rendering. You must NOT judge it here.",
|
|
7079
7293
|
"",
|
|
7080
7294
|
`Your agent-session diary ID is: ${ctx.diaryId}`,
|
|
7081
|
-
`This task's id is: ${ctx.taskId}
|
|
7082
|
-
|
|
7083
|
-
|
|
7084
|
-
"",
|
|
7295
|
+
`This task's id is: ${ctx.taskId}`
|
|
7296
|
+
].join("\n");
|
|
7297
|
+
const inputBlock = [
|
|
7085
7298
|
`- **Pack**: \`${packId}\``,
|
|
7086
7299
|
`- **Persist**: \`${persist}\``,
|
|
7087
|
-
`- **Pinned**: \`${pinned}
|
|
7088
|
-
|
|
7089
|
-
|
|
7090
|
-
"",
|
|
7300
|
+
`- **Pinned**: \`${pinned}\``
|
|
7301
|
+
].join("\n");
|
|
7302
|
+
const workflow = [
|
|
7091
7303
|
"1. Call `moltnet_pack_get` with `expandEntries: true` to inspect the",
|
|
7092
7304
|
" source entries. Read it — you need the entry count for your output.",
|
|
7093
7305
|
"2. Call `moltnet_pack_render` with:",
|
|
@@ -7095,16 +7307,14 @@ function buildRenderPackUserPrompt(input, ctx) {
|
|
|
7095
7307
|
` - \`persist\`: \`${persist}\``,
|
|
7096
7308
|
` - \`pinned\`: \`${pinned}\``,
|
|
7097
7309
|
" Record the returned `renderedPackId`, `cid`, `renderMethod`, and",
|
|
7098
|
-
" `content` byte length."
|
|
7099
|
-
|
|
7100
|
-
|
|
7101
|
-
"",
|
|
7310
|
+
" `content` byte length."
|
|
7311
|
+
].join("\n");
|
|
7312
|
+
const constraints = [
|
|
7102
7313
|
"- Do NOT modify the source pack or its entries.",
|
|
7103
7314
|
"- Do NOT write diary entries unless a genuine incident occurs",
|
|
7104
|
-
" (rendering failure, invariant violation)."
|
|
7105
|
-
|
|
7106
|
-
|
|
7107
|
-
"",
|
|
7315
|
+
" (rendering failure, invariant violation)."
|
|
7316
|
+
].join("\n");
|
|
7317
|
+
const fidelity = [
|
|
7108
7318
|
"These rules apply when you are producing the markdown yourself rather",
|
|
7109
7319
|
"than relying on a deterministic `server:*` renderer.",
|
|
7110
7320
|
"",
|
|
@@ -7124,25 +7334,63 @@ function buildRenderPackUserPrompt(input, ctx) {
|
|
|
7124
7334
|
" completeness. Optimize for \"no detectable drift across a",
|
|
7125
7335
|
" claim-by-claim audit\", not \"shorter at any cost\". When compressing, prefer",
|
|
7126
7336
|
" tightening prose around a quote rather than altering the quote,",
|
|
7127
|
-
" and prefer summarising a list over silently truncating it."
|
|
7128
|
-
"",
|
|
7129
|
-
buildSelfVerificationBlock(ctx.taskId),
|
|
7130
|
-
buildFinalOutputBlock({
|
|
7131
|
-
taskType: "render_pack",
|
|
7132
|
-
outputSchemaName: "RenderPackOutput",
|
|
7133
|
-
shapeSketch: [
|
|
7134
|
-
"{",
|
|
7135
|
-
" \"renderedPackId\": \"<uuid-or-null>\",",
|
|
7136
|
-
" \"renderedCid\": \"<cid>\",",
|
|
7137
|
-
" \"renderMethod\": \"<label>\",",
|
|
7138
|
-
" \"byteSize\": <int>,",
|
|
7139
|
-
" \"entriesRendered\": <int>,",
|
|
7140
|
-
" \"summary\": \"<1-3 sentence recap>\",",
|
|
7141
|
-
" \"verification\": <required iff input.successCriteria; see Self-verification>",
|
|
7142
|
-
"}"
|
|
7143
|
-
].join("\n")
|
|
7144
|
-
})
|
|
7337
|
+
" and prefer summarising a list over silently truncating it."
|
|
7145
7338
|
].join("\n");
|
|
7339
|
+
return assembleTaskPrompt("render_pack", [
|
|
7340
|
+
{
|
|
7341
|
+
id: "render_pack.header",
|
|
7342
|
+
source: "header",
|
|
7343
|
+
body: header
|
|
7344
|
+
},
|
|
7345
|
+
{
|
|
7346
|
+
id: "render_pack.input",
|
|
7347
|
+
source: "task_input",
|
|
7348
|
+
header: "Input",
|
|
7349
|
+
body: inputBlock
|
|
7350
|
+
},
|
|
7351
|
+
{
|
|
7352
|
+
id: "render_pack.workflow",
|
|
7353
|
+
source: "static",
|
|
7354
|
+
header: "Workflow",
|
|
7355
|
+
body: workflow
|
|
7356
|
+
},
|
|
7357
|
+
{
|
|
7358
|
+
id: "render_pack.constraints",
|
|
7359
|
+
source: "static",
|
|
7360
|
+
header: "Constraints",
|
|
7361
|
+
body: constraints
|
|
7362
|
+
},
|
|
7363
|
+
{
|
|
7364
|
+
id: "render_pack.fidelity",
|
|
7365
|
+
source: "static",
|
|
7366
|
+
header: "Fidelity Discipline",
|
|
7367
|
+
body: fidelity
|
|
7368
|
+
},
|
|
7369
|
+
{
|
|
7370
|
+
id: "render_pack.verification",
|
|
7371
|
+
source: "verification",
|
|
7372
|
+
body: buildSelfVerificationBlock(ctx.taskId)
|
|
7373
|
+
},
|
|
7374
|
+
{
|
|
7375
|
+
id: "render_pack.final_output",
|
|
7376
|
+
source: "final_output",
|
|
7377
|
+
body: buildFinalOutputBlock({
|
|
7378
|
+
taskType: "render_pack",
|
|
7379
|
+
outputSchemaName: "RenderPackOutput",
|
|
7380
|
+
shapeSketch: [
|
|
7381
|
+
"{",
|
|
7382
|
+
" \"renderedPackId\": \"<uuid-or-null>\",",
|
|
7383
|
+
" \"renderedCid\": \"<cid>\",",
|
|
7384
|
+
" \"renderMethod\": \"<label>\",",
|
|
7385
|
+
" \"byteSize\": <int>,",
|
|
7386
|
+
" \"entriesRendered\": <int>,",
|
|
7387
|
+
" \"summary\": \"<1-3 sentence recap>\",",
|
|
7388
|
+
" \"verification\": <required iff input.successCriteria; see Self-verification>",
|
|
7389
|
+
"}"
|
|
7390
|
+
].join("\n")
|
|
7391
|
+
})
|
|
7392
|
+
}
|
|
7393
|
+
]);
|
|
7146
7394
|
}
|
|
7147
7395
|
//#endregion
|
|
7148
7396
|
//#region ../../libs/agent-runtime/src/prompts/run-eval.ts
|
|
@@ -7151,8 +7399,7 @@ function buildRenderPackUserPrompt(input, ctx) {
|
|
|
7151
7399
|
*
|
|
7152
7400
|
* Free-form: no git workflow, no commit ceremony. The executor produces
|
|
7153
7401
|
* a textual response (and optional file artifacts) that later
|
|
7154
|
-
* `judge_eval_attempt` task(s) grade against their own hidden
|
|
7155
|
-
* rubric.
|
|
7402
|
+
* `judge_eval_attempt` task(s) grade against their own hidden rubric.
|
|
7156
7403
|
*
|
|
7157
7404
|
* Context delivery is handled by `resolveTaskContext` (see
|
|
7158
7405
|
* libs/agent-runtime/src/context-bindings.ts) and runs BEFORE this
|
|
@@ -7160,50 +7407,44 @@ function buildRenderPackUserPrompt(input, ctx) {
|
|
|
7160
7407
|
* the body, `skill` items are persisted at the runtime's skill path,
|
|
7161
7408
|
* and `user_inline` items are appended to the first user message. This
|
|
7162
7409
|
* builder does NOT inline `input.context[]` itself.
|
|
7410
|
+
*
|
|
7411
|
+
* Prompt-shape notes (issue #1175, area 1):
|
|
7412
|
+
* - No `Correlation` section: the agent never acts on it. The id is
|
|
7413
|
+
* still carried on attempt event metadata for cross-variant queries.
|
|
7414
|
+
* - No `Execution mode` section: the workspace already reflects the
|
|
7415
|
+
* chosen mode by its shape (scratch/shared mount/dedicated worktree).
|
|
7416
|
+
* Restating it as text adds noise without changing model behavior.
|
|
7417
|
+
* - The "Injected Task Context" phrase is used identically in this
|
|
7418
|
+
* prompt's discipline section and in the materialized context block
|
|
7419
|
+
* header (see context-bindings.ts) so weaker models see one repeated
|
|
7420
|
+
* anchor.
|
|
7421
|
+
* - The discipline copy demands the model encode injected constraints
|
|
7422
|
+
* into the code path itself, not into comments or the verification
|
|
7423
|
+
* field. Quoting the constraint back is not following the task.
|
|
7163
7424
|
*/
|
|
7164
7425
|
function buildRunEvalUserPrompt(input, ctx) {
|
|
7165
|
-
const { scenario, variantLabel,
|
|
7426
|
+
const { scenario, variantLabel, successCriteria } = input;
|
|
7166
7427
|
const hasContext = input.context.length > 0;
|
|
7167
7428
|
const hasInlineContext = input.context.some((entry) => entry.binding === "context_inline");
|
|
7168
|
-
const
|
|
7169
|
-
|
|
7170
|
-
"",
|
|
7171
|
-
|
|
7172
|
-
""
|
|
7173
|
-
].join("\n") : "";
|
|
7174
|
-
const verificationSection = successCriteria ? buildSelfVerificationBlock(ctx.taskId) : "";
|
|
7175
|
-
const correlationSection = ctx.correlationId ? [
|
|
7176
|
-
"### Correlation",
|
|
7177
|
-
"",
|
|
7178
|
-
`This task carries correlationId \`${ctx.correlationId}\`. It joins`,
|
|
7179
|
-
"this variant to its sibling `run_eval` tasks (other variants of the",
|
|
7180
|
-
"same scenario and to any later `judge_eval_attempt` tasks created",
|
|
7181
|
-
"against those variants. You do not need to act on it directly — it",
|
|
7182
|
-
"is recorded for cross-variant aggregation at query time.",
|
|
7183
|
-
""
|
|
7184
|
-
].join("\n") : "";
|
|
7185
|
-
const executionSection = [
|
|
7186
|
-
"### Execution mode",
|
|
7187
|
-
"",
|
|
7188
|
-
`Mode: \`${execution.mode}\``,
|
|
7189
|
-
`Workspace: \`${execution.workspace}\``,
|
|
7190
|
-
execution.workspace === "none" ? "You are running in a scratch workspace with no repository checkout mounted. Do not assume git history or repo files are present unless the scenario provided them explicitly." : execution.workspace === "shared_mount" ? "You are running against the daemon shared mount. Treat any repository mutations as affecting the mounted checkout directly." : "You are running in a dedicated disposable git worktree isolated from the daemon shared checkout.",
|
|
7191
|
-
""
|
|
7192
|
-
].join("\n");
|
|
7193
|
-
const contextDisciplineSection = hasContext ? [
|
|
7194
|
-
"### Injected context discipline",
|
|
7429
|
+
const header = `# Run Eval Agent\n\nYou are running an evaluation scenario as variant \`${variantLabel}\`.\nTask id: \`${ctx.taskId}\``;
|
|
7430
|
+
const contextDiscipline = hasContext ? [
|
|
7431
|
+
"This task includes Injected Task Context supplied by the task",
|
|
7432
|
+
"creator. You MUST inspect it BEFORE you write solution files or",
|
|
7433
|
+
"draft your final answer — not after.",
|
|
7195
7434
|
"",
|
|
7196
|
-
"
|
|
7197
|
-
"
|
|
7198
|
-
"
|
|
7199
|
-
"
|
|
7200
|
-
|
|
7201
|
-
|
|
7202
|
-
"
|
|
7203
|
-
"
|
|
7204
|
-
""
|
|
7435
|
+
"Reconcile every constraint from that context **into the code path",
|
|
7436
|
+
"itself**: function bodies, control flow, transaction boundaries,",
|
|
7437
|
+
"guard clauses. Quoting a constraint back in a comment, a",
|
|
7438
|
+
"`// note:` line, the task summary, or the `verification` field is",
|
|
7439
|
+
"NOT following the task. If the constraint affects behavior, it",
|
|
7440
|
+
"must affect behavior.",
|
|
7441
|
+
hasInlineContext ? "For `context_inline`, your FIRST content-inspection step is a `read` of `/workspace/context-pack.md` before your first `write` call. The same content is also mirrored in `/workspace/AGENTS.md` and may be referenced from `/workspace/.claude/CLAUDE.md`." : "When the context is delivered as a skill, inspect it before solving.",
|
|
7442
|
+
"If the Injected Task Context contains repo- or workflow-specific",
|
|
7443
|
+
"rules, those rules override your generic instincts."
|
|
7205
7444
|
].join("\n") : "";
|
|
7206
|
-
const
|
|
7445
|
+
const inputFiles = scenario.inputFiles?.length ? scenario.inputFiles.map((f) => `- \`${f}\``).join("\n") : "";
|
|
7446
|
+
const verification = successCriteria ? buildSelfVerificationBlock(ctx.taskId) : "";
|
|
7447
|
+
const finalOutput = buildFinalOutputBlock({
|
|
7207
7448
|
taskType: "run_eval",
|
|
7208
7449
|
outputSchemaName: "RunEvalOutput",
|
|
7209
7450
|
shapeSketch: [
|
|
@@ -7223,17 +7464,41 @@ function buildRunEvalUserPrompt(input, ctx) {
|
|
|
7223
7464
|
"}"
|
|
7224
7465
|
].join("\n")
|
|
7225
7466
|
});
|
|
7226
|
-
return [
|
|
7227
|
-
|
|
7228
|
-
|
|
7229
|
-
|
|
7230
|
-
|
|
7231
|
-
|
|
7232
|
-
|
|
7233
|
-
|
|
7234
|
-
|
|
7235
|
-
|
|
7236
|
-
|
|
7467
|
+
return assembleTaskPrompt("run_eval", [
|
|
7468
|
+
{
|
|
7469
|
+
id: "run_eval.header",
|
|
7470
|
+
source: "header",
|
|
7471
|
+
body: header
|
|
7472
|
+
},
|
|
7473
|
+
{
|
|
7474
|
+
id: "run_eval.context_discipline",
|
|
7475
|
+
source: "discipline",
|
|
7476
|
+
header: "Injected Task Context",
|
|
7477
|
+
body: contextDiscipline
|
|
7478
|
+
},
|
|
7479
|
+
{
|
|
7480
|
+
id: "run_eval.scenario",
|
|
7481
|
+
source: "task_input",
|
|
7482
|
+
header: "Scenario",
|
|
7483
|
+
body: scenario.prompt
|
|
7484
|
+
},
|
|
7485
|
+
{
|
|
7486
|
+
id: "run_eval.input_files",
|
|
7487
|
+
source: "task_input",
|
|
7488
|
+
header: "Input files",
|
|
7489
|
+
body: inputFiles
|
|
7490
|
+
},
|
|
7491
|
+
{
|
|
7492
|
+
id: "run_eval.verification",
|
|
7493
|
+
source: "verification",
|
|
7494
|
+
body: verification
|
|
7495
|
+
},
|
|
7496
|
+
{
|
|
7497
|
+
id: "run_eval.final_output",
|
|
7498
|
+
source: "final_output",
|
|
7499
|
+
body: finalOutput
|
|
7500
|
+
}
|
|
7501
|
+
]);
|
|
7237
7502
|
}
|
|
7238
7503
|
//#endregion
|
|
7239
7504
|
//#region ../../libs/agent-runtime/src/prompts/index.ts
|
|
@@ -16843,7 +17108,7 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
16843
17108
|
});
|
|
16844
17109
|
let taskPrompt;
|
|
16845
17110
|
try {
|
|
16846
|
-
|
|
17111
|
+
const assembled = buildTaskUserPrompt(task, {
|
|
16847
17112
|
diaryId,
|
|
16848
17113
|
taskId: task.id,
|
|
16849
17114
|
workspace: {
|
|
@@ -16854,6 +17119,12 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
16854
17119
|
},
|
|
16855
17120
|
extras: opts.promptExtras
|
|
16856
17121
|
});
|
|
17122
|
+
taskPrompt = assembled.text;
|
|
17123
|
+
await emit("info", {
|
|
17124
|
+
event: "prompt_assembled",
|
|
17125
|
+
taskType: assembled.taskType,
|
|
17126
|
+
sections: assembled.trace
|
|
17127
|
+
});
|
|
16857
17128
|
} catch (err) {
|
|
16858
17129
|
const message = err instanceof Error ? err.message : String(err);
|
|
16859
17130
|
await emit("error", {
|
|
@@ -17116,8 +17387,8 @@ async function executePiTask(claimedTask, reporter, opts) {
|
|
|
17116
17387
|
}
|
|
17117
17388
|
else if (submitToolHandle) {
|
|
17118
17389
|
parseError = {
|
|
17119
|
-
code: "
|
|
17120
|
-
message: "Agent did not
|
|
17390
|
+
code: "submit_output_missing",
|
|
17391
|
+
message: "Agent did not satisfy the promised submit-output criterion: no valid task submit tool call was captured before the session ended."
|
|
17121
17392
|
};
|
|
17122
17393
|
await emit("error", {
|
|
17123
17394
|
message: parseError.message,
|