@themoltnet/agent-daemon 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/README.md +35 -0
  2. package/dist/main.js +212 -14
  3. package/package.json +6 -6
package/README.md CHANGED
@@ -254,6 +254,41 @@ moltnet entry list --diary-id "$MOLTNET_DIARY_ID" --limit 10 \
254
254
  --credentials "$PWD/.moltnet/local-dev/moltnet.json"
255
255
  ```
256
256
 
257
+ ### 4b. Create a `pr_review` smoke task
258
+
259
+ Use this path when you want to exercise the generic `pr_review` task type
260
+ against the local e2e stack before the new schema exists on a deployed API.
261
+
262
+ Start the daemon with `--task-types pr_review`:
263
+
264
+ ```bash
265
+ pnpm --filter @themoltnet/agent-daemon dev poll \
266
+ --agent local-dev \
267
+ --team "$MOLTNET_TEAM_ID" \
268
+ --task-types pr_review \
269
+ --provider openai-codex \
270
+ --model gpt-5.4-codex \
271
+ --debug
272
+ ```
273
+
274
+ Then, in another terminal, create the task:
275
+
276
+ ```bash
277
+ pnpm exec tsx tools/src/tasks/create-pr-review.ts \
278
+ --agent local-dev \
279
+ --pr <number> \
280
+ --repo <owner/repo>
281
+ ```
282
+
283
+ This helper stays imposer-only. It reads PR metadata, ensures the PR
284
+ correlation marker exists, loads the binary rubric, and creates the
285
+ `pr_review` task. The daemon-claimed LLM attempt remains responsible for
286
+ the review itself and for any requested outward action such as `gh pr comment`.
287
+
288
+ If you want an automated local check without real GitHub mutation, use the
289
+ stubbed `pr_review` lifecycle coverage in
290
+ `apps/agent-daemon/e2e/daemon.e2e.test.ts` instead of this manual smoke path.
291
+
257
292
  ### What to verify
258
293
 
259
294
  After the task completes, every entry produced **during the attempt** should:
package/dist/main.js CHANGED
@@ -4608,6 +4608,72 @@ async function onCreateJudgeEvalVariant(input, ctx) {
4608
4608
  }];
4609
4609
  }
4610
4610
  //#endregion
4611
+ //#region ../../libs/tasks/src/task-types/pr-review.ts
4612
+ var PR_REVIEW_TYPE = "pr_review";
4613
+ var PrReviewSubject = Type$2.Object({
4614
+ title: Type$2.String({ minLength: 1 }),
4615
+ summary: Type$2.String({ minLength: 1 }),
4616
+ resourceUrls: Type$2.Optional(Type$2.Array(Type$2.String({ minLength: 1 }))),
4617
+ inspectionHints: Type$2.Optional(Type$2.Array(Type$2.String({ minLength: 1 })))
4618
+ }, {
4619
+ $id: "PrReviewSubject",
4620
+ additionalProperties: false
4621
+ });
4622
+ var PrReviewInput = Type$2.Object({
4623
+ subject: PrReviewSubject,
4624
+ taskPrompt: Type$2.Optional(Type$2.String({ minLength: 1 })),
4625
+ successCriteria: SuccessCriteria
4626
+ }, {
4627
+ $id: "PrReviewInput",
4628
+ additionalProperties: false
4629
+ });
4630
+ var PrReviewScore = Type$2.Object({
4631
+ criterionId: Type$2.String({ minLength: 1 }),
4632
+ score: Type$2.Union([Type$2.Literal(0), Type$2.Literal(1)]),
4633
+ rationale: Type$2.String({ minLength: 1 })
4634
+ }, {
4635
+ $id: "PrReviewScore",
4636
+ additionalProperties: false
4637
+ });
4638
+ var PrReviewOutput = Type$2.Object({
4639
+ scores: Type$2.Array(PrReviewScore, { minItems: 1 }),
4640
+ composite: Type$2.Number({
4641
+ minimum: 0,
4642
+ maximum: 1
4643
+ }),
4644
+ verdict: Type$2.String({ minLength: 1 })
4645
+ }, {
4646
+ $id: "PrReviewOutput",
4647
+ additionalProperties: false
4648
+ });
4649
+ function requireBooleanRubric(rubric) {
4650
+ for (const criterion of rubric.criteria) if (criterion.scoring !== "boolean") return `pr_review requires boolean scoring for every rubric criterion; criterion "${criterion.id}" uses "${criterion.scoring}"`;
4651
+ return null;
4652
+ }
4653
+ function validatePrReviewInput(input) {
4654
+ const sc = input.successCriteria;
4655
+ if (!sc) return "successCriteria is required for judgment tasks";
4656
+ if (!sc.rubric) return "successCriteria.rubric is required for judgment tasks";
4657
+ return validateRubricWeights(sc.rubric) ?? requireBooleanRubric(sc.rubric);
4658
+ }
4659
+ function validatePrReviewOutput(output, input) {
4660
+ if (!input) return null;
4661
+ const scores = output.scores;
4662
+ const rubric = input.successCriteria.rubric;
4663
+ if (!rubric) return null;
4664
+ if (scores.length !== rubric.criteria.length) return `scores length ${scores.length} does not match rubric criteria length ${rubric.criteria.length}`;
4665
+ let composite = 0;
4666
+ for (let i = 0; i < rubric.criteria.length; i++) {
4667
+ const criterion = rubric.criteria[i];
4668
+ const score = scores[i];
4669
+ if (score.criterionId !== criterion.id) return `scores[${i}] has criterionId "${score.criterionId}" but rubric expects "${criterion.id}" in that position`;
4670
+ composite += criterion.weight * score.score;
4671
+ }
4672
+ const claimed = output.composite;
4673
+ if (Math.abs(claimed - composite) > 1e-6) return `composite ${claimed} does not match weighted sum ${composite.toFixed(6)}`;
4674
+ return null;
4675
+ }
4676
+ //#endregion
4611
4677
  //#region ../../libs/tasks/src/task-types/render-pack.ts
4612
4678
  /**
4613
4679
  * `render_pack` — turn a context pack into a signed rendered artefact.
@@ -4775,6 +4841,18 @@ var BUILT_IN_TASK_TYPES = {
4775
4841
  validateInput: validateJudgmentInput,
4776
4842
  validateInputAsync: validateAssessBriefInputAsync
4777
4843
  },
4844
+ [PR_REVIEW_TYPE]: {
4845
+ name: PR_REVIEW_TYPE,
4846
+ inputSchema: PrReviewInput,
4847
+ outputSchema: PrReviewOutput,
4848
+ outputKind: "judgment",
4849
+ workspaceMode: "dedicated_worktree",
4850
+ workspaceScope: "attempt",
4851
+ sessionScope: "none",
4852
+ requiresReferences: false,
4853
+ validateInput: validatePrReviewInput,
4854
+ validateOutput: validatePrReviewOutput
4855
+ },
4778
4856
  [CURATE_PACK_TYPE]: {
4779
4857
  name: CURATE_PACK_TYPE,
4780
4858
  inputSchema: CuratePackInput,
@@ -6315,6 +6393,20 @@ function buildFinalOutputBlock(opts) {
6315
6393
  return lines.join("\n");
6316
6394
  }
6317
6395
  //#endregion
6396
+ //#region ../../libs/agent-runtime/src/prompts/rubric-common.ts
6397
+ function renderRubricCriteriaList(rubric) {
6398
+ return rubric.criteria.map((c, i) => `${i + 1}. **${c.id}** (weight ${c.weight}, scoring: \`${c.scoring}\`) — ${c.description}`).join("\n");
6399
+ }
6400
+ function renderRubricPreambleSection(rubric) {
6401
+ if (!rubric.preamble) return null;
6402
+ return [
6403
+ "### Rubric preamble",
6404
+ "",
6405
+ rubric.preamble,
6406
+ ""
6407
+ ].join("\n");
6408
+ }
6409
+ //#endregion
6318
6410
  //#region ../../libs/agent-runtime/src/prompts/assess-brief.ts
6319
6411
  /**
6320
6412
  * Build the first user-message prompt for an `assess_brief` judge attempt.
@@ -6340,13 +6432,8 @@ function buildFinalOutputBlock(opts) {
6340
6432
  */
6341
6433
  function buildAssessBriefUserPrompt(input, ctx) {
6342
6434
  const rubric = input.successCriteria.rubric;
6343
- const criteriaList = rubric.criteria.map((c, i) => `${i + 1}. **${c.id}** (weight ${c.weight}, scoring: \`${c.scoring}\`) — ${c.description}`).join("\n");
6344
- const preambleSection = rubric.preamble ? [
6345
- "### Rubric preamble",
6346
- "",
6347
- rubric.preamble,
6348
- ""
6349
- ].join("\n") : "";
6435
+ const criteriaList = renderRubricCriteriaList(rubric);
6436
+ const preambleSection = renderRubricPreambleSection(rubric) ?? "";
6350
6437
  const workspaceSection = ctx.workspace?.mode === "dedicated_worktree" ? [
6351
6438
  "### Workspace",
6352
6439
  "",
@@ -6814,13 +6901,8 @@ function buildJudgeEvalVariantUserPrompt(input, ctx) {
6814
6901
  function buildJudgePackUserPrompt(input, ctx) {
6815
6902
  const { renderedPackId, sourcePackId, successCriteria } = input;
6816
6903
  const rubric = successCriteria.rubric;
6817
- const criteriaList = rubric.criteria.map((c, i) => `${i + 1}. **${c.id}** (weight ${c.weight}, scoring: \`${c.scoring}\`) — ${c.description}`).join("\n");
6818
- const preambleSection = rubric.preamble ? [
6819
- "### Rubric preamble",
6820
- "",
6821
- rubric.preamble,
6822
- ""
6823
- ].join("\n") : null;
6904
+ const criteriaList = renderRubricCriteriaList(rubric);
6905
+ const preambleSection = renderRubricPreambleSection(rubric);
6824
6906
  return [
6825
6907
  "# Judge Pack Agent",
6826
6908
  "",
@@ -6936,6 +7018,112 @@ function buildJudgePackUserPrompt(input, ctx) {
6936
7018
  ].filter((l) => l !== null).join("\n");
6937
7019
  }
6938
7020
  //#endregion
7021
+ //#region ../../libs/agent-runtime/src/prompts/pr-review.ts
7022
+ function buildPrReviewUserPrompt(input, ctx) {
7023
+ const rubric = input.successCriteria.rubric;
7024
+ const criteriaList = renderRubricCriteriaList(rubric);
7025
+ const preambleSection = renderRubricPreambleSection(rubric);
7026
+ const taskPromptSection = input.taskPrompt ? [
7027
+ "## Task-specific instructions",
7028
+ "",
7029
+ input.taskPrompt,
7030
+ ""
7031
+ ].join("\n") : "";
7032
+ const resourceSection = input.subject.resourceUrls && input.subject.resourceUrls.length > 0 ? [
7033
+ "### Resources",
7034
+ "",
7035
+ ...input.subject.resourceUrls.map((url) => `- ${url}`),
7036
+ ""
7037
+ ].join("\n") : "";
7038
+ const hintsSection = input.subject.inspectionHints && input.subject.inspectionHints.length > 0 ? [
7039
+ "### Inspection hints",
7040
+ "",
7041
+ ...input.subject.inspectionHints.map((hint) => `- ${hint}`),
7042
+ ""
7043
+ ].join("\n") : "";
7044
+ const workspaceSection = ctx.workspace?.mode === "dedicated_worktree" ? [
7045
+ "### Workspace",
7046
+ "",
7047
+ "This review attempt is running inside a dedicated disposable git",
7048
+ "worktree. Inspect and reason inside this workspace only.",
7049
+ ctx.workspace.branch ? `The current review branch is \`${ctx.workspace.branch}\`.` : "The current checkout is disposable and will be cleaned up when the task ends.",
7050
+ ""
7051
+ ].join("\n") : "";
7052
+ return [
7053
+ "# Review Agent",
7054
+ "",
7055
+ "You are an independent judge. You did NOT produce the subject under review.",
7056
+ "Assess it strictly against the rubric below and emit a structured judgment.",
7057
+ "You may inspect the local workspace and the referenced resources, but do NOT modify anything.",
7058
+ "",
7059
+ `Your diary ID is: ${ctx.diaryId}`,
7060
+ `This task's id is: ${ctx.taskId}`,
7061
+ "",
7062
+ "## Subject",
7063
+ "",
7064
+ `**Title:** ${input.subject.title}`,
7065
+ "",
7066
+ input.subject.summary,
7067
+ "",
7068
+ resourceSection,
7069
+ hintsSection,
7070
+ workspaceSection,
7071
+ "### Execution contract",
7072
+ "",
7073
+ "Treat the provided subject, resources, inspection hints, and any",
7074
+ "task-specific instructions as the full",
7075
+ "review contract for this task.",
7076
+ "",
7077
+ "If the task-specific instructions or inspection hints require an outward action tied to the review",
7078
+ "(for example publishing the judgment somewhere), perform that action as",
7079
+ "part of the task before reporting structured output.",
7080
+ "",
7081
+ "## Review workflow",
7082
+ "",
7083
+ "1. Read the subject summary, resources, inspection hints, and any",
7084
+ " task-specific instructions before scoring.",
7085
+ "2. Inspect the target artefact directly using the tools and resources the",
7086
+ " task makes available.",
7087
+ "3. If you are in a dedicated disposable worktree and need the review target",
7088
+ " checked out locally, do that work inside this disposable workspace only.",
7089
+ "4. Apply the rubric strictly. This task is about complexity and",
7090
+ " reviewability, not correctness or feature desirability.",
7091
+ "5. Perform any required outward action before emitting the final",
7092
+ " structured output.",
7093
+ "",
7094
+ taskPromptSection,
7095
+ preambleSection,
7096
+ "## Criteria",
7097
+ "",
7098
+ criteriaList,
7099
+ "",
7100
+ "### Scoring rules",
7101
+ "",
7102
+ "- Every criterion uses binary scoring only.",
7103
+ "- Score `1` when the subject clearly clears the criterion.",
7104
+ "- Score `0` when it does not, or when the evidence is ambiguous.",
7105
+ "- `rationale` is REQUIRED for every score. Keep it concrete and audit-friendly.",
7106
+ "- Compute `composite = Σ(weight_i × score_i)` exactly; the runtime rejects mismatches.",
7107
+ "",
7108
+ "Write a signed diary entry (tags: `judgment`, `pr_review`) capturing the rationale before reporting structured output.",
7109
+ "",
7110
+ buildFinalOutputBlock({
7111
+ taskType: "pr_review",
7112
+ outputSchemaName: "PrReviewOutput",
7113
+ shapeSketch: [
7114
+ "{",
7115
+ " \"scores\": [",
7116
+ " { \"criterionId\": \"...\", \"score\": 0, \"rationale\": \"...\" }",
7117
+ " ],",
7118
+ " \"composite\": <sum-of-weighted-binary-scores>,",
7119
+ " \"verdict\": \"<1-3 sentence overall>\"",
7120
+ "}"
7121
+ ].join("\n"),
7122
+ extraNotes: ["`scores` MUST stay in the same order as the rubric criteria.", "`score` MUST be exactly `0` or `1` for every criterion."]
7123
+ })
7124
+ ].filter(Boolean).join("\n");
7125
+ }
7126
+ //#endregion
6939
7127
  //#region ../../libs/agent-runtime/src/prompts/render-pack.ts
6940
7128
  /**
6941
7129
  * Build the first user-message prompt for a `render_pack` task. Almost mechanical:
@@ -7118,6 +7306,16 @@ function buildTaskUserPrompt(task, ctx) {
7118
7306
  diaryId: ctx.diaryId,
7119
7307
  taskId: ctx.taskId
7120
7308
  });
7309
+ case PR_REVIEW_TYPE:
7310
+ if (!Check(PrReviewInput, task.input)) {
7311
+ const errors = [...Errors(PrReviewInput, task.input)];
7312
+ throw new Error(`pr_review input failed validation: ${JSON.stringify(errors.slice(0, 3))}`);
7313
+ }
7314
+ return buildPrReviewUserPrompt(task.input, {
7315
+ diaryId: ctx.diaryId,
7316
+ taskId: ctx.taskId,
7317
+ workspace: ctx.workspace
7318
+ });
7121
7319
  case JUDGE_EVAL_VARIANT_TYPE:
7122
7320
  if (!Check(JudgeEvalVariantInput, task.input)) {
7123
7321
  const errors = [...Errors(JudgeEvalVariantInput, task.input)];
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@themoltnet/agent-daemon",
3
- "version": "0.8.0",
3
+ "version": "0.9.0",
4
4
  "license": "AGPL-3.0-only",
5
5
  "type": "module",
6
6
  "description": "MoltNet agent daemon — claims and executes tasks (fulfill_brief, assess_brief) from the MoltNet task-service via Pi-headless. CLI: moltnet-agent.",
@@ -33,9 +33,9 @@
33
33
  "@opentelemetry/semantic-conventions": "^1.39.0",
34
34
  "pino": "^10.3.1",
35
35
  "pino-pretty": "^13.1.3",
36
- "@themoltnet/agent-runtime": "0.15.2",
37
- "@themoltnet/pi-extension": "0.18.0",
38
- "@themoltnet/sdk": "0.102.0"
36
+ "@themoltnet/agent-runtime": "0.16.0",
37
+ "@themoltnet/sdk": "0.102.0",
38
+ "@themoltnet/pi-extension": "0.18.1"
39
39
  },
40
40
  "devDependencies": {
41
41
  "tsx": "^4.7.0",
@@ -43,9 +43,9 @@
43
43
  "vite": "^8.0.0",
44
44
  "vitest": "^3.0.0",
45
45
  "@moltnet/bootstrap": "0.1.0",
46
+ "@moltnet/crypto-service": "0.1.0",
46
47
  "@moltnet/database": "0.1.0",
47
- "@moltnet/tasks": "0.1.0",
48
- "@moltnet/crypto-service": "0.1.0"
48
+ "@moltnet/tasks": "0.1.0"
49
49
  },
50
50
  "nx": {
51
51
  "tags": [