@axiom-lattice/core 3.0.6 → 3.0.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.mjs CHANGED
@@ -26109,7 +26109,7 @@ File content: ${files[key4]}`
26109
26109
  {
26110
26110
  dimension: "correctness",
26111
26111
  weight: 100,
26112
- description: "\u6574\u4F53\u6B63\u786E\u6027\uFF0C\u662F\u5426\u7B26\u5408\u9884\u671F\u8F93\u51FA\u63CF\u8FF0\u3002"
26112
+ description: "Overall correctness \u2014 whether the result matches the expected output description."
26113
26113
  }
26114
26114
  ];
26115
26115
  const evalRubrics = evalCase.eval.eval_rubrics && evalCase.eval.eval_rubrics.length > 0 ? evalCase.eval.eval_rubrics : defaultRubrics;
@@ -26123,53 +26123,53 @@ File content: ${files[key4]}`
26123
26123
  ${evalRubrics.map(
26124
26124
  (r) => `- **${r.dimension}**\uFF08\u6743\u91CD\uFF1A${r.weight}\uFF09\uFF1A${r.description}`
26125
26125
  ).join("\n")}`;
26126
- const testPrompt = `# \u89D2\u8272
26127
- \u4F60\u662F\u4E00\u540D\u8D44\u6DF1\u7684 AI Agent \u8BC4\u4F30\u4E13\u5BB6\uFF0C\u8D1F\u8D23\u6839\u636E\u9884\u8BBE\u7684\u6307\u6807\uFF08Rubrics\uFF09\u5BF9 Agent \u7684\u6267\u884C\u8FC7\u7A0B\u4E0E\u7ED3\u679C\u8FDB\u884C"\u9ED1\u76D2\u6D4B\u8BD5"\u5224\u5B9A\u3002
26126
+ const testPrompt = `# Role
26127
+ You are a senior AI Agent evaluation expert. Your job is to perform a "black-box test" judgment of the agent's execution process and results against the preset evaluation rubrics.
26128
26128
 
26129
- # \u8F93\u5165\u4FE1\u606F
26130
- \u6D4B\u8BD5\u6846\u67B6\u5C06\u4E3A\u4F60\u63D0\u4F9B\u4EE5\u4E0B\u4E94\u4E2A\u6838\u5FC3\u4E0A\u4E0B\u6587\uFF1A
26129
+ # Input Information
26130
+ The test framework provides you with the following five core contexts:
26131
26131
 
26132
- 1. **\u7528\u6237\u610F\u56FE\uFF08User Intent\uFF09**\uFF1A${evalCase.input.message}
26132
+ 1. **User Intent**: ${evalCase.input.message}
26133
26133
 
26134
- 2. **\u8F93\u5165\u6587\u4EF6\uFF08Input Files\uFF09**\uFF1A${testCaseFilesContent || "\u65E0"}
26134
+ 2. **Input Files**: ${testCaseFilesContent || "None"}
26135
26135
 
26136
- 3. **\u6267\u884C\u8FC7\u7A0B\uFF08Execution Trajectory\uFF0CAgent \u5168\u7A0B\u7684\u6D88\u606F/\u5DE5\u5177\u8C03\u7528\u8BB0\u5F55\uFF09**\uFF1A
26136
+ 3. **Execution Trajectory** (the agent's full message/tool-call record):
26137
26137
  ${trajectory}
26138
26138
 
26139
- 4. **\u6700\u7EC8\u8F93\u51FA\uFF08Final Output\uFF0CAgent \u7684\u6700\u540E\u4E00\u6761\u56DE\u590D\uFF09**\uFF1A
26139
+ 4. **Final Output** (the agent's last reply):
26140
26140
  ${finalOutput}
26141
26141
 
26142
- 5. **\u671F\u671B\u8F93\u51FA\u63CF\u8FF0\uFF08Expected Output Description\uFF09**\uFF1A${evalCase.eval.content_assertion}
26142
+ 5. **Expected Output Description**: ${evalCase.eval.content_assertion}
26143
26143
  ${rubricsSection}
26144
26144
 
26145
- # \u4EFB\u52A1
26146
- \u4F60\u5FC5\u987B\u4E25\u683C\u5BF9\u7167"\u8BC4\u4F30\u6307\u6807\uFF08Evaluation Rubrics\uFF09"\u4E2D\u7684\u6BCF\u4E00\u9879\u6307\u6807\uFF0C\u7ED3\u5408"\u6267\u884C\u8FC7\u7A0B"\u4E0E"\u6700\u7EC8\u8F93\u51FA"\uFF0C\u5206\u6790 Agent \u662F\u5426\u8FBE\u6807\u3002\u8BC4\u4F30\u65F6\u65E2\u8981\u68C0\u67E5\u6700\u7EC8\u7ED3\u679C\uFF0C\u4E5F\u8981\u68C0\u67E5\u8FC7\u7A0B\u4E2D\u662F\u5426\u6B63\u786E\u5B8C\u6210\u4E86\u5FC5\u8981\u7684\u6B65\u9AA4\uFF08\u5982\u5DE5\u5177\u8C03\u7528\u3001\u4FE1\u606F\u68C0\u7D22\u7B49\uFF09\u3002
26145
+ # Task
26146
+ You must strictly evaluate the agent against every rubric in the "Evaluation Rubrics" section, using both the "Execution Trajectory" and the "Final Output". Evaluate the final result AND whether the process correctly performed the required intermediate steps (tool calls, information retrieval, etc.).
26147
26147
 
26148
- # \u89C4\u5219
26149
- 1. **\u5BA2\u89C2\u6027**\uFF1A\u4EC5\u6839\u636E\u63D0\u4F9B\u7684\u4E0A\u4E0B\u6587\u5224\u5B9A\u3002\u5982\u679C\u6807\u51C6\u8981\u6C42"\u5305\u542B\u6570\u5B57"\uFF0C\u4F46\u8F93\u51FA\u53EA\u6709\u6587\u5B57\uFF0C\u5373\u4F7F\u8BED\u6C14\u518D\u597D\u4E5F\u5FC5\u987B\u6263\u5206\u3002
26150
- 2. **\u7ED3\u679C\u6821\u9A8C**\uFF1A\u5982\u679C"\u6700\u7EC8\u8F93\u51FA"\u4E2D\u7F3A\u5931\u9884\u671F\u7684\u5185\u5BB9\uFF0C\u6216\u5185\u5BB9\u4E0D\u7B26\u5408"\u8BC4\u4F30\u6307\u6807"\u4E2D\u7684\u6807\u51C6\uFF0C\u5BF9\u5E94\u7684\u6307\u6807\u5E94\u5224\u5B9A\u4E3A\u5931\u8D25\u3002
26151
- 3. **\u8FC7\u7A0B\u6821\u9A8C**\uFF1A\u5982\u679C"\u6267\u884C\u8FC7\u7A0B"\u663E\u793A Agent \u672A\u6267\u884C\u5FC5\u8981\u7684\u4E2D\u95F4\u6B65\u9AA4\uFF08\u5982\u5E94\u8C03\u7528\u5DE5\u5177\u800C\u672A\u8C03\u7528\uFF09\uFF0C\u5373\u4F7F\u6700\u7EC8\u8F93\u51FA\u770B\u4F3C\u5408\u7406\uFF0C\u4E5F\u5E94\u5728\u5BF9\u5E94\u6307\u6807\u4E0A\u6263\u5206\u3002
26152
- 4. **\u8BC1\u636E\u5BFC\u5411**\uFF1A\u5728\u7ED9\u51FA\u539F\u56E0\uFF08reason\uFF09\u65F6\uFF0C\u5FC5\u987B\u5F15\u7528\u6267\u884C\u8FC7\u7A0B\u6216\u6700\u7EC8\u8F93\u51FA\u4E2D\u7684\u5177\u4F53\u5185\u5BB9\u3002
26153
- 5. **\u52A0\u6743\u8BA1\u7B97**\uFF1A\u6700\u7EC8\u5206\u6570\u4E3A\u5404\u9879\u6307\u6807\u5F97\u5206\u4E0E\u5176\u6743\u91CD\u7684\u4E58\u79EF\u4E4B\u548C\uFF080-100\u5206\u5236\uFF09\u3002
26154
- 6. **HITL \u4E2D\u65AD\u5224\u5B9A**\uFF1A\u5982\u679C\u6267\u884C\u8FC7\u7A0B\u4E2D\u51FA\u73B0\u300CHITL \u6682\u505C\uFF1AAgent \u8BF7\u6C42\u4EBA\u5DE5\u8F93\u5165\u300D\u6761\u76EE\uFF0C\u8BF4\u660E Agent \u5728\u7B49\u5F85\u4EBA\u5DE5\u786E\u8BA4\u3002\u8BF7\u628A\u5B83\u5F53\u4F5C\u88AB\u6D4B\u4E1A\u52A1\u884C\u4E3A\u6765\u8BC4\u5224\uFF1A\u82E5\u671F\u671B\u8F93\u51FA\u8981\u6C42\u81EA\u4E3B\u5B8C\u6210\uFF08\u5982"\u65E0\u9700\u786E\u8BA4\u81EA\u52A8\u6267\u884C"\uFF09\uFF0C\u8BF7\u6C42\u4EBA\u5DE5\u8F93\u5165\u5E94\u5224\u5931\u8D25\uFF1B\u82E5\u671F\u671B\u8F93\u51FA\u8981\u6C42\u5148\u83B7\u5F97\u786E\u8BA4\u6216\u6279\u51C6\uFF08\u5982"\u6267\u884C\u524D\u5FC5\u987B\u8BF7\u6C42\u6279\u51C6"\uFF09\uFF0C\u8BF7\u6C42\u4EBA\u5DE5\u8F93\u5165\u662F\u6B63\u786E\u884C\u4E3A\uFF0C\u5E94\u7ED3\u5408\u5176\u65F6\u673A\u4E0E\u5185\u5BB9\u5224\u5B9A\u4E3A\u901A\u8FC7\u6216\u6309\u6307\u6807\u7ED9\u5206\u3002
26155
- 7. **HITL \u81EA\u52A8\u54CD\u5E94\u5224\u5B9A**\uFF1A\u5982\u679C\u300CHITL \u6682\u505C\u300D\u6761\u76EE\u4E4B\u540E\u51FA\u73B0\u300C\u5DF2\u81EA\u52A8\u54CD\u5E94\uFF08\u6D4B\u8BD5\u7B56\u7565 auto-approve/auto-reject/canned-response\uFF09\u300D\u6761\u76EE\uFF0C\u8BF4\u660E\u6D4B\u8BD5\u6846\u67B6\u6CE8\u5165\u4E86\u4EBA\u5DE5\u56DE\u590D\u3001\u6D41\u7A0B\u5DF2\u7EE7\u7EED\u2014\u2014\u8BF7\u6309**\u5B8C\u6574\u6D41\u7A0B**\u8BC4\u5224\u6682\u505C\u4E4B\u540E\u7684\u884C\u4E3A\uFF08\u5982\u6279\u51C6\u540E\u662F\u5426\u6B63\u786E\u6267\u884C\u4E86\u64CD\u4F5C\uFF09\uFF0C\u5E76\u6838\u5BF9\u81EA\u52A8\u54CD\u5E94\u5185\u5BB9\u662F\u5426\u7B26\u5408\u4EBA\u5DE5\u56DE\u590D\u7684\u5408\u7406\u9884\u671F\u3002
26148
+ # Rules
26149
+ 1. **Objectivity**: Judge solely from the provided context. If the standard requires "contains a number" but the output has only text, points must be deducted even if the tone is good.
26150
+ 2. **Result verification**: If the "Final Output" is missing expected content, or does not meet the criteria in the "Evaluation Rubrics", the corresponding rubric must be marked as failed.
26151
+ 3. **Process verification**: If the "Execution Trajectory" shows the agent did not perform a necessary intermediate step (e.g., should have called a tool but did not), deduct points on the corresponding rubric even if the final output looks plausible.
26152
+ 4. **Evidence-based**: When giving a reason, you must quote specific content from the execution trajectory or final output.
26153
+ 5. **Weighted scoring**: The final score is the weighted sum of the rubric scores (on a 0-100 scale).
26154
+ 6. **HITL interrupt judgment**: If the trajectory contains a "HITL pause: agent requested human input" entry, the agent is waiting for human confirmation. Treat this as the business behavior under test: if the expected output requires autonomous completion (e.g., "execute automatically without confirmation"), requesting human input should be judged a failure; if the expected output requires confirmation or approval first (e.g., "must request approval before executing"), requesting human input is correct behavior \u2014 judge its timing and content, passing or scoring according to the rubrics.
26155
+ 7. **HITL auto-response judgment**: If a "HITL pause" entry is followed by an "auto-responded (test policy auto-approve/auto-reject/canned-response)" entry, the test framework injected a human reply and the flow continued \u2014 evaluate the behavior AFTER the pause as the complete flow (e.g., whether the operation was correctly executed after approval), and check whether the auto-response content matches a reasonable human reply.
26156
26156
 
26157
- # \u8F93\u51FA\u683C\u5F0F\uFF08\u4EC5JSON\uFF09
26158
- \u4F60\u5FC5\u987B\u4EC5\u4EE5 JSON \u683C\u5F0F\u56DE\u590D\uFF0C\u7ED3\u6784\u5982\u4E0B\uFF1A
26157
+ # Output Format (JSON only)
26158
+ You MUST reply with JSON only, using this structure:
26159
26159
  {
26160
26160
  "pass": true | false,
26161
26161
  "final_score": number,
26162
26162
  "dimension_results": [
26163
26163
  {
26164
- "name": "\u6307\u6807\u540D\u79F0",
26164
+ "name": "rubric name",
26165
26165
  "score": number,
26166
- "reason": "\u5177\u4F53\u7684\u6263\u5206\u6216\u7ED9\u5206\u7406\u7531\uFF0C\u9700\u5F15\u7528\u8BC1\u636E"
26166
+ "reason": "specific reason for deduction or credit, citing evidence"
26167
26167
  }
26168
26168
  ],
26169
- "summary": "\u5BF9 Agent \u8868\u73B0\u7684\u6574\u4F53\u8BC4\u4EF7"
26169
+ "summary": "overall evaluation of the agent's performance"
26170
26170
  }
26171
26171
 
26172
- \u6CE8\u610F\uFF1A\u5982\u679C final_score >= 80 \u4E14\u6CA1\u6709\u81F4\u547D\u6027\u9519\u8BEF\uFF0Cpass \u5E94\u4E3A true\uFF1B\u5426\u5219\u4E3A false\u3002`;
26172
+ Note: if final_score >= 80 and there are no fatal errors, pass should be true; otherwise false.`;
26173
26173
  this.lastTestPrompt = testPrompt;
26174
26174
  const judgeThreadId = v44();
26175
26175
  this.lastJudgeThreadId = judgeThreadId;
@@ -26552,6 +26552,32 @@ var LatticeEvalSuite = class {
26552
26552
  import { AgentType as AgentType6 } from "@axiom-lattice/protocols";
26553
26553
  import { HumanMessage as HumanMessage5 } from "@langchain/core/messages";
26554
26554
  import { v4 as uuidv46 } from "uuid";
26555
+ var DEFAULT_CALIBRATION_PROBES = [
26556
+ {
26557
+ id: "catch-up-correct",
26558
+ task: "A train leaves Station A at 10:00 AM traveling at 60 mph. A second train leaves Station A at 10:30 AM traveling at 90 mph on the same route. The final answer must be the time (HH:MM) at which the second train catches up to the first.",
26559
+ finalOutput: "11:30",
26560
+ expectedPass: true
26561
+ },
26562
+ {
26563
+ id: "catch-up-wrong",
26564
+ task: "A train leaves Station A at 10:00 AM traveling at 60 mph. A second train leaves Station A at 10:30 AM traveling at 90 mph on the same route. The final answer must be the time (HH:MM) at which the second train catches up to the first.",
26565
+ finalOutput: "12:30",
26566
+ expectedPass: false
26567
+ },
26568
+ {
26569
+ id: "discount-tax-correct",
26570
+ task: "A store offers a 20% discount on an item priced at $150, then adds 8% sales tax to the discounted price. The final answer must be the total price in dollars.",
26571
+ finalOutput: "129.6",
26572
+ expectedPass: true
26573
+ },
26574
+ {
26575
+ id: "discount-tax-wrong",
26576
+ task: "A store offers a 20% discount on an item priced at $150, then adds 8% sales tax to the discounted price. The final answer must be the total price in dollars.",
26577
+ finalOutput: "162",
26578
+ expectedPass: false
26579
+ }
26580
+ ];
26555
26581
  var LatticeEvalProject = class {
26556
26582
  constructor(project, onCaseComplete) {
26557
26583
  this.suites = /* @__PURE__ */ new Map();
@@ -26659,21 +26685,22 @@ var LatticeEvalProject = class {
26659
26685
  }
26660
26686
  /**
26661
26687
  * Verify the judge agent can produce parseable, correct verdicts
26662
- * before committing to a full run. Uses two known-answer cases
26663
- * (one expected PASS, one expected FAIL) to catch broken judges.
26688
+ * before committing to a full run. Uses known-answer probes (default
26689
+ * multi-step reasoning set, overridable per project via
26690
+ * judge_agent_config.calibration_cases) to catch broken judges.
26691
+ * Checks both the pass verdict and the final_score direction.
26664
26692
  */
26665
26693
  async calibrateJudge() {
26666
26694
  const tenantId2 = this.project.lattice_server_config.tenant_id || "default";
26667
26695
  const judgeAgent = await getAgentClient(tenantId2, this.judgeAgentKey);
26668
- const cases = [
26669
- { output: "7", expected: "7", expectedPass: true },
26670
- { output: "7", expected: "999", expectedPass: false }
26671
- ];
26672
- for (const c of cases) {
26673
- const prompt = `\u4F60\u662F\u8BC4\u4F30\u4E13\u5BB6\u3002\u5224\u5B9A\u6700\u7EC8\u8F93\u51FA\u662F\u5426\u7B26\u5408\u671F\u671B\u3002
26674
- \u6700\u7EC8\u8F93\u51FA\uFF1A${c.output}
26675
- \u671F\u671B\u8F93\u51FA\uFF1A${c.expected}
26676
- \u4EC5\u8F93\u51FA JSON\uFF1A{"pass": true|false, "final_score": 0-100, "summary": "\u7406\u7531"}`;
26696
+ const probes = this.project.judge_agent_config.calibration_cases && this.project.judge_agent_config.calibration_cases.length > 0 ? this.project.judge_agent_config.calibration_cases : DEFAULT_CALIBRATION_PROBES;
26697
+ for (const probe of probes) {
26698
+ const trajectorySection = probe.trajectory ? `
26699
+ Trajectory: ${probe.trajectory}` : "";
26700
+ const prompt = `You are an evaluation expert. Judge whether the final output meets the expected output description.
26701
+ Expected output description: ${probe.task}
26702
+ Final output: ${probe.finalOutput}${trajectorySection}
26703
+ Respond with JSON only: {"pass": true|false, "final_score": 0-100, "summary": "reason"}`;
26677
26704
  let raw = "";
26678
26705
  let invokeError = null;
26679
26706
  for (let attempt = 0; attempt < 2; attempt++) {
@@ -26698,13 +26725,24 @@ var LatticeEvalProject = class {
26698
26725
  return { ok: false, reason: `Calibration output unparseable: ${parsed.error}`, bypassed: true };
26699
26726
  }
26700
26727
  const actualPass = parsed.pass !== void 0 ? parsed.pass : (parsed.final_score ?? 0) >= 80;
26701
- if (actualPass !== c.expectedPass) {
26728
+ if (actualPass !== probe.expectedPass) {
26702
26729
  return {
26703
26730
  ok: false,
26704
- reason: `Calibration mismatch: output="${c.output}" expected="${c.expected}" \u2014 judge said ${actualPass ? "PASS" : "FAIL"}, expected ${c.expectedPass ? "PASS" : "FAIL"}`,
26731
+ reason: `Calibration mismatch (probe=${probe.id}): expected ${probe.expectedPass ? "PASS" : "FAIL"}, judge said ${actualPass ? "PASS" : "FAIL"}`,
26705
26732
  bypassed: true
26706
26733
  };
26707
26734
  }
26735
+ if (parsed.final_score !== void 0) {
26736
+ const minScore = probe.expectedScoreMin ?? 80;
26737
+ const scoreOk = probe.expectedPass ? parsed.final_score >= minScore : parsed.final_score < minScore;
26738
+ if (!scoreOk) {
26739
+ return {
26740
+ ok: false,
26741
+ reason: `Calibration score mismatch (probe=${probe.id}): expected ${probe.expectedPass ? "final_score >= " + minScore : "final_score < " + minScore}, judge gave ${parsed.final_score}`,
26742
+ bypassed: true
26743
+ };
26744
+ }
26745
+ }
26708
26746
  }
26709
26747
  return { ok: true };
26710
26748
  }
@@ -31096,6 +31134,7 @@ export {
31096
31134
  ConsoleLoggerClient,
31097
31135
  CustomMetricsClient,
31098
31136
  CustomMiddlewareRegistry,
31137
+ DEFAULT_CALIBRATION_PROBES,
31099
31138
  DaytonaInstance,
31100
31139
  DaytonaProvider,
31101
31140
  DefaultScheduleClient,