npm - @tangle-network/agent-eval - Versions diffs - 0.29.1 → 0.31.0 - Mend

@tangle-network/agent-eval 0.29.1 → 0.31.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (80) hide show

package/dist/{baseline-BwdCXUS8.d.ts → baseline-4R5deP0N.d.ts} RENAMED Viewed

@@ -1,4 +1,4 @@
-import { T as TraceStore } from './store-BP5be6s7.js';
+import { T as TraceStore } from './store-Db2Bv8Cf.js';
 /**
  * Tool-use metrics — derived purely from trace data.

package/dist/benchmarks/index.d.ts CHANGED Viewed

@@ -1,3 +1,3 @@
-export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as deterministicSplit, e as routing } from '../index--fVrWDiR.js';
-import '../run-record-CqzahIbx.js';
-import '../errors-BZ9sTdz7.js';
+export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as deterministicSplit, e as routing } from '../index-TVjRYWRm.js';
+import '../run-record-nYf9x2hU.js';
+import '../errors-mje_cKOs.js';

package/dist/builder-eval/index.d.ts CHANGED Viewed

@@ -1,6 +1,6 @@
-import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult, T as TestGradedScenario, b as TestGradedRunResult } from '../test-graded-scenario-BJ54PDan.js';
-import { T as TraceEmitter } from '../emitter-BqjeOvJh.js';
-import { T as TraceStore, R as Run } from '../store-BP5be6s7.js';
+import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult, T as TestGradedScenario, b as TestGradedRunResult } from '../test-graded-scenario-B2kWEdh9.js';
+import { T as TraceEmitter } from '../emitter-DP_cSSiw.js';
+import { T as TraceStore, R as Run } from '../store-Db2Bv8Cf.js';
 /**
  * BuilderSession — ties a builder-of-builders workflow together.

package/dist/builder-eval/index.js CHANGED Viewed

@@ -1,7 +1,7 @@
 import {
   SandboxHarness,
   runTestGradedScenario
-} from "../chunk-QHF6EQKK.js";
+} from "../chunk-YTMXBHFM.js";
 import {
   judgeSpans
 } from "../chunk-47X6LRCE.js";
@@ -9,7 +9,7 @@ import "../chunk-5BKGXME7.js";
 import {
   TraceEmitter
 } from "../chunk-TVVP3ZZQ.js";
-import "../chunk-NG236HPC.js";
+import "../chunk-QYJT52YW.js";
 import "../chunk-PZ5AY32C.js";
 // src/builder-eval/builder-session.ts

package/dist/{chunk-R5UQJNKC.js → chunk-4L3WJXQJ.js} RENAMED Viewed

@@ -1,6 +1,6 @@
 import {
   ValidationError
-} from "./chunk-NG236HPC.js";
+} from "./chunk-QYJT52YW.js";
 // src/judge-calibration.ts
 function calibrateJudge(golden, candidate) {
@@ -719,4 +719,4 @@ export {
   corpusInterRaterAgreement,
   corpusInterRaterAgreementFromJudgeScores
 };
-//# sourceMappingURL=chunk-R5UQJNKC.js.map
+//# sourceMappingURL=chunk-4L3WJXQJ.js.map

package/dist/{chunk-RUI6SIHY.js → chunk-75ZREHD7.js} RENAMED Viewed

@@ -1,13 +1,13 @@
 import {
   assertLlmRoute
-} from "./chunk-4S4BM3QQ.js";
+} from "./chunk-M6RZ5LJN.js";
 import {
   researchReport
-} from "./chunk-5AKPEK5L.js";
+} from "./chunk-CXJOVDJR.js";
 import {
   RunIntegrityError,
   assertRunCaptured
-} from "./chunk-KTGTIOFD.js";
+} from "./chunk-UBPIXOC4.js";
 import {
   FileSystemRawProviderSink
 } from "./chunk-PC4UYEBM.js";
@@ -284,4 +284,4 @@ function defaultRunId(params) {
 export {
   runEvalCampaign
 };
-//# sourceMappingURL=chunk-RUI6SIHY.js.map
+//# sourceMappingURL=chunk-75ZREHD7.js.map

package/dist/{chunk-5AKPEK5L.js → chunk-CXJOVDJR.js} RENAMED Viewed

@@ -2,7 +2,7 @@ import {
   cohensD,
   confidenceInterval,
   wilcoxonSignedRank
-} from "./chunk-R5UQJNKC.js";
+} from "./chunk-4L3WJXQJ.js";
 import {
   canonicalize,
   hashJson
@@ -1047,4 +1047,4 @@ export {
   RESEARCH_REPORT_HARD_PAIR_FLOOR,
   researchReport
 };
-//# sourceMappingURL=chunk-5AKPEK5L.js.map
+//# sourceMappingURL=chunk-CXJOVDJR.js.map

package/dist/{chunk-K33INZHH.js → chunk-GVQT44CS.js} RENAMED Viewed

@@ -1,6 +1,6 @@
 import {
   cohensD
-} from "./chunk-R5UQJNKC.js";
+} from "./chunk-4L3WJXQJ.js";
 import {
   argHash,
   groupBy,
@@ -615,4 +615,4 @@ export {
   iqr,
   welchsTTest
 };
-//# sourceMappingURL=chunk-K33INZHH.js.map
+//# sourceMappingURL=chunk-GVQT44CS.js.map

package/dist/{chunk-UW4NOOZI.js → chunk-HIO4UIS5.js} RENAMED Viewed

@@ -5,7 +5,7 @@ import {
 import {
   NotFoundError,
   ReplayError
-} from "./chunk-NG236HPC.js";
+} from "./chunk-QYJT52YW.js";
 // src/trace-analyst/prompts.ts
 var TRACE_ANALYST_ACTOR_DESCRIPTION = `You answer questions about an OTLP-shaped JSONL trace dataset using the trace tools provided in the \`traces\` namespace.
@@ -986,6 +986,302 @@ function normalizeRecordArray(value) {
   );
 }
+// src/trace-analyst/hook.ts
+var DEFAULT_QUESTION = "Summarise what happened in this run. Surface any failure modes, surprising findings, or evidence that the run's verdict is wrong.";
+function traceAnalystOnRunComplete(opts) {
+  return async (ctx) => {
+    if (opts.shouldRun && !opts.shouldRun(ctx)) return;
+    const source = opts.analyze.source;
+    if (source === void 0) {
+      await ctx.store.appendEvent({
+        eventId: `analyst-skip-${ctx.runId}`,
+        runId: ctx.runId,
+        kind: "log",
+        timestamp: Date.now(),
+        payload: { source: "trace_analyst_hook", reason: "no source configured" }
+      });
+      return;
+    }
+    const result = await analyzeTraces({ question: opts.question ?? DEFAULT_QUESTION }, {
+      ...opts.analyze,
+      source
+    });
+    if (opts.save) await opts.save(result, ctx);
+    if (opts.gateOn && !opts.gateOn(result, ctx)) {
+      await ctx.store.appendEvent({
+        eventId: `analyst-gate-${ctx.runId}`,
+        runId: ctx.runId,
+        kind: "log",
+        timestamp: Date.now(),
+        payload: {
+          source: "trace_analyst_hook",
+          reason: "analyst_gate_failed",
+          findings: result.findings
+        }
+      });
+    }
+  };
+}
+// src/trace-analyst/insights.ts
+var DOMAIN_STOP_WORDS = /* @__PURE__ */ new Set([
+  "and",
+  "advanced",
+  "app",
+  "build",
+  "create",
+  "easy",
+  "expert",
+  "extreme",
+  "for",
+  "from",
+  "hard",
+  "implementation",
+  "integrate",
+  "medium",
+  "project",
+  "task",
+  "the",
+  "this",
+  "with",
+  "workflow"
+]);
+function tokenizeDomainWords(value) {
+  return [...value.matchAll(/[A-Za-z][A-Za-z0-9.+#-]{2,}/g)].map((match) => match[0].toLowerCase()).filter((word) => !DOMAIN_STOP_WORDS.has(word));
+}
+function inferDomainKeywords(suite) {
+  const suiteWords = new Set(tokenizeDomainWords(`${suite.name} ${suite.collectionId ?? ""}`));
+  const source = [
+    suite.name,
+    suite.collectionId ?? "",
+    ...suite.tasks.flatMap((task) => [
+      task.id,
+      task.name,
+      task.prompt ?? "",
+      task.difficulty ?? "",
+      ...task.tags ?? [],
+      ...task.gaps ?? []
+    ])
+  ].join(" ");
+  const counts = /* @__PURE__ */ new Map();
+  for (const word of tokenizeDomainWords(source)) counts.set(word, (counts.get(word) ?? 0) + 1);
+  return [...counts.entries()].filter(([word, count]) => count >= 2 || suiteWords.has(word)).sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).map(([word]) => word).slice(0, 18);
+}
+function domainEvidencePattern(keywords) {
+  const escaped = keywords.filter((keyword) => keyword.length >= 3).map((keyword) => keyword.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"));
+  return escaped.length > 0 ? new RegExp(`(?<![A-Za-z0-9])(?:${escaped.join("|")})(?![A-Za-z0-9])`, "i") : /(?<![A-Za-z0-9])(?:sdk|api|css|dns|xml|provider|client|service|integration|webhook|transaction|auth|oauth|graphql|rest)(?![A-Za-z0-9])/i;
+}
+function describeTraceInsightScope(suite) {
+  const taskLabel = suite.tasks.length === 1 ? "1 implementation task" : `${suite.tasks.length} implementation tasks`;
+  const tags = /* @__PURE__ */ new Map();
+  for (const task of suite.tasks) {
+    for (const tag of task.tags ?? []) tags.set(tag, (tags.get(tag) ?? 0) + 1);
+  }
+  const topTags = [...tags.entries()].sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).slice(0, 8).map(([tag]) => tag);
+  if (topTags.length > 0) return `${taskLabel} across ${topTags.join(", ")}.`;
+  const difficulties = [
+    ...new Set(
+      suite.tasks.map((task) => task.difficulty).filter((value) => Boolean(value))
+    )
+  ].join(", ");
+  return `${taskLabel} across ${difficulties || "the selected benchmark scope"}.`;
+}
+function planTraceInsightQuestions(input) {
+  const hasFailures = input.suite.tasks.some((task) => task.outcome && task.outcome !== "satisfied");
+  const hasMultipleShots = input.suite.tasks.some(
+    (task) => (task.gaps ?? []).some((gap) => /shot|review|retry|continue/i.test(gap))
+  );
+  const questions = [
+    {
+      id: "execution-path",
+      question: "What did the worker actually do before the first meaningful implementation edit?",
+      why: "Separates grounded execution from polished but shallow output."
+    },
+    {
+      id: "research-grounding",
+      question: "Did the worker inspect docs, source, examples, or package references before committing to an implementation path?",
+      why: "Identifies whether failures came from weak retrieval, weak examples, or premature coding."
+    },
+    {
+      id: "domain-proof",
+      question: "Which tasks produced executable domain proof versus UI copy, placeholders, or inferred behavior?",
+      why: "Keeps product-quality claims tied to concrete evidence."
+    },
+    {
+      id: "root-cause",
+      question: "For each major failure cluster, is the likely root cause prompt/scaffold, docs/examples, SDK/API ergonomics, evaluator, runtime, or model behavior?",
+      why: "Turns trace observations into actionable ownership."
+    },
+    {
+      id: "evidence-quality",
+      question: "Which external-facing claims are directly supported by trace ids, span ids, verifier findings, reviewer notes, or generated code?",
+      why: "Prevents unsupported customer-report conclusions."
+    }
+  ];
+  if (hasMultipleShots) {
+    questions.push({
+      id: "reviewer-lift",
+      question: "Where did reviewer feedback improve score, stall, or regress across shots?",
+      why: "Shows whether the driver loop is learning or merely repeating work."
+    });
+  }
+  if (hasFailures) {
+    questions.push({
+      id: "optimization-targets",
+      question: "Which prompt, evaluator, scaffold, or workflow changes should feed the next GEPA/autoresearch optimization run?",
+      why: "Connects benchmark evidence to the optimization loop."
+    });
+  }
+  return questions;
+}
+function buildTraceInsightContext(input) {
+  return {
+    suite: input.suite,
+    scope: describeTraceInsightScope(input.suite),
+    keywords: inferDomainKeywords(input.suite),
+    questions: planTraceInsightQuestions(input),
+    panel: defaultTraceInsightPanel(),
+    findings: input.findings ?? [],
+    agent: input.agent ?? null,
+    totals: input.totals ?? null
+  };
+}
+function scoreTraceInsightReadiness(context) {
+  const failedTasks = context.suite.tasks.filter(
+    (task) => task.outcome && task.outcome !== "satisfied"
+  );
+  const findingTaskIds = new Set(context.findings.flatMap((finding) => finding.taskIds));
+  const failedTasksWithFindings = failedTasks.filter((task) => findingTaskIds.has(task.id));
+  const tasksWithGaps = context.suite.tasks.filter((task) => (task.gaps ?? []).length > 0);
+  const gates = [
+    {
+      id: "domain-context",
+      label: "Domain context inferred",
+      passed: context.keywords.length > 0,
+      severity: "high",
+      detail: context.keywords.length > 0 ? `${context.keywords.length} domain terms inferred: ${context.keywords.slice(0, 8).join(", ")}` : "No domain terms were inferred from suite, tasks, prompts, tags, or gaps."
+    },
+    {
+      id: "panel-coverage",
+      label: "Analyst panel planned",
+      passed: context.panel.length >= 4 && context.questions.length >= 5,
+      severity: "high",
+      detail: `${context.panel.length} panel roles and ${context.questions.length} investigation questions planned.`
+    },
+    {
+      id: "failure-coverage",
+      label: "Failures mapped to findings",
+      passed: failedTasks.length === 0 || failedTasksWithFindings.length / failedTasks.length >= 0.5,
+      severity: "critical",
+      detail: failedTasks.length === 0 ? "No failed tasks in suite." : `${failedTasksWithFindings.length}/${failedTasks.length} failed tasks appear in finding clusters.`
+    },
+    {
+      id: "gap-evidence",
+      label: "Task gaps captured",
+      passed: failedTasks.length === 0 || tasksWithGaps.length / failedTasks.length >= 0.5,
+      severity: "medium",
+      detail: `${tasksWithGaps.length} tasks include explicit evaluator or analyst gaps.`
+    }
+  ];
+  const penalty = gates.reduce((sum, gate) => {
+    if (gate.passed) return sum;
+    if (gate.severity === "critical") return sum + 35;
+    if (gate.severity === "high") return sum + 20;
+    if (gate.severity === "medium") return sum + 10;
+    return sum + 5;
+  }, 0);
+  const score = Math.max(0, Math.min(1, 1 - penalty / 100));
+  return {
+    score,
+    grade: score >= 0.9 ? "external-ready" : score >= 0.7 ? "internal-review" : "raw-analysis",
+    gates
+  };
+}
+function defaultTraceInsightPanel() {
+  return [
+    {
+      id: "trace-forensics",
+      name: "Trace Forensics",
+      responsibility: "Reconstruct what the worker did in order, including research, edits, reviewer interventions, verifier feedback, and stop reason."
+    },
+    {
+      id: "root-cause",
+      name: "Root Cause",
+      responsibility: "Map failures to prompt/scaffold, docs/examples, SDK/API/product ergonomics, evaluator, runtime, or model behavior."
+    },
+    {
+      id: "optimization",
+      name: "Optimization",
+      responsibility: "Identify prompt, reviewer, evaluator, scaffold, and GEPA/autoresearch changes that should be tested next."
+    },
+    {
+      id: "external-evidence",
+      name: "External Evidence",
+      responsibility: "Separate customer-safe claims from internal harness findings and reject conclusions without task, trace, span, code, reviewer, or verifier evidence."
+    }
+  ];
+}
+function buildTraceInsightPrompt(input) {
+  const context = buildTraceInsightContext(input);
+  const maxRepresentativeTraces = input.maxRepresentativeTraces ?? 6;
+  return `Analyze this benchmark run and produce evidence-backed trace intelligence.
+Audience:
+- internal AI/product leadership
+- possible customer-facing report for ${input.suite.name}
+Investigation plan:
+${context.questions.map((item, index) => `${index + 1}. ${item.question} (${item.why})`).join("\n")}
+Analyst panel:
+${context.panel.map((role) => `- ${role.name}: ${role.responsibility}`).join("\n")}
+If the task branches are independent, use subagents for the panel roles above and aggregate their findings. Do not run a panel role unless its answer will change the final report.
+Required output:
+1. Executive verdict: what this run proves and does not prove.
+2. The investigation questions you answered and the evidence used.
+3. Failure taxonomy: agent prompting, evaluator/harness, docs/examples, SDK/API/product integration, infra.
+4. Evidence-backed examples with trace ids/task ids and concrete verifier findings.
+5. Highest-ROI fixes for the benchmark harness, prompt/GEPA optimization, and customer-facing product/docs surface.
+6. What is safe for an external report versus what must stay internal.
+7. One rerun plan that would validate lift after optimization.
+Budget:
+- Inspect the dataset overview, the failure summary, and at most ${maxRepresentativeTraces} representative traces.
+- Prefer traces named in the failure summary over broad exploration.
+- Do not do exhaustive trace sweeps.
+- Return the final report as soon as the taxonomy and examples are supported.
+Run summary:
+${JSON.stringify(
+    {
+      suite: input.suite.name,
+      scope: context.scope,
+      inferredKeywords: context.keywords,
+      agent: context.agent,
+      totals: context.totals,
+      findings: context.findings.map((finding) => ({
+        kind: finding.kind,
+        severity: finding.severity,
+        taskCount: finding.taskIds.length,
+        proposedFixClass: finding.proposedFixClass
+      })),
+      failures: input.suite.tasks.filter((task) => task.outcome && task.outcome !== "satisfied").map((task) => ({
+        task: task.id,
+        difficulty: task.difficulty,
+        outcome: task.outcome,
+        score: task.score,
+        gaps: task.gaps ?? []
+      }))
+    },
+    null,
+    2
+  )}
+Use the trace tools. Do not invent facts. Cite task ids. Separate customer-facing claims from internal harness/model findings.`;
+}
 // src/trace/store.ts
 var InMemoryTraceStore = class {
   runs = /* @__PURE__ */ new Map();
@@ -1545,6 +1841,16 @@ export {
   buildTraceAnalystTools,
   traceAnalystFunctionGroup,
   analyzeTraces,
+  traceAnalystOnRunComplete,
+  tokenizeDomainWords,
+  inferDomainKeywords,
+  domainEvidencePattern,
+  describeTraceInsightScope,
+  planTraceInsightQuestions,
+  buildTraceInsightContext,
+  scoreTraceInsightReadiness,
+  defaultTraceInsightPanel,
+  buildTraceInsightPrompt,
   InMemoryTraceStore,
   FileSystemTraceStore,
   OTEL_AGENT_EVAL_SCOPE,
@@ -1558,4 +1864,4 @@ export {
   createReplayFetch,
   iterateRawCalls
 };
-//# sourceMappingURL=chunk-UW4NOOZI.js.map
+//# sourceMappingURL=chunk-HIO4UIS5.js.map