@agentproto/eval 0.2.11 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.mjs CHANGED
@@ -415,6 +415,550 @@ function toVitest(suite, opts, hooks) {
415
415
  }
416
416
  });
417
417
  }
418
+ var FIRST_PERSON_TERMS = [
419
+ "je",
420
+ "j'",
421
+ "j\u2019",
422
+ "moi",
423
+ "mon",
424
+ "ma",
425
+ "mes",
426
+ "me",
427
+ "m'",
428
+ "m\u2019",
429
+ "nous",
430
+ "notre",
431
+ "nos",
432
+ "mien(?:ne)?s?"
433
+ ];
434
+ var FIRST_PERSON_RE = new RegExp(
435
+ FIRST_PERSON_TERMS.map((term) => {
436
+ const isElision = term.endsWith("'") || term.endsWith("\u2019");
437
+ return `(?<![\\p{L}\\p{N}_])${term}${isElision ? "" : "(?![\\p{L}\\p{N}_])"}`;
438
+ }).join("|"),
439
+ "iu"
440
+ );
441
+ var ABBREVIATIONS = /* @__PURE__ */ new Set([
442
+ "m",
443
+ "mme",
444
+ "mlle",
445
+ "dr",
446
+ "st",
447
+ "ste",
448
+ "etc",
449
+ "cf",
450
+ "vs",
451
+ "p",
452
+ "ex",
453
+ "no",
454
+ "art",
455
+ "vol",
456
+ "pp",
457
+ "chap"
458
+ ]);
459
+ function splitLines(text) {
460
+ return text.split("\n").map((line) => line.trim()).filter((line) => line.length > 0);
461
+ }
462
+ function endsWithAbbreviation(chunk) {
463
+ const match = chunk.match(/(\p{L}+)\.$/u);
464
+ if (!match) return false;
465
+ const word = match[1];
466
+ if (ABBREVIATIONS.has(word.toLowerCase())) return true;
467
+ return word.length === 1 && /\p{Lu}/u.test(word);
468
+ }
469
+ function splitSentences(text) {
470
+ const chunks = text.split(/(?<=[.!?])\s+|\n+/u).map((s) => s.trim()).filter((s) => s.length > 0);
471
+ const sentences = [];
472
+ for (const chunk of chunks) {
473
+ const prevIndex = sentences.length - 1;
474
+ const prev = prevIndex >= 0 ? sentences[prevIndex] : void 0;
475
+ if (prev !== void 0 && endsWithAbbreviation(prev)) {
476
+ sentences[prevIndex] = `${prev} ${chunk}`;
477
+ } else {
478
+ sentences.push(chunk);
479
+ }
480
+ }
481
+ return sentences;
482
+ }
483
+ function bulletsRatio(text) {
484
+ const lines = splitLines(text);
485
+ if (lines.length === 0) return 0;
486
+ const bulletLines = lines.filter((line) => /^([-*•]|\d+\.)\s+/u.test(line));
487
+ return bulletLines.length / lines.length;
488
+ }
489
+ function firstPersonRatio(text) {
490
+ const sentences = splitSentences(text);
491
+ if (sentences.length === 0) return 0;
492
+ const firstPersonSentences = sentences.filter((s) => FIRST_PERSON_RE.test(s));
493
+ return firstPersonSentences.length / sentences.length;
494
+ }
495
+ function questionRate(text) {
496
+ const sentences = splitSentences(text);
497
+ if (sentences.length === 0) return 0;
498
+ const questions = sentences.filter((s) => s.endsWith("?"));
499
+ return questions.length / sentences.length;
500
+ }
501
+ function meanSentenceLength(text) {
502
+ const sentences = splitSentences(text);
503
+ if (sentences.length === 0) return 0;
504
+ const total = sentences.reduce((sum, s) => {
505
+ const words = s.split(/\s+/u).filter((w) => w.length > 0);
506
+ return sum + words.length;
507
+ }, 0);
508
+ return total / sentences.length;
509
+ }
510
+ var DEFAULT_LENGTH_BAND = { min: 5, max: 30 };
511
+ function computeTextStats(text, band = DEFAULT_LENGTH_BAND) {
512
+ const length = meanSentenceLength(text);
513
+ return {
514
+ bulletsRatio: bulletsRatio(text),
515
+ firstPersonRatio: firstPersonRatio(text),
516
+ questionRate: questionRate(text),
517
+ meanSentenceLength: length,
518
+ inBand: length >= band.min && length <= band.max
519
+ };
520
+ }
521
+ var textStatsThresholdsSchema = z.object({
522
+ maxBulletsRatio: z.number().min(0).max(1).optional().describe("Default 0.02."),
523
+ minFirstPersonRatio: z.number().min(0).max(1).optional().describe("Default 0.6."),
524
+ lengthBand: z.object({ min: z.number().min(0), max: z.number().min(0) }).refine((band) => band.min <= band.max, { message: "lengthBand.min must be <= lengthBand.max" }).optional().describe("Word-count band for mean sentence length. Default { min: 5, max: 30 }.")
525
+ });
526
+ var textStatsTool = defineTool({
527
+ id: "eval.text-stats",
528
+ description: "Deterministic scorer: computes bullet-line ratio, first-person ratio, question rate, and mean sentence length (+ inBand) over French text. `passed` is derived from caller-supplied `thresholds` \u2014 this tool owns no fixed gate.",
529
+ version: "0.1.0",
530
+ inputSchema: z.object({
531
+ text: z.string().describe("The French text to analyze."),
532
+ thresholds: textStatsThresholdsSchema.optional()
533
+ }),
534
+ outputSchema: scoreSchema,
535
+ mutates: [],
536
+ approval: "auto",
537
+ riskLevel: 0
538
+ });
539
+ var textStatsImpl = implementTool(textStatsTool, ({ input }) => {
540
+ const thresholds = input.thresholds ?? {};
541
+ const maxBulletsRatio = thresholds.maxBulletsRatio ?? 0.02;
542
+ const minFirstPersonRatio = thresholds.minFirstPersonRatio ?? 0.6;
543
+ const band = thresholds.lengthBand ?? DEFAULT_LENGTH_BAND;
544
+ const stats = computeTextStats(input.text, band);
545
+ const bulletsOk = stats.bulletsRatio <= maxBulletsRatio;
546
+ const firstPersonOk = stats.firstPersonRatio >= minFirstPersonRatio;
547
+ const checks = [bulletsOk, firstPersonOk, stats.inBand];
548
+ const value = checks.filter(Boolean).length / checks.length;
549
+ const passed = checks.every(Boolean);
550
+ return {
551
+ value,
552
+ passed,
553
+ label: "text-stats",
554
+ rationale: `bulletsRatio=${stats.bulletsRatio.toFixed(3)} (max ${maxBulletsRatio}), firstPersonRatio=${stats.firstPersonRatio.toFixed(3)} (min ${minFirstPersonRatio}), questionRate=${stats.questionRate.toFixed(3)}, meanSentenceLength=${stats.meanSentenceLength.toFixed(1)}, inBand=${stats.inBand} (band ${band.min}-${band.max})`
555
+ };
556
+ });
557
+ var WORD_RE = /\p{L}[\p{L}'’-]*/gu;
558
+ var FRENCH_STOPWORDS = /* @__PURE__ */ new Set([
559
+ "le",
560
+ "la",
561
+ "les",
562
+ "un",
563
+ "une",
564
+ "des",
565
+ "de",
566
+ "du",
567
+ "et",
568
+ "\xE0",
569
+ "au",
570
+ "aux",
571
+ "ce",
572
+ "ces",
573
+ "cet",
574
+ "cette",
575
+ "que",
576
+ "qui",
577
+ "quoi",
578
+ "dont",
579
+ "o\xF9",
580
+ "je",
581
+ "tu",
582
+ "il",
583
+ "elle",
584
+ "on",
585
+ "nous",
586
+ "vous",
587
+ "ils",
588
+ "elles",
589
+ "est",
590
+ "sont",
591
+ "\xE9t\xE9",
592
+ "\xEAtre",
593
+ "avoir",
594
+ "ai",
595
+ "as",
596
+ "a",
597
+ "avons",
598
+ "avez",
599
+ "ont",
600
+ "pour",
601
+ "par",
602
+ "sur",
603
+ "sous",
604
+ "dans",
605
+ "avec",
606
+ "sans",
607
+ "ne",
608
+ "pas",
609
+ "plus",
610
+ "mais",
611
+ "ou",
612
+ "si",
613
+ "se",
614
+ "sa",
615
+ "son",
616
+ "ses",
617
+ "leur",
618
+ "leurs",
619
+ "en",
620
+ "y"
621
+ ]);
622
+ function tokenize(text) {
623
+ return (text.normalize("NFC").toLowerCase().match(WORD_RE) ?? []).map((w) => w.trim()).filter((w) => w.length > 0);
624
+ }
625
+ function containsWord(text, term) {
626
+ const escaped = term.normalize("NFC").replace(/[.*+?^${}()|[\]\\]/gu, "\\$&");
627
+ const re = new RegExp(`(?<![\\p{L}\\p{N}_])${escaped}(?![\\p{L}\\p{N}_])`, "iu");
628
+ return re.test(text.normalize("NFC"));
629
+ }
630
+ function countTerms(texts, minLen) {
631
+ const counts = /* @__PURE__ */ new Map();
632
+ for (const text of texts) {
633
+ for (const word of tokenize(text)) {
634
+ if (word.length < minLen) continue;
635
+ if (FRENCH_STOPWORDS.has(word)) continue;
636
+ counts.set(word, (counts.get(word) ?? 0) + 1);
637
+ }
638
+ }
639
+ return counts;
640
+ }
641
+ function totalCount(counts) {
642
+ let total = 0;
643
+ for (const c of counts.values()) total += c;
644
+ return total;
645
+ }
646
+ function extractLexicon(corpusTexts, opts) {
647
+ const top = opts?.top ?? 20;
648
+ const minLen = opts?.minLen ?? 3;
649
+ const counts = countTerms(corpusTexts, minLen);
650
+ if (opts?.background && opts.background.length > 0) {
651
+ const bgCounts = countTerms(opts.background, minLen);
652
+ const fgTotal = totalCount(counts);
653
+ const bgTotal = totalCount(bgCounts);
654
+ return [...counts.entries()].map(([word, fg]) => {
655
+ const bg = bgCounts.get(word) ?? 0;
656
+ const score = Math.log((fg + 1) / (fgTotal - fg + 1)) - Math.log((bg + 1) / (bgTotal - bg + 1));
657
+ return [word, score];
658
+ }).sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).slice(0, top).map(([word]) => word);
659
+ }
660
+ return [...counts.entries()].sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).slice(0, top).map(([word]) => word);
661
+ }
662
+ var lexiconHitRateTool = defineTool({
663
+ id: "eval.lexicon-hit-rate",
664
+ description: "Deterministic scorer: fraction of `lexicon` terms present as whole words in `text` (case-insensitive). `passed = hitRate >= threshold` (threshold default 0.5).",
665
+ version: "0.1.0",
666
+ inputSchema: z.object({
667
+ text: z.string().describe("The text to score."),
668
+ lexicon: z.array(z.string()).min(1).describe("Signature terms to look for, e.g. from extractLexicon."),
669
+ threshold: z.number().min(0).max(1).optional().describe("Minimum hit rate to pass. Default 0.5.")
670
+ }),
671
+ outputSchema: scoreSchema,
672
+ mutates: [],
673
+ approval: "auto",
674
+ riskLevel: 0
675
+ });
676
+ var lexiconHitRateImpl = implementTool(lexiconHitRateTool, ({ input }) => {
677
+ const threshold = input.threshold ?? 0.5;
678
+ const lexicon = input.lexicon;
679
+ const hits = lexicon.filter((term) => containsWord(input.text, term));
680
+ const hitRate = lexicon.length === 0 ? 0 : hits.length / lexicon.length;
681
+ const passed = hitRate >= threshold;
682
+ return {
683
+ value: hitRate,
684
+ passed,
685
+ label: "lexicon-hit-rate",
686
+ rationale: `${hits.length}/${lexicon.length} lexicon terms found (threshold ${threshold}): ${hits.join(", ") || "none"}`
687
+ };
688
+ });
689
+
690
+ // src/style/verdict.ts
691
+ function parseVerdict(raw) {
692
+ const parsed = judgeVerdictSchema.safeParse(raw);
693
+ return parsed.success ? parsed.data : null;
694
+ }
695
+
696
+ // src/style/pairwise.ts
697
+ var stylePairwiseTool = defineTool({
698
+ id: "eval.style-pairwise",
699
+ description: "Model-backed scorer: asks an injected judge which of `a`/`b` is closer in style to `reference` under `criteria`. value encodes preference (1 = a wins, 0 = b wins, 0.5 = tie). The judge lives in the driver (see makeStylePairwiseDriver) \u2014 this tool contract carries no model call.",
700
+ version: "0.1.0",
701
+ inputSchema: z.object({
702
+ reference: z.string().describe("The reference style text."),
703
+ a: z.string().describe("Candidate A."),
704
+ b: z.string().describe("Candidate B."),
705
+ criteria: z.string().describe("Free-form grading criteria/rubric for the judge.")
706
+ }),
707
+ outputSchema: scoreSchema,
708
+ mutates: [],
709
+ approval: "auto",
710
+ riskLevel: 0
711
+ });
712
+ function clamp012(x) {
713
+ return Math.min(1, Math.max(0, x));
714
+ }
715
+ function makeStylePairwiseDriver(judge) {
716
+ return defineDriver({
717
+ id: "eval-style-pairwise",
718
+ name: "Eval Style Pairwise (model-backed)",
719
+ description: "Model-backed scorer driver: implements eval.style-pairwise by awaiting an injected JudgeFn over {reference, a, b} and mapping its verdict.value to an a-vs-b preference score.",
720
+ version: "0.1.0",
721
+ kind: "builtin",
722
+ implements: [{ tool: "eval.style-pairwise", version: "0.1.0" }],
723
+ implementations: [
724
+ implementTool(stylePairwiseTool, async ({ input }) => {
725
+ const raw = await judge({
726
+ output: { reference: input.reference, a: input.a, b: input.b },
727
+ criteria: input.criteria,
728
+ expected: input.reference
729
+ });
730
+ const verdict = parseVerdict(raw);
731
+ if (!verdict) {
732
+ return {
733
+ value: 0,
734
+ passed: false,
735
+ label: "style-pairwise",
736
+ rationale: "judge returned a malformed verdict"
737
+ };
738
+ }
739
+ const value = clamp012(verdict.value);
740
+ const passed = verdict.passed ?? value >= 0.5;
741
+ return {
742
+ value,
743
+ passed,
744
+ label: "style-pairwise",
745
+ ...verdict.rationale ? { rationale: verdict.rationale } : {}
746
+ };
747
+ })
748
+ ]
749
+ });
750
+ }
751
+ function unswap(winner, order) {
752
+ if (order === "normal" || winner === "tie") return winner;
753
+ return winner === "a" ? "b" : "a";
754
+ }
755
+ function aWinScore(winner) {
756
+ if (winner === "a") return 1;
757
+ if (winner === "tie") return 0.5;
758
+ return 0;
759
+ }
760
+ function cohenKappa(r1, r2) {
761
+ const n = r1.length;
762
+ let agree = 0;
763
+ const categories = ["a", "b", "tie"];
764
+ const counts1 = new Map(categories.map((c) => [c, 0]));
765
+ const counts2 = new Map(categories.map((c) => [c, 0]));
766
+ for (let i = 0; i < n; i++) {
767
+ const x1 = r1[i];
768
+ const x2 = r2[i];
769
+ if (x1 === x2) agree++;
770
+ counts1.set(x1, (counts1.get(x1) ?? 0) + 1);
771
+ counts2.set(x2, (counts2.get(x2) ?? 0) + 1);
772
+ }
773
+ const po = agree / n;
774
+ const pe = categories.reduce((sum, c) => sum + (counts1.get(c) ?? 0) / n * ((counts2.get(c) ?? 0) / n), 0);
775
+ if (pe >= 1) return po === 1 ? 1 : 0;
776
+ return (po - pe) / (1 - pe);
777
+ }
778
+ function pairwiseWinRate(verdicts) {
779
+ const n = verdicts.length;
780
+ const normalized = verdicts.map((v) => ({ ...v, winner: unswap(v.winner, v.order) }));
781
+ const normalGroup = normalized.filter((v) => v.order === "normal");
782
+ const swappedGroup = normalized.filter((v) => v.order === "swapped");
783
+ const nNormal = normalGroup.length;
784
+ const nSwapped = swappedGroup.length;
785
+ const rateNormal = nNormal === 0 ? null : normalGroup.reduce((sum, v) => sum + aWinScore(v.winner), 0) / nNormal;
786
+ const rateSwapped = nSwapped === 0 ? null : swappedGroup.reduce((sum, v) => sum + aWinScore(v.winner), 0) / nSwapped;
787
+ const balanced = nNormal > 0 && nSwapped > 0;
788
+ const winRate = rateNormal !== null && rateSwapped !== null ? (rateNormal + rateSwapped) / 2 : rateNormal ?? rateSwapped ?? 0;
789
+ const byJudge = /* @__PURE__ */ new Map();
790
+ for (const v of normalized) {
791
+ let items = byJudge.get(v.judge);
792
+ if (!items) {
793
+ items = /* @__PURE__ */ new Map();
794
+ byJudge.set(v.judge, items);
795
+ }
796
+ if (!items.has(v.item)) items.set(v.item, v.winner);
797
+ }
798
+ const judges = [...byJudge.entries()].sort((a, b) => b[1].size - a[1].size);
799
+ let kappa = null;
800
+ if (judges.length >= 2) {
801
+ const [, items1] = judges[0];
802
+ const [, items2] = judges[1];
803
+ const commonItems = [...items1.keys()].filter((item) => items2.has(item)).sort();
804
+ if (commonItems.length >= 2) {
805
+ const r1 = commonItems.map((item) => items1.get(item));
806
+ const r2 = commonItems.map((item) => items2.get(item));
807
+ kappa = cohenKappa(r1, r2);
808
+ }
809
+ }
810
+ return { winRate, kappa, n, nNormal, nSwapped, balanced };
811
+ }
812
+ function dot(a, b) {
813
+ let sum = 0;
814
+ for (let i = 0; i < a.length; i++) sum += (a[i] ?? 0) * (b[i] ?? 0);
815
+ return sum;
816
+ }
817
+ function norm(v) {
818
+ return Math.sqrt(dot(v, v));
819
+ }
820
+ var EmbeddingDimensionError = class extends RangeError {
821
+ };
822
+ function centroid(vectors) {
823
+ const dims = vectors[0]?.length ?? 0;
824
+ for (const v of vectors) {
825
+ if (v.length !== dims) {
826
+ throw new EmbeddingDimensionError(`inconsistent embedding dimensions: expected ${dims}, got ${v.length}`);
827
+ }
828
+ }
829
+ const sum = new Array(dims).fill(0);
830
+ for (const v of vectors) {
831
+ for (let i = 0; i < dims; i++) sum[i] = (sum[i] ?? 0) + (v[i] ?? 0);
832
+ }
833
+ return sum.map((x) => x / vectors.length);
834
+ }
835
+ function cosineToCentroid(candidate, references) {
836
+ if (references.length === 0) return 0;
837
+ const c = centroid(references);
838
+ if (candidate.length !== c.length) {
839
+ throw new EmbeddingDimensionError(
840
+ `inconsistent embedding dimensions: candidate has ${candidate.length}, references have ${c.length}`
841
+ );
842
+ }
843
+ const cn = norm(c);
844
+ const vn = norm(candidate);
845
+ if (cn === 0 || vn === 0) return 0;
846
+ const cosine = dot(candidate, c) / (vn * cn);
847
+ return Math.max(0, Math.min(1, cosine));
848
+ }
849
+ var styleEmbeddingTool = defineTool({
850
+ id: "eval.style-embedding",
851
+ description: "Model-backed scorer: embeds `candidate` and `references[]` via an injected EmbedFn and scores value = cosine similarity of `candidate` to the references' centroid, clamped to [0, 1] via max(0, cosine) (orthogonal or opposing candidates score 0, not 0.5). The embedding call lives in the driver (see makeStyleEmbeddingDriver) \u2014 this tool contract carries no model call of its own.",
852
+ version: "0.1.0",
853
+ inputSchema: z.object({
854
+ candidate: z.string().describe("The text to score."),
855
+ references: z.array(z.string()).min(1).describe("Reference texts defining the style centroid.")
856
+ }),
857
+ outputSchema: scoreSchema,
858
+ mutates: [],
859
+ approval: "auto",
860
+ riskLevel: 0
861
+ });
862
+ function makeStyleEmbeddingDriver(embed, opts) {
863
+ const threshold = opts?.threshold ?? 0.5;
864
+ return defineDriver({
865
+ id: "eval-style-embedding",
866
+ name: "Eval Style Embedding (model-backed)",
867
+ description: "Model-backed scorer driver: implements eval.style-embedding by awaiting an injected EmbedFn and scoring cosine similarity of the candidate to the references' centroid. No LLM SDK or network call here.",
868
+ version: "0.1.0",
869
+ kind: "builtin",
870
+ implements: [{ tool: "eval.style-embedding", version: "0.1.0" }],
871
+ implementations: [
872
+ implementTool(styleEmbeddingTool, async ({ input }) => {
873
+ const vectors = await embed([input.candidate, ...input.references]);
874
+ const [candidateVec, ...referenceVecs] = vectors;
875
+ let value;
876
+ try {
877
+ value = cosineToCentroid(candidateVec ?? [], referenceVecs);
878
+ } catch (err) {
879
+ return {
880
+ value: 0,
881
+ passed: false,
882
+ label: "style-embedding",
883
+ rationale: err instanceof EmbeddingDimensionError ? err.message : "embedding failed"
884
+ };
885
+ }
886
+ const passed = value >= threshold;
887
+ return {
888
+ value,
889
+ passed,
890
+ label: "style-embedding",
891
+ rationale: `cosine-to-centroid over ${referenceVecs.length} reference(s), threshold ${threshold}`
892
+ };
893
+ })
894
+ ]
895
+ });
896
+ }
897
+ var outlineFidelityTool = defineTool({
898
+ id: "eval.outline-fidelity",
899
+ description: "Model-backed scorer: asks an injected judge whether `answer` covers every point in `outline` with zero added facts. value = covered/total as judged; passed = value >= 0.95, a fixed gate (not the judge's own passed). The judge lives in the driver (see makeOutlineFidelityDriver).",
900
+ version: "0.1.0",
901
+ inputSchema: z.object({
902
+ outline: z.string().describe("The source outline \u2014 bullet points the answer must cover."),
903
+ answer: z.string().describe("The produced prose answer to check for fidelity.")
904
+ }),
905
+ outputSchema: scoreSchema,
906
+ mutates: [],
907
+ approval: "auto",
908
+ riskLevel: 0
909
+ });
910
+ var FIDELITY_THRESHOLD = 0.95;
911
+ var FIDELITY_CRITERIA = "Score how faithfully the answer covers the outline: value = (outline points covered) / (total outline points), where any fact in the answer not present in the outline counts as a violation and caps value at the fraction covered. 1.0 means every outline point is covered and nothing was added.";
912
+ function makeOutlineFidelityDriver(judge) {
913
+ return defineDriver({
914
+ id: "eval-outline-fidelity",
915
+ name: "Eval Outline Fidelity (model-backed)",
916
+ description: "Model-backed scorer driver: implements eval.outline-fidelity by awaiting an injected JudgeFn over {answer, outline} and gating on a fixed 0.95 threshold, not the judge's own passed.",
917
+ version: "0.1.0",
918
+ kind: "builtin",
919
+ implements: [{ tool: "eval.outline-fidelity", version: "0.1.0" }],
920
+ implementations: [
921
+ implementTool(outlineFidelityTool, async ({ input }) => {
922
+ const raw = await judge({
923
+ output: input.answer,
924
+ criteria: FIDELITY_CRITERIA,
925
+ expected: input.outline
926
+ });
927
+ const verdict = parseVerdict(raw);
928
+ if (!verdict) {
929
+ return {
930
+ value: 0,
931
+ passed: false,
932
+ label: "outline-fidelity",
933
+ rationale: "judge returned a malformed verdict"
934
+ };
935
+ }
936
+ const value = Math.min(1, Math.max(0, verdict.value));
937
+ const passed = value >= FIDELITY_THRESHOLD;
938
+ return {
939
+ value,
940
+ passed,
941
+ label: "outline-fidelity",
942
+ ...verdict.rationale ? { rationale: verdict.rationale } : {}
943
+ };
944
+ })
945
+ ]
946
+ });
947
+ }
948
+
949
+ // src/style/index.ts
950
+ var styleScorersProvider = defineDriver({
951
+ id: "eval-style-scorers",
952
+ name: "Eval Style Scorers (built-in)",
953
+ description: "In-process deterministic style scorers: text-stats and lexicon-hit-rate. Each is an AIP-14 TOOL whose output is the shared Score shape.",
954
+ version: "0.1.0",
955
+ kind: "builtin",
956
+ implements: [
957
+ { tool: "eval.text-stats", version: "0.1.0" },
958
+ { tool: "eval.lexicon-hit-rate", version: "0.1.0" }
959
+ ],
960
+ implementations: [textStatsImpl, lexiconHitRateImpl]
961
+ });
418
962
 
419
963
  // src/index.ts
420
964
  var SPEC_NAME = "agenteval/v1";
@@ -439,6 +983,6 @@ var evalScorersProvider = defineDriver({
439
983
  ]
440
984
  });
441
985
 
442
- export { EVAL_EVENT_SCHEMA, SPEC_NAME, SPEC_VERSION, bindScorer, evalScorersProvider, exactMatchImpl, exactMatchTool, jsonSchemaValidImpl, jsonSchemaValidTool, jsonValueSchema, judgeVerdictSchema, latencyBudgetImpl, latencyBudgetTool, llmJudge, llmJudgeTool, makeLlmJudgeDriver, regexMatchImpl, regexMatchTool, runEval, scoreSchema, toVitest };
986
+ export { EVAL_EVENT_SCHEMA, EmbeddingDimensionError, SPEC_NAME, SPEC_VERSION, bindScorer, bulletsRatio, computeTextStats, cosineToCentroid, evalScorersProvider, exactMatchImpl, exactMatchTool, extractLexicon, firstPersonRatio, jsonSchemaValidImpl, jsonSchemaValidTool, jsonValueSchema, judgeVerdictSchema, latencyBudgetImpl, latencyBudgetTool, lexiconHitRateImpl, lexiconHitRateTool, llmJudge, llmJudgeTool, makeLlmJudgeDriver, makeOutlineFidelityDriver, makeStyleEmbeddingDriver, makeStylePairwiseDriver, meanSentenceLength, outlineFidelityTool, pairwiseWinRate, parseVerdict, questionRate, regexMatchImpl, regexMatchTool, runEval, scoreSchema, splitSentences, styleEmbeddingTool, stylePairwiseTool, styleScorersProvider, textStatsImpl, textStatsTool, toVitest };
443
987
  //# sourceMappingURL=index.mjs.map
444
988
  //# sourceMappingURL=index.mjs.map