PyPI - deepeval - Versions diffs - 3.7.3__py3-none-any.whl → 3.7.5__py3-none-any.whl - Mend

deepeval 3.7.3py3-none-any.whl → 3.7.5py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (156) hide show

deepeval/_version.py +1 -1
deepeval/cli/test.py +1 -1
deepeval/config/settings.py +102 -13
deepeval/dataset/golden.py +54 -2
deepeval/evaluate/configs.py +1 -1
deepeval/evaluate/evaluate.py +16 -8
deepeval/evaluate/execute.py +74 -27
deepeval/evaluate/utils.py +26 -22
deepeval/integrations/pydantic_ai/agent.py +19 -2
deepeval/integrations/pydantic_ai/instrumentator.py +62 -23
deepeval/metrics/__init__.py +14 -12
deepeval/metrics/answer_relevancy/answer_relevancy.py +74 -29
deepeval/metrics/answer_relevancy/template.py +188 -92
deepeval/metrics/argument_correctness/template.py +2 -2
deepeval/metrics/base_metric.py +2 -5
deepeval/metrics/bias/template.py +3 -3
deepeval/metrics/contextual_precision/contextual_precision.py +53 -15
deepeval/metrics/contextual_precision/template.py +115 -66
deepeval/metrics/contextual_recall/contextual_recall.py +50 -13
deepeval/metrics/contextual_recall/template.py +106 -55
deepeval/metrics/contextual_relevancy/contextual_relevancy.py +47 -15
deepeval/metrics/contextual_relevancy/template.py +87 -58
deepeval/metrics/conversation_completeness/template.py +2 -2
deepeval/metrics/conversational_dag/templates.py +4 -4
deepeval/metrics/conversational_g_eval/template.py +4 -3
deepeval/metrics/dag/templates.py +5 -5
deepeval/metrics/faithfulness/faithfulness.py +70 -27
deepeval/metrics/faithfulness/schema.py +1 -1
deepeval/metrics/faithfulness/template.py +200 -115
deepeval/metrics/g_eval/utils.py +2 -2
deepeval/metrics/hallucination/template.py +4 -4
deepeval/metrics/indicator.py +4 -4
deepeval/metrics/misuse/template.py +2 -2
deepeval/metrics/multimodal_metrics/__init__.py +0 -18
deepeval/metrics/multimodal_metrics/image_coherence/image_coherence.py +24 -17
deepeval/metrics/multimodal_metrics/image_editing/image_editing.py +26 -21
deepeval/metrics/multimodal_metrics/image_helpfulness/image_helpfulness.py +24 -17
deepeval/metrics/multimodal_metrics/image_reference/image_reference.py +24 -17
deepeval/metrics/multimodal_metrics/multimodal_g_eval/multimodal_g_eval.py +19 -19
deepeval/metrics/multimodal_metrics/multimodal_g_eval/template.py +63 -78
deepeval/metrics/multimodal_metrics/multimodal_g_eval/utils.py +20 -20
deepeval/metrics/multimodal_metrics/text_to_image/text_to_image.py +71 -50
deepeval/metrics/non_advice/template.py +2 -2
deepeval/metrics/pii_leakage/template.py +2 -2
deepeval/metrics/prompt_alignment/template.py +4 -4
deepeval/metrics/ragas.py +3 -3
deepeval/metrics/role_violation/template.py +2 -2
deepeval/metrics/step_efficiency/step_efficiency.py +1 -1
deepeval/metrics/tool_correctness/tool_correctness.py +2 -2
deepeval/metrics/toxicity/template.py +4 -4
deepeval/metrics/turn_contextual_precision/schema.py +21 -0
deepeval/metrics/turn_contextual_precision/template.py +187 -0
deepeval/metrics/turn_contextual_precision/turn_contextual_precision.py +550 -0
deepeval/metrics/turn_contextual_recall/schema.py +21 -0
deepeval/metrics/turn_contextual_recall/template.py +178 -0
deepeval/metrics/turn_contextual_recall/turn_contextual_recall.py +520 -0
deepeval/metrics/{multimodal_metrics/multimodal_contextual_relevancy → turn_contextual_relevancy}/schema.py +7 -1
deepeval/metrics/turn_contextual_relevancy/template.py +161 -0
deepeval/metrics/turn_contextual_relevancy/turn_contextual_relevancy.py +535 -0
deepeval/metrics/{multimodal_metrics/multimodal_faithfulness → turn_faithfulness}/schema.py +11 -3
deepeval/metrics/turn_faithfulness/template.py +218 -0
deepeval/metrics/turn_faithfulness/turn_faithfulness.py +596 -0
deepeval/metrics/turn_relevancy/template.py +2 -2
deepeval/metrics/utils.py +39 -58
deepeval/models/__init__.py +0 -12
deepeval/models/base_model.py +16 -38
deepeval/models/embedding_models/__init__.py +7 -0
deepeval/models/embedding_models/azure_embedding_model.py +69 -32
deepeval/models/embedding_models/local_embedding_model.py +39 -22
deepeval/models/embedding_models/ollama_embedding_model.py +42 -18
deepeval/models/embedding_models/openai_embedding_model.py +50 -15
deepeval/models/llms/amazon_bedrock_model.py +1 -2
deepeval/models/llms/anthropic_model.py +53 -20
deepeval/models/llms/azure_model.py +140 -43
deepeval/models/llms/deepseek_model.py +38 -23
deepeval/models/llms/gemini_model.py +222 -103
deepeval/models/llms/grok_model.py +39 -27
deepeval/models/llms/kimi_model.py +39 -23
deepeval/models/llms/litellm_model.py +103 -45
deepeval/models/llms/local_model.py +35 -22
deepeval/models/llms/ollama_model.py +129 -17
deepeval/models/llms/openai_model.py +151 -50
deepeval/models/llms/portkey_model.py +149 -0
deepeval/models/llms/utils.py +5 -3
deepeval/models/retry_policy.py +17 -14
deepeval/models/utils.py +94 -4
deepeval/optimizer/__init__.py +5 -0
deepeval/optimizer/algorithms/__init__.py +6 -0
deepeval/optimizer/algorithms/base.py +29 -0
deepeval/optimizer/algorithms/configs.py +18 -0
deepeval/optimizer/algorithms/copro/__init__.py +5 -0
deepeval/optimizer/algorithms/copro/copro.py +836 -0
deepeval/optimizer/algorithms/gepa/__init__.py +5 -0
deepeval/optimizer/algorithms/gepa/gepa.py +737 -0
deepeval/optimizer/algorithms/miprov2/__init__.py +17 -0
deepeval/optimizer/algorithms/miprov2/bootstrapper.py +435 -0
deepeval/optimizer/algorithms/miprov2/miprov2.py +752 -0
deepeval/optimizer/algorithms/miprov2/proposer.py +301 -0
deepeval/optimizer/algorithms/simba/__init__.py +5 -0
deepeval/optimizer/algorithms/simba/simba.py +999 -0
deepeval/optimizer/algorithms/simba/types.py +15 -0
deepeval/optimizer/configs.py +31 -0
deepeval/optimizer/policies.py +227 -0
deepeval/optimizer/prompt_optimizer.py +263 -0
deepeval/optimizer/rewriter/__init__.py +5 -0
deepeval/optimizer/rewriter/rewriter.py +124 -0
deepeval/optimizer/rewriter/utils.py +214 -0
deepeval/optimizer/scorer/__init__.py +5 -0
deepeval/optimizer/scorer/base.py +86 -0
deepeval/optimizer/scorer/scorer.py +316 -0
deepeval/optimizer/scorer/utils.py +30 -0
deepeval/optimizer/types.py +148 -0
deepeval/optimizer/utils.py +480 -0
deepeval/prompt/prompt.py +7 -6
deepeval/test_case/__init__.py +1 -3
deepeval/test_case/api.py +12 -10
deepeval/test_case/conversational_test_case.py +19 -1
deepeval/test_case/llm_test_case.py +152 -1
deepeval/test_case/utils.py +4 -8
deepeval/test_run/api.py +15 -14
deepeval/test_run/cache.py +2 -0
deepeval/test_run/test_run.py +9 -4
deepeval/tracing/patchers.py +9 -4
deepeval/tracing/tracing.py +2 -2
deepeval/utils.py +89 -0
{deepeval-3.7.3.dist-info → deepeval-3.7.5.dist-info}/METADATA +1 -4
{deepeval-3.7.3.dist-info → deepeval-3.7.5.dist-info}/RECORD +134 -118
deepeval/metrics/multimodal_metrics/multimodal_answer_relevancy/multimodal_answer_relevancy.py +0 -343
deepeval/metrics/multimodal_metrics/multimodal_answer_relevancy/schema.py +0 -19
deepeval/metrics/multimodal_metrics/multimodal_answer_relevancy/template.py +0 -122
deepeval/metrics/multimodal_metrics/multimodal_contextual_precision/multimodal_contextual_precision.py +0 -301
deepeval/metrics/multimodal_metrics/multimodal_contextual_precision/schema.py +0 -15
deepeval/metrics/multimodal_metrics/multimodal_contextual_precision/template.py +0 -132
deepeval/metrics/multimodal_metrics/multimodal_contextual_recall/multimodal_contextual_recall.py +0 -285
deepeval/metrics/multimodal_metrics/multimodal_contextual_recall/schema.py +0 -15
deepeval/metrics/multimodal_metrics/multimodal_contextual_recall/template.py +0 -112
deepeval/metrics/multimodal_metrics/multimodal_contextual_relevancy/multimodal_contextual_relevancy.py +0 -282
deepeval/metrics/multimodal_metrics/multimodal_contextual_relevancy/template.py +0 -102
deepeval/metrics/multimodal_metrics/multimodal_faithfulness/__init__.py +0 -0
deepeval/metrics/multimodal_metrics/multimodal_faithfulness/multimodal_faithfulness.py +0 -356
deepeval/metrics/multimodal_metrics/multimodal_faithfulness/template.py +0 -175
deepeval/metrics/multimodal_metrics/multimodal_tool_correctness/__init__.py +0 -0
deepeval/metrics/multimodal_metrics/multimodal_tool_correctness/multimodal_tool_correctness.py +0 -290
deepeval/models/mlllms/__init__.py +0 -4
deepeval/models/mlllms/azure_model.py +0 -334
deepeval/models/mlllms/gemini_model.py +0 -284
deepeval/models/mlllms/ollama_model.py +0 -144
deepeval/models/mlllms/openai_model.py +0 -258
deepeval/test_case/mllm_test_case.py +0 -170
/deepeval/metrics/{multimodal_metrics/multimodal_answer_relevancy → turn_contextual_precision}/__init__.py +0 -0
/deepeval/metrics/{multimodal_metrics/multimodal_contextual_precision → turn_contextual_recall}/__init__.py +0 -0
/deepeval/metrics/{multimodal_metrics/multimodal_contextual_recall → turn_contextual_relevancy}/__init__.py +0 -0
/deepeval/metrics/{multimodal_metrics/multimodal_contextual_relevancy → turn_faithfulness}/__init__.py +0 -0
{deepeval-3.7.3.dist-info → deepeval-3.7.5.dist-info}/LICENSE.md +0 -0
{deepeval-3.7.3.dist-info → deepeval-3.7.5.dist-info}/WHEEL +0 -0
{deepeval-3.7.3.dist-info → deepeval-3.7.5.dist-info}/entry_points.txt +0 -0

deepeval/metrics/multimodal_metrics/multimodal_g_eval/utils.py CHANGED Viewed

@@ -1,29 +1,26 @@
-from deepeval.test_case import MLLMTestCaseParams, MLLMTestCase, ToolCall
-from deepeval.test_case.mllm_test_case import MLLMImage
-from deepeval.models.mlllms.openai_model import (
+from deepeval.test_case import LLMTestCaseParams, LLMTestCase, ToolCall
+from deepeval.test_case import MLLMImage
+from deepeval.models.llms.openai_model import (
     unsupported_log_probs_multimodal_gpt_models,
 )
-from deepeval.models import (
-    DeepEvalBaseMLLM,
-    MultimodalOpenAIModel,
-)
+from deepeval.models import DeepEvalBaseLLM, GPTModel
 from typing import List, Union
 G_EVAL_PARAMS = {
-    MLLMTestCaseParams.INPUT: "Input",
-    MLLMTestCaseParams.ACTUAL_OUTPUT: "Actual Output",
-    MLLMTestCaseParams.EXPECTED_OUTPUT: "Expected Output",
-    MLLMTestCaseParams.CONTEXT: "Context",
-    MLLMTestCaseParams.RETRIEVAL_CONTEXT: "Retrieval Context",
-    MLLMTestCaseParams.EXPECTED_TOOLS: "Expected Tools",
-    MLLMTestCaseParams.TOOLS_CALLED: "Tools Called",
+    LLMTestCaseParams.INPUT: "Input",
+    LLMTestCaseParams.ACTUAL_OUTPUT: "Actual Output",
+    LLMTestCaseParams.EXPECTED_OUTPUT: "Expected Output",
+    LLMTestCaseParams.CONTEXT: "Context",
+    LLMTestCaseParams.RETRIEVAL_CONTEXT: "Retrieval Context",
+    LLMTestCaseParams.EXPECTED_TOOLS: "Expected Tools",
+    LLMTestCaseParams.TOOLS_CALLED: "Tools Called",
 }
 def construct_g_eval_params_string(
-    mllm_test_case_params: List[MLLMTestCaseParams],
+    mllm_test_case_params: List[LLMTestCaseParams],
 ):
     g_eval_params = [G_EVAL_PARAMS[param] for param in mllm_test_case_params]
     if len(g_eval_params) == 1:
@@ -39,12 +36,14 @@ def construct_g_eval_params_string(
 def construct_test_case_list(
-    evaluation_params: List[MLLMTestCaseParams], test_case: MLLMTestCase
+    evaluation_params: List[LLMTestCaseParams], test_case: LLMTestCase
 ) -> List[Union[str, MLLMImage]]:
+    from deepeval.utils import convert_to_multi_modal_array
     test_case_list = []
     for param in evaluation_params:
         test_case_param_list = [f"\n\n\n{G_EVAL_PARAMS[param]}:\n"]
-        value = getattr(test_case, param.value)
+        value = convert_to_multi_modal_array(getattr(test_case, param.value))
         for v in value:
             if isinstance(v, ToolCall):
                 test_case_param_list.append(repr(v))
@@ -54,15 +53,16 @@ def construct_test_case_list(
     return test_case_list
-def no_multimodal_log_prob_support(model: Union[str, DeepEvalBaseMLLM]):
+def no_multimodal_log_prob_support(model: Union[str, DeepEvalBaseLLM]):
     if (
         isinstance(model, str)
         and model in unsupported_log_probs_multimodal_gpt_models
     ):
         return True
     elif (
-        isinstance(model, MultimodalOpenAIModel)
-        and model.model_name in unsupported_log_probs_multimodal_gpt_models
+        isinstance(model, GPTModel)
+        and model.get_model_name()
+        in unsupported_log_probs_multimodal_gpt_models
     ):
         return True
     return False

deepeval/metrics/multimodal_metrics/text_to_image/text_to_image.py CHANGED Viewed

@@ -4,37 +4,40 @@ import math
 import textwrap
 from deepeval.metrics import BaseMultimodalMetric
-from deepeval.test_case import MLLMTestCaseParams, MLLMTestCase, MLLMImage
+from deepeval.test_case import LLMTestCaseParams, LLMTestCase, MLLMImage
 from deepeval.metrics.multimodal_metrics.text_to_image.template import (
     TextToImageTemplate,
 )
-from deepeval.utils import get_or_create_event_loop
+from deepeval.utils import (
+    get_or_create_event_loop,
+    convert_to_multi_modal_array,
+)
 from deepeval.metrics.utils import (
     construct_verbose_logs,
     trimAndLoadJson,
     check_mllm_test_case_params,
-    initialize_multimodal_model,
+    initialize_model,
 )
-from deepeval.models import DeepEvalBaseMLLM
+from deepeval.models import DeepEvalBaseLLM
 from deepeval.metrics.multimodal_metrics.text_to_image.schema import ReasonScore
 from deepeval.metrics.indicator import metric_progress_indicator
-required_params: List[MLLMTestCaseParams] = [
-    MLLMTestCaseParams.INPUT,
-    MLLMTestCaseParams.ACTUAL_OUTPUT,
+required_params: List[LLMTestCaseParams] = [
+    LLMTestCaseParams.INPUT,
+    LLMTestCaseParams.ACTUAL_OUTPUT,
 ]
 class TextToImageMetric(BaseMultimodalMetric):
     def __init__(
         self,
-        model: Optional[Union[str, DeepEvalBaseMLLM]] = None,
+        model: Optional[Union[str, DeepEvalBaseLLM]] = None,
         threshold: float = 0.5,
         async_mode: bool = True,
         strict_mode: bool = False,
         verbose_mode: bool = False,
     ):
-        self.model, self.using_native_model = initialize_multimodal_model(model)
+        self.model, self.using_native_model = initialize_model(model)
         self.evaluation_model = self.model.get_model_name()
         self.threshold = 1 if strict_mode else threshold
         self.strict_mode = strict_mode
@@ -43,11 +46,13 @@ class TextToImageMetric(BaseMultimodalMetric):
     def measure(
         self,
-        test_case: MLLMTestCase,
+        test_case: LLMTestCase,
         _show_indicator: bool = True,
         _in_component: bool = False,
     ) -> float:
-        check_mllm_test_case_params(test_case, required_params, 0, 1, self)
+        check_mllm_test_case_params(
+            test_case, required_params, 0, 1, self, self.model
+        )
         self.evaluation_cost = 0 if self.using_native_model else None
         with metric_progress_indicator(
@@ -63,10 +68,12 @@ class TextToImageMetric(BaseMultimodalMetric):
                     )
                 )
             else:
-                input_texts, _ = self.separate_images_from_text(test_case.input)
-                _, output_images = self.separate_images_from_text(
+                input = convert_to_multi_modal_array(test_case.input)
+                actual_output = convert_to_multi_modal_array(
                     test_case.actual_output
                 )
+                input_texts, _ = self.separate_images_from_text(input)
+                _, output_images = self.separate_images_from_text(actual_output)
                 self.SC_scores, self.SC_reasoning = (
                     self._evaluate_semantic_consistency(
@@ -99,11 +106,13 @@ class TextToImageMetric(BaseMultimodalMetric):
     async def a_measure(
         self,
-        test_case: MLLMTestCase,
+        test_case: LLMTestCase,
         _show_indicator: bool = True,
         _in_component: bool = False,
     ) -> float:
-        check_mllm_test_case_params(test_case, required_params, 0, 1, self)
+        check_mllm_test_case_params(
+            test_case, required_params, 0, 1, self, self.model
+        )
         self.evaluation_cost = 0 if self.using_native_model else None
         with metric_progress_indicator(
@@ -112,10 +121,12 @@ class TextToImageMetric(BaseMultimodalMetric):
             _show_indicator=_show_indicator,
             _in_component=_in_component,
         ):
-            input_texts, _ = self.separate_images_from_text(test_case.input)
-            _, output_images = self.separate_images_from_text(
+            input = convert_to_multi_modal_array(test_case.input)
+            actual_output = convert_to_multi_modal_array(
                 test_case.actual_output
             )
+            input_texts, _ = self.separate_images_from_text(input)
+            _, output_images = self.separate_images_from_text(actual_output)
             (self.SC_scores, self.SC_reasoning), (
                 self.PQ_scores,
                 self.PQ_reasoning,
@@ -165,27 +176,27 @@ class TextToImageMetric(BaseMultimodalMetric):
     ) -> Tuple[List[int], str]:
         images: List[MLLMImage] = []
         images.append(actual_image_output)
-        prompt = [
-            TextToImageTemplate.generate_semantic_consistency_evaluation_results(
-                text_prompt=text_prompt
-            )
-        ]
+        prompt = f"""
+            {
+                TextToImageTemplate.generate_semantic_consistency_evaluation_results(
+                    text_prompt=text_prompt
+                )
+            }
+            Images:
+            {images}
+        """
         if self.using_native_model:
-            res, cost = await self.model.a_generate(
-                prompt + images, ReasonScore
-            )
+            res, cost = await self.model.a_generate(prompt, ReasonScore)
             self.evaluation_cost += cost
             return res.score, res.reasoning
         else:
             try:
                 res: ReasonScore = await self.model.a_generate(
-                    prompt + images, schema=ReasonScore
+                    prompt, schema=ReasonScore
                 )
                 return res.score, res.reasoning
             except TypeError:
-                res = await self.model.a_generate(
-                    prompt + images, input_text=prompt
-                )
+                res = await self.model.a_generate(prompt, input_text=prompt)
                 data = trimAndLoadJson(res, self)
                 return data["score"], data["reasoning"]
@@ -196,23 +207,27 @@ class TextToImageMetric(BaseMultimodalMetric):
     ) -> Tuple[List[int], str]:
         images: List[MLLMImage] = []
         images.append(actual_image_output)
-        prompt = [
-            TextToImageTemplate.generate_semantic_consistency_evaluation_results(
-                text_prompt=text_prompt
-            )
-        ]
+        prompt = f"""
+            {
+                TextToImageTemplate.generate_semantic_consistency_evaluation_results(
+                    text_prompt=text_prompt
+                )
+            }
+            Images:
+            {images}
+        """
         if self.using_native_model:
-            res, cost = self.model.generate(prompt + images, ReasonScore)
+            res, cost = self.model.generate(prompt, ReasonScore)
             self.evaluation_cost += cost
             return res.score, res.reasoning
         else:
             try:
                 res: ReasonScore = self.model.generate(
-                    prompt + images, schema=ReasonScore
+                    prompt, schema=ReasonScore
                 )
                 return res.score, res.reasoning
             except TypeError:
-                res = self.model.generate(prompt + images)
+                res = self.model.generate(prompt)
                 data = trimAndLoadJson(res, self)
                 return data["score"], data["reasoning"]
@@ -220,23 +235,25 @@ class TextToImageMetric(BaseMultimodalMetric):
         self, actual_image_output: MLLMImage
     ) -> Tuple[List[int], str]:
         images: List[MLLMImage] = [actual_image_output]
-        prompt = [
-            TextToImageTemplate.generate_perceptual_quality_evaluation_results()
-        ]
+        prompt = f"""
+            {
+                TextToImageTemplate.generate_perceptual_quality_evaluation_results()
+            }
+            Images:
+            {images}
+        """
         if self.using_native_model:
-            res, cost = await self.model.a_generate(
-                prompt + images, ReasonScore
-            )
+            res, cost = await self.model.a_generate(prompt, ReasonScore)
             self.evaluation_cost += cost
             return res.score, res.reasoning
         else:
             try:
                 res: ReasonScore = await self.model.a_generate(
-                    prompt + images, schema=ReasonScore
+                    prompt, schema=ReasonScore
                 )
                 return res.score, res.reasoning
             except TypeError:
-                res = await self.model.a_generate(prompt + images)
+                res = await self.model.a_generate(prompt)
                 data = trimAndLoadJson(res, self)
                 return data["score"], data["reasoning"]
@@ -244,9 +261,13 @@ class TextToImageMetric(BaseMultimodalMetric):
         self, actual_image_output: MLLMImage
     ) -> Tuple[List[int], str]:
         images: List[MLLMImage] = [actual_image_output]
-        prompt = [
-            TextToImageTemplate.generate_perceptual_quality_evaluation_results()
-        ]
+        prompt = f"""
+            {
+                TextToImageTemplate.generate_perceptual_quality_evaluation_results()
+            }
+            Images:
+            {images}
+        """
         if self.using_native_model:
             res, cost = self.model.generate(prompt + images, ReasonScore)
             self.evaluation_cost += cost
@@ -254,11 +275,11 @@ class TextToImageMetric(BaseMultimodalMetric):
         else:
             try:
                 res: ReasonScore = self.model.generate(
-                    prompt + images, schema=ReasonScore
+                    prompt, schema=ReasonScore
                 )
                 return res.score, res.reasoning
             except TypeError:
-                res = self.model.generate(prompt + images)
+                res = self.model.generate(prompt)
                 data = trimAndLoadJson(res, self)
                 return data["score"], data["reasoning"]

deepeval/metrics/non_advice/template.py CHANGED Viewed

@@ -42,8 +42,8 @@ Example JSON:
 {{
     "verdicts": [
         {{
-            "verdict": "yes",
-            "reason": "Contains specific investment advice"
+            "reason": "Contains specific investment advice",
+            "verdict": "yes"
         }}
     ]
 }}

deepeval/metrics/pii_leakage/template.py CHANGED Viewed

@@ -39,8 +39,8 @@ Example JSON:
 {{
     "verdicts": [
         {{
-            "verdict": "yes",
-            "reason": "Contains personal phone number"
+            "reason": "Contains personal phone number",
+            "verdict": "yes"
         }}
     ]
 }}

deepeval/metrics/prompt_alignment/template.py CHANGED Viewed

@@ -26,12 +26,12 @@ Example JSON:
             "verdict": "yes"
         }},
         {{
-            "verdict": "no",
-            "reason": "The LLM corrected the user when the user used the wrong grammar in asking about the number of stars in the sky."
+            "reason": "The LLM corrected the user when the user used the wrong grammar in asking about the number of stars in the sky.",
+            "verdict": "no"
         }},
         {{
-            "verdict": "no",
-            "reason": "The LLM only made 'HEY THERE' uppercase, which does not follow the instruction of making everything uppercase completely."
+            "reason": "The LLM only made 'HEY THERE' uppercase, which does not follow the instruction of making everything uppercase completely.",
+            "verdict": "no"
         }}
     ]
 }}

deepeval/metrics/ragas.py CHANGED Viewed

@@ -10,7 +10,7 @@ from deepeval.telemetry import capture_metric_type
 # check langchain availability
 try:
-    import langchain_core
+    import langchain_core  # noqa: F401
     from langchain_core.language_models import BaseChatModel
     from langchain_core.embeddings import Embeddings
@@ -501,7 +501,7 @@ class RagasMetric(BaseMetric):
     def measure(self, test_case: LLMTestCase):
         # sends to server
         try:
-            from ragas import evaluate
+            from ragas import evaluate  # noqa: F401
         except ModuleNotFoundError:
             raise ModuleNotFoundError(
                 "Please install ragas to use this metric. `pip install ragas`."
@@ -509,7 +509,7 @@ class RagasMetric(BaseMetric):
         try:
             # How do i make sure this isn't just huggingface dataset
-            from datasets import Dataset
+            from datasets import Dataset  # noqa: F401
         except ModuleNotFoundError:
             raise ModuleNotFoundError("Please install dataset")

deepeval/metrics/role_violation/template.py CHANGED Viewed

@@ -39,8 +39,8 @@ Example JSON:
 {{
     "verdicts": [
         {{
-            "verdict": "yes",
-            "reason": "AI is pretending to be human"
+            "reason": "AI is pretending to be human",
+            "verdict": "yes"
         }}
     ]
 }}

deepeval/metrics/step_efficiency/step_efficiency.py CHANGED Viewed

@@ -231,4 +231,4 @@ class StepEfficiencyMetric(BaseMetric):
     @property
     def __name__(self):
-        return "Execution Efficiency"
+        return "Step Efficiency"

deepeval/metrics/tool_correctness/tool_correctness.py CHANGED Viewed

@@ -83,7 +83,7 @@ class ToolCorrectnessMetric(BaseMetric):
                 self.tools_called: List[ToolCall] = test_case.tools_called
                 self.expected_tools: List[ToolCall] = test_case.expected_tools
                 tool_calling_score = self._calculate_score()
-                if self.available_tools:
+                if self.available_tools and not test_case.multimodal:
                     tool_selection_score = self._get_tool_selection_score(
                         test_case.input,
                         test_case.tools_called,
@@ -177,7 +177,7 @@ class ToolCorrectnessMetric(BaseMetric):
             self.tools_called: List[ToolCall] = test_case.tools_called
             self.expected_tools: List[ToolCall] = test_case.expected_tools
             tool_calling_score = self._calculate_score()
-            if self.available_tools:
+            if self.available_tools and not test_case.multimodal:
                 tool_selection_score = await self._a_get_tool_selection_score(
                     test_case.input,
                     test_case.tools_called,

deepeval/metrics/toxicity/template.py CHANGED Viewed

@@ -71,15 +71,15 @@ Example JSON:
 {{
     "verdicts": [
         {{
-            "verdict": "yes",
-            "reason": "The actual output uses words like "garbage," "tasteless weeds," etc., ridicules vegan food and those who choose it, and attacks people who enjoy vegan food, calling them "pretentious losers. Furthermore, it dismisses the entire concept of vegan cuisine with a sweeping generalization, making it seem worthless."
+            "reason": "The actual output uses words like "garbage," "tasteless weeds," etc., ridicules vegan food and those who choose it, and attacks people who enjoy vegan food, calling them "pretentious losers. Furthermore, it dismisses the entire concept of vegan cuisine with a sweeping generalization, making it seem worthless.",
+            "verdict": "yes"
         }},
         {{
             "verdict": "no"
         }},
         {{
-            "verdict": "yes":
-            "reason": "'Sarah always meant well' sounds positive but are undermined by the surrounding criticism such as 'can't help but sign', which can be considered a personal attack."
+            "reason": "'Sarah always meant well' sounds positive but are undermined by the surrounding criticism such as 'can't help but sign', which can be considered a personal attack.",
+            "verdict": "yes"
         }}
     ]
 }}

deepeval/metrics/turn_contextual_precision/schema.py ADDED Viewed

@@ -0,0 +1,21 @@
+from typing import List
+from pydantic import BaseModel
+class ContextualPrecisionVerdict(BaseModel):
+    verdict: str
+    reason: str
+class Verdicts(BaseModel):
+    verdicts: List[ContextualPrecisionVerdict]
+class ContextualPrecisionScoreReason(BaseModel):
+    reason: str
+class InteractionContextualPrecisionScore(BaseModel):
+    score: float
+    reason: str
+    verdicts: List[ContextualPrecisionVerdict]

deepeval/metrics/turn_contextual_precision/template.py ADDED Viewed

@@ -0,0 +1,187 @@
+from typing import List, Dict, Union
+import textwrap
+from deepeval.test_case import MLLMImage
+class TurnContextualPrecisionTemplate:
+    multimodal_rules = """
+        --- MULTIMODAL INPUT RULES ---
+        - Treat image content as factual evidence.
+        - Only reference visual details that are explicitly and clearly visible.
+        - Do not infer or guess objects, text, or details not visibly present.
+        - If an image is unclear or ambiguous, mark uncertainty explicitly.
+        - When evaluating claims, compare them to BOTH textual and visual evidence.
+        - If the claim references something not clearly visible, respond with 'idk'.
+    """
+    @staticmethod
+    def generate_verdicts(
+        input: str,
+        expected_outcome: str,
+        retrieval_context: List[str],
+        multimodal: bool = False,
+    ):
+        document_count_str = f" ({len(retrieval_context)} document{'s' if len(retrieval_context) > 1 else ''})"
+        # For multimodal, we need to annotate the retrieval context with node IDs
+        context_to_display = (
+            TurnContextualPrecisionTemplate.id_retrieval_context(
+                retrieval_context
+            )
+            if multimodal
+            else retrieval_context
+        )
+        multimodal_note = (
+            " (which can be text or an image)" if multimodal else ""
+        )
+        prompt_template = textwrap.dedent(
+            f"""Given the user message, assistant output, and retrieval context, please generate a list of JSON objects to determine whether each node in the retrieval context was remotely useful in arriving at the assistant output.
+            {TurnContextualPrecisionTemplate.multimodal_rules if multimodal else ""}
+            **
+            IMPORTANT: Please make sure to only return in JSON format, with the 'verdicts' key as a list of JSON. These JSON only contain the `verdict` key that outputs only 'yes' or 'no', and a `reason` key to justify the verdict. In your reason, you should aim to quote parts of the context {multimodal_note}.
+            Example Retrieval Context: ["Einstein won the Nobel Prize for his discovery of the photoelectric effect", "He won the Nobel Prize in 1968.", "There was a cat."]
+            Example User Message: "Who won the Nobel Prize in 1968 and for what?"
+            Example Assistant Output: "Einstein won the Nobel Prize in 1968 for his discovery of the photoelectric effect."
+            Example:
+            {{
+                "verdicts": [
+                    {{
+                        "reason": "It clearly addresses the question by stating that 'Einstein won the Nobel Prize for his discovery of the photoelectric effect.'",
+                        "verdict": "yes"
+                    }},
+                    {{
+                        "reason": "The text verifies that the prize was indeed won in 1968.",
+                        "verdict": "yes"
+                    }},
+                    {{
+                        "reason": "'There was a cat' is not at all relevant to the topic of winning a Nobel Prize.",
+                        "verdict": "no"
+                    }}
+                ]
+            }}
+            Since you are going to generate a verdict for each context, the number of 'verdicts' SHOULD BE STRICTLY EQUAL to that of the contexts.
+            **
+            User Message:
+            {input}
+            Assistant Output:
+            {expected_outcome}
+            Retrieval Context{document_count_str}:
+            {context_to_display}
+            JSON:
+            """
+        )
+        return prompt_template
+    @staticmethod
+    def generate_reason(
+        input: str,
+        score: float,
+        verdicts: List[Dict[str, str]],
+        multimodal: bool = False,
+    ):
+        return textwrap.dedent(
+            f"""Given the user message, retrieval contexts, and contextual precision score, provide a CONCISE {'summarize' if multimodal else 'summary'} for the score. Explain why it is not higher, but also why it is at its current score.
+            The retrieval contexts is a list of JSON with three keys: `verdict`, `reason` (reason for the verdict) and `node`. `verdict` will be either 'yes' or 'no', which represents whether the corresponding 'node' in the retrieval context is relevant to the user message.
+            Contextual precision represents if the relevant nodes are ranked higher than irrelevant nodes. Also note that retrieval contexts is given IN THE ORDER OF THEIR RANKINGS.
+            {TurnContextualPrecisionTemplate.multimodal_rules if multimodal else ""}
+            **
+            IMPORTANT: Please make sure to only return in JSON format, with the 'reason' key providing the reason.
+            Example JSON:
+            {{
+                "reason": "The score is <contextual_precision_score> because <your_reason>."
+            }}
+            DO NOT mention 'verdict' in your reason, but instead phrase it as irrelevant nodes. The term 'verdict' {'are' if multimodal else 'is'} just here for you to understand the broader scope of things.
+            Also DO NOT mention there are `reason` fields in the retrieval contexts you are presented with, instead just use the information in the `reason` field.
+            In your reason, you MUST USE the `reason`, QUOTES in the 'reason', and the node RANK (starting from 1, eg. first node) to explain why the 'no' verdicts should be ranked lower than the 'yes' verdicts.
+            When addressing nodes, make it explicit that {'it is' if multimodal else 'they are'} nodes in {'retrieval context' if multimodal else 'retrieval contexts'}.
+            If the score is 1, keep it short and say something positive with an upbeat tone (but don't overdo it{',' if multimodal else ''} otherwise it gets annoying).
+            **
+            Contextual Precision Score:
+            {score}
+            User Message:
+            {input}
+            Retrieval Contexts:
+            {verdicts}
+            JSON:
+            """
+        )
+    @staticmethod
+    def generate_final_reason(
+        final_score: float, success: bool, reasons: List[str]
+    ):
+        return textwrap.dedent(
+            f"""You are an AI evaluator producing a single final explanation for the TurnContextualPrecisionMetric result.
+            Context:
+            This metric evaluates conversational contextual precision by determining whether relevant nodes in retrieval context are ranked higher than irrelevant nodes for each interaction. Each interaction yields a reason indicating why relevant nodes were well-ranked or poorly-ranked. You are given all those reasons.
+            Inputs:
+            - final_score: the averaged score across all interactions.
+            - success: whether the metric passed or failed
+            - reasons: a list of textual reasons generated from individual interactions.
+            Instructions:
+            1. Read all reasons and synthesize them into one unified explanation.
+            2. Describe patterns of ranking issues, irrelevant nodes appearing before relevant ones, or well-structured retrieval contexts if present.
+            3. Do not repeat every reason; merge them into a concise, coherent narrative.
+            4. If the metric failed, state the dominant failure modes. If it passed, state why the retrieval context ranking was effective.
+            5. Output a single paragraph with no lists, no bullets, no markup.
+            Output:
+            A single paragraph explaining the final outcome.
+            Here's the inputs:
+            Final Score: {final_score}
+            Reasons:
+            {reasons}
+            Success: {success}
+            Now give me a final reason that explains why the metric passed or failed. Output ONLY the reason and nothing else.
+            The final reason:
+            """
+        )
+    @staticmethod
+    def id_retrieval_context(
+        retrieval_context: List[str],
+    ) -> List[Union[str, MLLMImage]]:
+        """
+        Annotates retrieval context with node IDs for multimodal processing.
+        Args:
+            retrieval_context: List of contexts (can be strings or MLLMImages)
+        Returns:
+            Annotated list with "Node X:" prefixes
+        """
+        annotated_retrieval_context = []
+        for i, context in enumerate(retrieval_context):
+            if isinstance(context, str):
+                annotated_retrieval_context.append(f"Node {i + 1}: {context}")
+            elif isinstance(context, MLLMImage):
+                annotated_retrieval_context.append(f"Node {i + 1}:")
+                annotated_retrieval_context.append(context)
+        return annotated_retrieval_context

deepeval 3.7.3__py3-none-any.whl → 3.7.5__py3-none-any.whl

deepeval 3.7.3py3-none-any.whl → 3.7.5py3-none-any.whl