PyPI - mteb - Versions diffs - 2.1.7__py3-none-any.whl → 2.1.9__py3-none-any.whl - Mend

mteb 2.1.7py3-none-any.whl → 2.1.9py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (243) hide show

mteb/tasks/retrieval/vie/cqa_dupstack_mathematica_vn_retrieval.py CHANGED Viewed

@@ -5,11 +5,7 @@ from mteb.abstasks.task_metadata import TaskMetadata
 class CQADupstackMathematicaVN(AbsTaskRetrieval):
     metadata = TaskMetadata(
         name="CQADupstackMathematica-VN",
-        description="""A translated dataset from CQADupStack: A Benchmark Data Set for Community Question-Answering Research
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from CQADupStack: A Benchmark Data Set for Community Question-Answering Research The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="http://nlp.cis.unimelb.edu.au/resources/cqadupstack/",
         dataset={
             "path": "GreenNode/cqadupstack-mathematica-vn",

mteb/tasks/retrieval/vie/cqa_dupstack_physics_vn_retrieval.py CHANGED Viewed

@@ -5,11 +5,7 @@ from mteb.abstasks.task_metadata import TaskMetadata
 class CQADupstackPhysicsVN(AbsTaskRetrieval):
     metadata = TaskMetadata(
         name="CQADupstackPhysics-VN",
-        description="""A translated dataset from CQADupStack: A Benchmark Data Set for Community Question-Answering Research
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from CQADupStack: A Benchmark Data Set for Community Question-Answering Research The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="http://nlp.cis.unimelb.edu.au/resources/cqadupstack/",
         dataset={
             "path": "GreenNode/cqadupstack-physics-vn",

mteb/tasks/retrieval/vie/cqa_dupstack_programmers_vn_retrieval.py CHANGED Viewed

@@ -5,11 +5,7 @@ from mteb.abstasks.task_metadata import TaskMetadata
 class CQADupstackProgrammersRetrievalVN(AbsTaskRetrieval):
     metadata = TaskMetadata(
         name="CQADupstackProgrammers-VN",
-        description="""A translated dataset from CQADupStack: A Benchmark Data Set for Community Question-Answering Research
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from CQADupStack: A Benchmark Data Set for Community Question-Answering Research The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="http://nlp.cis.unimelb.edu.au/resources/cqadupstack/",
         dataset={
             "path": "GreenNode/cqadupstack-programmers-vn",

mteb/tasks/retrieval/vie/cqa_dupstack_stats_vn_retrieval.py CHANGED Viewed

@@ -5,11 +5,7 @@ from mteb.abstasks.task_metadata import TaskMetadata
 class CQADupstackStatsVN(AbsTaskRetrieval):
     metadata = TaskMetadata(
         name="CQADupstackStats-VN",
-        description="""A translated dataset from CQADupStack: A Benchmark Data Set for Community Question-Answering Research
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from CQADupStack: A Benchmark Data Set for Community Question-Answering Research The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="http://nlp.cis.unimelb.edu.au/resources/cqadupstack/",
         dataset={
             "path": "GreenNode/cqadupstack-stats-vn",

mteb/tasks/retrieval/vie/cqa_dupstack_tex_vn_retrieval.py CHANGED Viewed

@@ -5,11 +5,7 @@ from mteb.abstasks.task_metadata import TaskMetadata
 class CQADupstackTexVN(AbsTaskRetrieval):
     metadata = TaskMetadata(
         name="CQADupstackTex-VN",
-        description="""A translated dataset from CQADupStack: A Benchmark Data Set for Community Question-Answering Research
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from CQADupStack: A Benchmark Data Set for Community Question-Answering Research The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="http://nlp.cis.unimelb.edu.au/resources/cqadupstack/",
         dataset={
             "path": "GreenNode/cqadupstack-tex-vn",

mteb/tasks/retrieval/vie/cqa_dupstack_unix_vn_retrieval.py CHANGED Viewed

@@ -5,11 +5,7 @@ from mteb.abstasks.task_metadata import TaskMetadata
 class CQADupstackUnixVN(AbsTaskRetrieval):
     metadata = TaskMetadata(
         name="CQADupstackUnix-VN",
-        description="""A translated dataset from CQADupStack: A Benchmark Data Set for Community Question-Answering Research
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from CQADupStack: A Benchmark Data Set for Community Question-Answering Research The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="http://nlp.cis.unimelb.edu.au/resources/cqadupstack/",
         dataset={
             "path": "GreenNode/cqadupstack-unix-vn",

mteb/tasks/retrieval/vie/cqa_dupstack_webmasters_vn_retrieval.py CHANGED Viewed

@@ -5,11 +5,7 @@ from mteb.abstasks.task_metadata import TaskMetadata
 class CQADupstackWebmastersVN(AbsTaskRetrieval):
     metadata = TaskMetadata(
         name="CQADupstackWebmasters-VN",
-        description="""A translated dataset from CQADupStack: A Benchmark Data Set for Community Question-Answering Research
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from CQADupStack: A Benchmark Data Set for Community Question-Answering Research The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="http://nlp.cis.unimelb.edu.au/resources/cqadupstack/",
         dataset={
             "path": "GreenNode/cqadupstack-webmasters-vn",

mteb/tasks/retrieval/vie/cqa_dupstack_wordpress_vn_retrieval.py CHANGED Viewed

@@ -9,11 +9,7 @@ class CQADupstackWordpressVN(AbsTaskRetrieval):
             "path": "GreenNode/cqadupstack-wordpress-vn",
             "revision": "2230f80e1baf42aa005731ca86577621c566fcd7",
         },
-        description="""A translated dataset from CQADupStack: A Benchmark Data Set for Community Question-Answering Research
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from CQADupStack: A Benchmark Data Set for Community Question-Answering Research The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="http://nlp.cis.unimelb.edu.au/resources/cqadupstack/",
         type="Retrieval",
         category="t2t",

mteb/tasks/retrieval/vie/db_pedia_vn_retrieval.py CHANGED Viewed

@@ -5,11 +5,7 @@ from mteb.abstasks.task_metadata import TaskMetadata
 class DBPediaVN(AbsTaskRetrieval):
     metadata = TaskMetadata(
         name="DBPedia-VN",
-        description="""A translated dataset from DBpedia-Entity is a standard test collection for entity search over the DBpedia knowledge base
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from DBpedia-Entity is a standard test collection for entity search over the DBpedia knowledge base The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="https://github.com/iai-group/DBpedia-Entity/",
         dataset={
             "path": "GreenNode/dbpedia-vn",

mteb/tasks/retrieval/vie/fevervn_retrieval.py CHANGED Viewed

@@ -9,13 +9,7 @@ class FEVERVN(AbsTaskRetrieval):
             "path": "GreenNode/fever-vn",
             "revision": "a543dd8b98aed3603110c01d26db05ba39b87d49",
         },
-        description="""A translated dataset from FEVER (Fact Extraction and VERification) consists of 185,445 claims generated by altering sentences
-            extracted from Wikipedia and subsequently verified without knowledge of the sentence they were
-            derived from.
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from FEVER (Fact Extraction and VERification) consists of 185,445 claims generated by altering sentences extracted from Wikipedia and subsequently verified without knowledge of the sentence they were derived from. The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="https://fever.ai/",
         type="Retrieval",
         category="t2t",

mteb/tasks/retrieval/vie/fi_qa2018_vn_retrieval.py CHANGED Viewed

@@ -5,11 +5,7 @@ from mteb.abstasks.task_metadata import TaskMetadata
 class FiQA2018VN(AbsTaskRetrieval):
     metadata = TaskMetadata(
         name="FiQA2018-VN",
-        description="""A translated dataset from Financial Opinion Mining and Question Answering
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from Financial Opinion Mining and Question Answering The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="https://sites.google.com/view/fiqa/",
         dataset={
             "path": "GreenNode/fiqa-vn",

mteb/tasks/retrieval/vie/hotpot_qavn_retrieval.py CHANGED Viewed

@@ -9,12 +9,7 @@ class HotpotQAVN(AbsTaskRetrieval):
             "path": "GreenNode/hotpotqa-vn",
             "revision": "8a5220c7af5084f0d5d2afeb74f9c2b41b759ff0",
         },
-        description="""A translated dataset from HotpotQA is a question answering dataset featuring natural, multi-hop questions, with strong
-            supervision for supporting facts to enable more explainable question answering systems.
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from HotpotQA is a question answering dataset featuring natural, multi-hop questions, with strong supervision for supporting facts to enable more explainable question answering systems. The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="https://hotpotqa.github.io/",
         type="Retrieval",
         category="t2t",

mteb/tasks/retrieval/vie/msmarcovn_retrieval.py CHANGED Viewed

@@ -9,11 +9,7 @@ class MSMARCOVN(AbsTaskRetrieval):
             "path": "GreenNode/msmarco-vn",
             "revision": "85d1ad4cc9070b8d019d65f5af1631a2ab91e294",
         },
-        description="""A translated dataset from MS MARCO is a collection of datasets focused on deep learning in search
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from MS MARCO is a collection of datasets focused on deep learning in search The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="https://microsoft.github.io/msmarco/",
         type="Retrieval",
         category="t2t",

mteb/tasks/retrieval/vie/nf_corpus_vn_retrieval.py CHANGED Viewed

@@ -9,11 +9,7 @@ class NFCorpusVN(AbsTaskRetrieval):
             "path": "GreenNode/nfcorpus-vn",
             "revision": "a13d72fbb859be3dc19ab669d1ec9510407d2dcd",
         },
-        description="""A translated dataset from NFCorpus: A Full-Text Learning to Rank Dataset for Medical Information Retrieval
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from NFCorpus: A Full-Text Learning to Rank Dataset for Medical Information Retrieval The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="https://www.cl.uni-heidelberg.de/statnlpgroup/nfcorpus/",
         type="Retrieval",
         category="t2t",

mteb/tasks/retrieval/vie/nqvn_retrieval.py CHANGED Viewed

@@ -9,11 +9,7 @@ class NQVN(AbsTaskRetrieval):
             "path": "GreenNode/nq-vn",
             "revision": "40a6d7f343b9c9f4855a426d8c431ad5f8aaf56b",
         },
-        description="""A translated dataset from NFCorpus: A Full-Text Learning to Rank Dataset for Medical Information Retrieval
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from NFCorpus: A Full-Text Learning to Rank Dataset for Medical Information Retrieval The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="https://ai.google.com/research/NaturalQuestions/",
         type="Retrieval",
         category="t2t",

mteb/tasks/retrieval/vie/quora_vn_retrieval.py CHANGED Viewed

@@ -9,12 +9,7 @@ class QuoraVN(AbsTaskRetrieval):
             "path": "GreenNode/quora-vn",
             "revision": "3363d81e41b67c1032bf3b234882a03d271e2289",
         },
-        description="""A translated dataset from QuoraRetrieval is based on questions that are marked as duplicates on the Quora platform. Given a
-            question, find other (duplicate) questions.
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from QuoraRetrieval is based on questions that are marked as duplicates on the Quora platform. Given a question, find other (duplicate) questions. The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="https://quoradata.quora.com/First-Quora-Dataset-Release-Question-Pairs",
         type="Retrieval",
         category="t2t",

mteb/tasks/retrieval/vie/sci_fact_vn_retrieval.py CHANGED Viewed

@@ -9,11 +9,7 @@ class SciFactVN(AbsTaskRetrieval):
             "path": "GreenNode/scifact-vn",
             "revision": "483a7cf890c523c954e7751d328c5bb65061dcff",
         },
-        description="""A translated dataset from SciFact verifies scientific claims using evidence from the research literature containing scientific paper abstracts.
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from SciFact verifies scientific claims using evidence from the research literature containing scientific paper abstracts. The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="https://github.com/allenai/scifact",
         type="Retrieval",
         category="t2t",

mteb/tasks/retrieval/vie/scidocsvn_retrieval.py CHANGED Viewed

@@ -9,12 +9,7 @@ class SCIDOCSVN(AbsTaskRetrieval):
             "path": "GreenNode/scidocs-vn",
             "revision": "724cddfa9d328a193f303a0a9b7789468ac79f26",
         },
-        description="""A translated dataset from SciDocs, a new evaluation benchmark consisting of seven document-level tasks ranging from citation
-            prediction, to document classification and recommendation.
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from SciDocs, a new evaluation benchmark consisting of seven document-level tasks ranging from citation prediction, to document classification and recommendation. The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="https://allenai.org/data/scidocs",
         type="Retrieval",
         category="t2t",

mteb/tasks/retrieval/vie/touche2020_vn_retrieval.py CHANGED Viewed

@@ -5,11 +5,7 @@ from mteb.abstasks.task_metadata import TaskMetadata
 class Touche2020VN(AbsTaskRetrieval):
     metadata = TaskMetadata(
         name="Touche2020-VN",
-        description="""A translated dataset from Touché Task 1: Argument Retrieval for Controversial Questions
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from Touché Task 1: Argument Retrieval for Controversial Questions The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="https://webis.de/events/touche-20/shared-task-1.html",
         dataset={
             "path": "GreenNode/webis-touche2020-vn",

mteb/tasks/retrieval/vie/treccovidvn_retrieval.py CHANGED Viewed

@@ -5,11 +5,7 @@ from mteb.abstasks.task_metadata import TaskMetadata
 class TRECCOVIDVN(AbsTaskRetrieval):
     metadata = TaskMetadata(
         name="TRECCOVID-VN",
-        description="""A translated dataset from TRECCOVID is an ad-hoc search challenge based on the COVID-19 dataset containing scientific articles related to the COVID-19 pandemic.
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from TRECCOVID is an ad-hoc search challenge based on the COVID-19 dataset containing scientific articles related to the COVID-19 pandemic. The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="https://ir.nist.gov/covidSubmit/index.html",
         dataset={
             "path": "GreenNode/trec-covid-vn",

mteb/tasks/sts/vie/biosses_stsvn.py CHANGED Viewed

@@ -9,11 +9,7 @@ class BiossesSTSVN(AbsTaskSTS):
             "path": "GreenNode/biosses-sts-vn",
             "revision": "1dae4a6df91c0852680cd4ab48c8c1d8a9ed49b2",
         },
-        description="""A translated dataset from Biomedical Semantic Similarity Estimation.
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from Biomedical Semantic Similarity Estimation. The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="https://tabilab.cmpe.boun.edu.tr/BIOSSES/DataSet.html",
         type="STS",
         category="t2c",

mteb/tasks/sts/vie/sickr_stsvn.py CHANGED Viewed

@@ -9,11 +9,7 @@ class SickrSTSVN(AbsTaskSTS):
             "path": "GreenNode/sickr-sts-vn",
             "revision": "bc89f0401983c456b609f7fb324278f346b2cccf",
         },
-        description="""A translated dataset from Semantic Textual Similarity SICK-R dataset as described here:
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from Semantic Textual Similarity SICK-R dataset as described here: The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="https://aclanthology.org/2020.lrec-1.207",
         type="STS",
         category="t2c",

mteb/tasks/sts/vie/sts_benchmark_stsvn.py CHANGED Viewed

@@ -9,11 +9,7 @@ class STSBenchmarkSTSVN(AbsTaskSTS):
             "path": "GreenNode/stsbenchmark-sts-vn",
             "revision": "f24d66738cda4a02138ada5af7689a92ce1fcad6",
         },
-        description="""A translated dataset from Semantic Textual Similarity Benchmark (STSbenchmark) dataset.
-            The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system:
-            - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation.
-            - Applies advanced embedding models to filter the translations.
-            - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.""",
+        description="A translated dataset from Semantic Textual Similarity Benchmark (STSbenchmark) dataset. The process of creating the VN-MTEB (Vietnamese Massive Text Embedding Benchmark) from English samples involves a new automated system: - The system uses large language models (LLMs), specifically Coherence's Aya model, for translation. - Applies advanced embedding models to filter the translations. - Use LLM-as-a-judge to scoring the quality of the samples base on multiple criteria.",
         reference="https://github.com/PhilipMay/stsb-multi-mt/",
         type="STS",
         category="t2c",

mteb/tasks/zeroshot_classification/eng/gtsrb.py CHANGED Viewed

@@ -9,7 +9,7 @@ from mteb.abstasks.zeroshot_classification import (
 class GTSRBZeroShotClassification(AbsTaskZeroShotClassification):
     metadata = TaskMetadata(
         name="GTSRBZeroShot",
-        description="""The German Traffic Sign Recognition Benchmark (GTSRB) is a multi-class classification dataset for traffic signs. It consists of dataset of more than 50,000 traffic sign images. The dataset comprises 43 classes with unbalanced class frequencies.""",
+        description="The German Traffic Sign Recognition Benchmark (GTSRB) is a multi-class classification dataset for traffic signs. It consists of dataset of more than 50,000 traffic sign images. The dataset comprises 43 classes with unbalanced class frequencies.",
         reference="https://benchmark.ini.rub.de/",
         dataset={
             "path": "clip-benchmark/wds_gtsrb",

mteb/tasks/zeroshot_classification/eng/patch_camelyon.py CHANGED Viewed

@@ -9,7 +9,7 @@ from mteb.abstasks.zeroshot_classification import (
 class PatchCamelyonZeroShotClassification(AbsTaskZeroShotClassification):
     metadata = TaskMetadata(
         name="PatchCamelyonZeroShot",
-        description="""Histopathology diagnosis classification dataset.""",
+        description="Histopathology diagnosis classification dataset.",
         reference="https://link.springer.com/chapter/10.1007/978-3-030-00934-2_24",
         dataset={
             "path": "clip-benchmark/wds_vtab-pcam",

mteb/tasks/zeroshot_classification/eng/ucf101.py CHANGED Viewed

@@ -7,11 +7,7 @@ from mteb.abstasks.zeroshot_classification import (
 class UCF101ZeroShotClassification(AbsTaskZeroShotClassification):
     metadata = TaskMetadata(
         name="UCF101ZeroShot",
-        description="""UCF101 is an action recognition data set of realistic
-action videos collected from YouTube, having 101 action categories. This
-version of the dataset does not contain images but images saved frame by
-frame. Train and test splits are generated based on the authors' first
-version train/test list.""",
+        description="UCF101 is an action recognition data set of realistic action videos collected from YouTube, having 101 action categories. This version of the dataset does not contain images but images saved frame by frame. Train and test splits are generated based on the authors' first version train/test list.",
         reference="https://huggingface.co/datasets/flwrlabs/ucf101",
         dataset={
             "path": "flwrlabs/ucf101",

{mteb-2.1.7.dist-info → mteb-2.1.9.dist-info}/METADATA RENAMED Viewed

@@ -1,6 +1,6 @@
 Metadata-Version: 2.4
 Name: mteb
-Version: 2.1.7
+Version: 2.1.9
 Summary: Massive Text Embedding Benchmark
 Author-email: MTEB Contributors <niklas@huggingface.co>, Kenneth Enevoldsen <kenneth.enevoldsen@cas.au.dk>, Nouamane Tazi <nouamane@huggingface.co>, Nils Reimers <info@nils-reimers.de>
 Maintainer-email: Kenneth Enevoldsen <kenneth.enevoldsen@cas.au.dk>, Roman Solomatin <risolomatin@gmail.com>, Isaac Chung <chungisaac1217@gmail.com>

mteb 2.1.7__py3-none-any.whl → 2.1.9__py3-none-any.whl

mteb 2.1.7py3-none-any.whl → 2.1.9py3-none-any.whl