From 537b974dfcbf958c7e6292dde9bb2f816342a20f Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" Date: Wed, 6 Nov 2024 13:09:56 +0000 Subject: [PATCH] Update tasks table --- docs/tasks.md | 1106 ++++++++++++++++++++++++------------------------- 1 file changed, 553 insertions(+), 553 deletions(-) diff --git a/docs/tasks.md b/docs/tasks.md index d90ac1816..4c1ab08e3 100644 --- a/docs/tasks.md +++ b/docs/tasks.md @@ -8,593 +8,593 @@ The following tables give you an overview of the tasks in MTEB. | Name | Languages | Type | Category | Domains | # Samples | Dataset statistics | |------|-----------|------|----------|---------|-----------|--------------------| | [AFQMC](https://aclanthology.org/2021.emnlp-main.357) | ['cmn'] | STS | s2s | | None | None | -| [AILACasedocs](https://zenodo.org/records/4063986) | ['eng'] | Retrieval | p2p | [Legal, Written] | None | {'test': {'average_document_length': 26948.344086021505, 'average_query_length': 3038.42, 'num_documents': 186, 'num_queries': 50, 'average_relevant_docs_per_query': 3.9}} | -| [AILAStatutes](https://zenodo.org/records/4063986) | ['eng'] | Retrieval | p2p | [Legal, Written] | None | {'test': {'average_document_length': 1973.6341463414635, 'average_query_length': 3038.42, 'num_documents': 82, 'num_queries': 50, 'average_relevant_docs_per_query': 4.34}} | -| [AJGT](https://link.springer.com/chapter/10.1007/978-3-319-60042-0_66/) (Alomari et al., 2017) | ['ara'] | Classification | s2s | [Social, Written] | {'train': 1800} | {'train': 46.81} | -| [ARCChallenge](https://allenai.org/data/arc) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | {'test': 1172} | {'test': {'average_document_length': 30.94235294117647, 'average_query_length': 131.56569965870307, 'num_documents': 9350, 'num_queries': 1172, 'average_relevant_docs_per_query': 1.0}} | +| [AILACasedocs](https://zenodo.org/records/4063986) | ['eng'] | Retrieval | p2p | [Legal, Written] | None | None | +| [AILAStatutes](https://zenodo.org/records/4063986) | ['eng'] | Retrieval | p2p | [Legal, Written] | None | None | +| [AJGT](https://link.springer.com/chapter/10.1007/978-3-319-60042-0_66/) (Alomari et al., 2017) | ['ara'] | Classification | s2s | [Social, Written] | None | None | +| [ARCChallenge](https://allenai.org/data/arc) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | None | None | | [ATEC](https://aclanthology.org/2021.emnlp-main.357) | ['cmn'] | STS | s2s | | None | None | -| [AfriSentiClassification](https://arxiv.org/abs/2302.08956) | ['amh', 'arq', 'ary', 'hau', 'ibo', 'kin', 'pcm', 'por', 'swa', 'tso', 'twi', 'yor'] | Classification | s2s | [Social, Written] | {'test': 2048} | {'test': 74.77} | -| [AfriSentiLangClassification](https://huggingface.co/datasets/HausaNLP/afrisenti-lid-data/) | ['amh', 'arq', 'ary', 'hau', 'ibo', 'kin', 'pcm', 'por', 'swa', 'tso', 'twi', 'yor'] | Classification | s2s | [Social, Written] | {'test': 5754} | {'test': 77.84} | -| [AllegroReviews](https://aclanthology.org/2020.acl-main.111.pdf) | ['pol'] | Classification | s2s | | {'test': 1006} | {'test': 477.2} | -| [AlloProfClusteringP2P.v2](https://huggingface.co/datasets/lyon-nlp/alloprof) (Lefebvre-Brossard et al., 2023) | ['fra'] | Clustering | p2p | [Encyclopaedic, Written] | {'test': 2556} | {'test': 3539.5} | -| [AlloProfClusteringS2S.v2](https://huggingface.co/datasets/lyon-nlp/alloprof) (Lefebvre-Brossard et al., 2023) | ['fra'] | Clustering | s2s | [Encyclopaedic, Written] | {'test': 2556} | {'test': 32.8} | -| [AlloprofReranking](https://huggingface.co/datasets/antoinelb7/alloprof) (Lefebvre-Brossard et al., 2023) | ['fra'] | Reranking | s2p | [Web, Academic, Written] | {'test': 2316, 'train': 9264} | None | -| [AlloprofRetrieval](https://huggingface.co/datasets/antoinelb7/alloprof) (Lefebvre-Brossard et al., 2023) | ['fra'] | Retrieval | s2p | [Encyclopaedic, Written] | {'train': 2048} | {'test': {'average_document_length': 3505.705399061033, 'average_query_length': 170.71286701208982, 'num_documents': 2556, 'num_queries': 2316, 'average_relevant_docs_per_query': 1.0}} | -| [AlphaNLI](https://leaderboard.allenai.org/anli/submissions/get-started) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | {'test': 1532} | {'test': {'average_document_length': 43.42647308646886, 'average_query_length': 103.05483028720627, 'num_documents': 241347, 'num_queries': 1532, 'average_relevant_docs_per_query': 1.0}} | -| [AmazonCounterfactualClassification](https://arxiv.org/abs/2104.06893) | ['deu', 'eng', 'jpn'] | Classification | s2s | [Reviews, Written] | {'validation': 335, 'test': 670} | {'validation': 109.2, 'test': 106.1} | -| [AmazonPolarityClassification](https://huggingface.co/datasets/amazon_polarity) (Julian McAuley, 2013) | ['eng'] | Classification | p2p | [Reviews, Written] | {'test': 400000} | {'test': 431.4} | -| [AmazonReviewsClassification](https://arxiv.org/abs/2010.02573) (Phillip Keung, 2020) | ['cmn', 'deu', 'eng', 'fra', 'jpn', 'spa'] | Classification | s2s | [Reviews, Written] | {'validation': 30000, 'test': 30000} | {'validation': 159.2, 'test': 160.4} | -| [AngryTweetsClassification](https://aclanthology.org/2021.nodalida-main.53/) (Pauli et al., 2021) | ['dan'] | Classification | s2s | [Social, Written] | {'test': 1050} | {'test': 156.1} | -| [AppsRetrieval](https://arxiv.org/abs/2105.09938) (Dan Hendrycks, 2021) | ['eng', 'python'] | Retrieval | p2p | [Programming, Written] | {'test': 1000} | {'test': {'average_document_length': 575.0086708499715, 'average_query_length': 1669.8284196547145, 'num_documents': 8765, 'num_queries': 3765, 'average_relevant_docs_per_query': 1.0}} | -| [ArEntail](https://link.springer.com/article/10.1007/s10579-024-09731-1) (Obeidat et al., 2024) | ['ara'] | PairClassification | s2s | [News, Written] | {'test': 1000} | {'test': 65.77} | -| [ArXivHierarchicalClusteringP2P](https://www.kaggle.com/Cornell-University/arxiv) | ['eng'] | Clustering | p2p | [Academic, Written] | {'test': 2048} | {'test': {'num_samples': 2048, 'average_text_length': 1008.439453125, 'average_labels_per_text': 1.46337890625, 'unique_labels': 129, 'labels': {'cs': {'count': 356}, 'math': {'count': 381}, 'OC': {'count': 11}, 'hep-lat': {'count': 13}, 'hep': {'count': 98}, 'astro-ph': {'count': 213}, 'eess': {'count': 76}, 'quant-ph': {'count': 135}, 'DC': {'count': 5}, 'cond-mat': {'count': 274}, 'hep-th': {'count': 66}, 'SP': {'count': 33}, 'hep-ph': {'count': 69}, 'FA': {'count': 6}, 'nucl-th': {'count': 17}, 'q-bio': {'count': 80}, 'HE': {'count': 22}, 'HC': {'count': 2}, 'stat': {'count': 60}, 'ML': {'count': 16}, 'IV': {'count': 13}, 'stat-mech': {'count': 47}, 'DS': {'count': 14}, 'ME': {'count': 12}, 'CC': {'count': 2}, 'mtrl-sci': {'count': 22}, 'PE': {'count': 16}, 'NT': {'count': 11}, 'SC': {'count': 6}, 'AG': {'count': 13}, 'physics': {'count': 81}, 'ins-det': {'count': 9}, 'GA': {'count': 18}, 'BM': {'count': 6}, 'GN': {'count': 17}, 'NA': {'count': 15}, 'app-ph': {'count': 7}, 'RT': {'count': 6}, 'other': {'count': 37}, 'soft': {'count': 15}, 'CO': {'count': 33}, 'supr-con': {'count': 21}, 'chem-ph': {'count': 3}, 'DM': {'count': 2}, 'MN': {'count': 12}, 'q-fin': {'count': 27}, 'PM': {'count': 2}, 'AP': {'count': 27}, 'gr-qc': {'count': 15}, 'quant-gas': {'count': 8}, 'mes-hall': {'count': 33}, 'IT': {'count': 19}, 'SI': {'count': 6}, 'SG': {'count': 3}, 'bio-ph': {'count': 2}, 'SR': {'count': 16}, 'soc-ph': {'count': 5}, 'hep-ex': {'count': 15}, 'DG': {'count': 11}, 'NE': {'count': 5}, 'CR': {'count': 6}, 'CL': {'count': 12}, 'RM': {'count': 3}, 'econ': {'count': 17}, 'nlin': {'count': 5}, 'PS': {'count': 1}, 'LG': {'count': 26}, 'QA': {'count': 9}, 'str-el': {'count': 26}, 'CV': {'count': 34}, 'MF': {'count': 6}, 'IM': {'count': 7}, 'EM': {'count': 6}, 'TH': {'count': 5}, 'PR': {'count': 20}, 'AT': {'count': 4}, 'OA': {'count': 4}, 'CP': {'count': 6}, 'LO': {'count': 14}, 'flu-dyn': {'count': 6}, 'atom-ph': {'count': 8}, 'class-ph': {'count': 1}, 'SY': {'count': 20}, 'IR': {'count': 1}, 'plasm-ph': {'count': 8}, 'CE': {'count': 2}, 'AO': {'count': 1}, 'comp-ph': {'count': 3}, 'optics': {'count': 12}, 'MG': {'count': 4}, 'ST': {'count': 6}, 'nucl-ex': {'count': 6}, 'CY': {'count': 9}, 'ao-ph': {'count': 2}, 'DB': {'count': 1}, 'math-ph': {'count': 10}, 'NC': {'count': 13}, 'GT': {'count': 11}, 'TO': {'count': 2}, 'AI': {'count': 9}, 'NI': {'count': 2}, 'gen-ph': {'count': 4}, 'OT': {'count': 4}, 'SD': {'count': 2}, 'dis-nn': {'count': 4}, 'RO': {'count': 7}, 'CA': {'count': 6}, 'FL': {'count': 1}, 'SE': {'count': 5}, 'EP': {'count': 9}, 'hist-ph': {'count': 1}, 'QM': {'count': 9}, 'ed-ph': {'count': 2}, 'GR': {'count': 4}, 'MS': {'count': 1}, 'CD': {'count': 1}, 'ET': {'count': 1}, 'acc-ph': {'count': 5}, 'AC': {'count': 2}, 'OH': {'count': 1}, 'EC': {'count': 2}, 'DL': {'count': 1}, 'AS': {'count': 3}, 'geo-ph': {'count': 2}, 'CG': {'count': 3}, 'CB': {'count': 1}, 'AR': {'count': 1}, 'TR': {'count': 1}, 'atm-clus': {'count': 1}}}} | -| [ArXivHierarchicalClusteringS2S](https://www.kaggle.com/Cornell-University/arxiv) | ['eng'] | Clustering | p2p | [Academic, Written] | {'test': 2048} | {'test': 1009.98} | -| [ArguAna](http://argumentation.bplaced.net/arguana/data) (Boteva et al., 2016) | ['eng'] | Retrieval | s2p | [Medical, Written] | None | {'test': {'average_document_length': 1029.2327645838136, 'average_query_length': 1192.7204836415362, 'num_documents': 8674, 'num_queries': 1406, 'average_relevant_docs_per_query': 1.0}} | -| [ArguAna-PL](https://huggingface.co/datasets/clarin-knext/arguana-pl) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | | None | {'test': {'average_document_length': 1060.702674659903, 'average_query_length': 1224.8022759601706, 'num_documents': 8674, 'num_queries': 1406, 'average_relevant_docs_per_query': 1.0}} | -| [ArmenianParaphrasePC](https://github.com/ivannikov-lab/arpa-paraphrase-corpus) (Arthur Malajyan, 2020) | ['hye'] | PairClassification | s2s | [News, Written] | {'train': 4023, 'test': 1470} | {'train': 243.81, 'test': 241.37} | -| [ArxivClassification](https://ieeexplore.ieee.org/document/8675939) (He et al., 2019) | ['eng'] | Classification | s2s | [Academic, Written] | {'test': 2048} | {} | -| [AskUbuntuDupQuestions](https://github.com/taolei87/askubuntu) | ['eng'] | Reranking | s2s | | {'test': 2255} | {'test': {'num_samples': 375, 'num_positive': 375, 'num_negative': 375, 'avg_query_len': 50.205333333333336, 'avg_positive_len': 6.013333333333334, 'avg_negative_len': 13.986666666666666}} | -| [Assin2RTE](https://link.springer.com/chapter/10.1007/978-3-030-41505-1_39) (Real et al., 2020) | ['por'] | PairClassification | s2s | [Written] | {'test': 2448} | {'test': 53.55} | -| [Assin2STS](https://link.springer.com/chapter/10.1007/978-3-030-41505-1_39) (Real et al., 2020) | ['por'] | STS | s2s | [Written] | {'test': 2448} | {'test': 53.55} | +| [AfriSentiClassification](https://arxiv.org/abs/2302.08956) | ['amh', 'arq', 'ary', 'hau', 'ibo', 'kin', 'pcm', 'por', 'swa', 'tso', 'twi', 'yor'] | Classification | s2s | [Social, Written] | None | None | +| [AfriSentiLangClassification](https://huggingface.co/datasets/HausaNLP/afrisenti-lid-data/) | ['amh', 'arq', 'ary', 'hau', 'ibo', 'kin', 'pcm', 'por', 'swa', 'tso', 'twi', 'yor'] | Classification | s2s | [Social, Written] | None | None | +| [AllegroReviews](https://aclanthology.org/2020.acl-main.111.pdf) | ['pol'] | Classification | s2s | | None | None | +| [AlloProfClusteringP2P.v2](https://huggingface.co/datasets/lyon-nlp/alloprof) (Lefebvre-Brossard et al., 2023) | ['fra'] | Clustering | p2p | [Encyclopaedic, Written] | None | None | +| [AlloProfClusteringS2S.v2](https://huggingface.co/datasets/lyon-nlp/alloprof) (Lefebvre-Brossard et al., 2023) | ['fra'] | Clustering | s2s | [Encyclopaedic, Written] | None | None | +| [AlloprofReranking](https://huggingface.co/datasets/antoinelb7/alloprof) (Lefebvre-Brossard et al., 2023) | ['fra'] | Reranking | s2p | [Web, Academic, Written] | None | None | +| [AlloprofRetrieval](https://huggingface.co/datasets/antoinelb7/alloprof) (Lefebvre-Brossard et al., 2023) | ['fra'] | Retrieval | s2p | [Encyclopaedic, Written] | None | None | +| [AlphaNLI](https://leaderboard.allenai.org/anli/submissions/get-started) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | None | None | +| [AmazonCounterfactualClassification](https://arxiv.org/abs/2104.06893) | ['deu', 'eng', 'jpn'] | Classification | s2s | [Reviews, Written] | None | None | +| [AmazonPolarityClassification](https://huggingface.co/datasets/amazon_polarity) (Julian McAuley, 2013) | ['eng'] | Classification | p2p | [Reviews, Written] | None | None | +| [AmazonReviewsClassification](https://arxiv.org/abs/2010.02573) (Phillip Keung, 2020) | ['cmn', 'deu', 'eng', 'fra', 'jpn', 'spa'] | Classification | s2s | [Reviews, Written] | None | None | +| [AngryTweetsClassification](https://aclanthology.org/2021.nodalida-main.53/) (Pauli et al., 2021) | ['dan'] | Classification | s2s | [Social, Written] | None | None | +| [AppsRetrieval](https://arxiv.org/abs/2105.09938) (Dan Hendrycks, 2021) | ['eng', 'python'] | Retrieval | p2p | [Programming, Written] | {'test': 12530} | {'test': {'number_of_characters': 2245.84, 'num_samples': 12530, 'num_queries': 3765, 'num_documents': 8765, 'average_document_length': 0.07, 'average_query_length': 0.44, 'average_relevant_docs_per_query': 1.0}} | +| [ArEntail](https://link.springer.com/article/10.1007/s10579-024-09731-1) (Obeidat et al., 2024) | ['ara'] | PairClassification | s2s | [News, Written] | None | None | +| [ArXivHierarchicalClusteringP2P](https://www.kaggle.com/Cornell-University/arxiv) | ['eng'] | Clustering | p2p | [Academic, Written] | {'test': 2048} | {'test': {'num_samples': 2048, 'number_of_characters': 2065284, 'average_text_length': 1008.44, 'average_labels_per_text': 1.46, 'unique_labels': 129, 'labels': {'cs': {'count': 356}, 'math': {'count': 381}, 'OC': {'count': 11}, 'hep-lat': {'count': 13}, 'hep': {'count': 98}, 'astro-ph': {'count': 213}, 'eess': {'count': 76}, 'quant-ph': {'count': 135}, 'DC': {'count': 5}, 'cond-mat': {'count': 274}, 'hep-th': {'count': 66}, 'SP': {'count': 33}, 'hep-ph': {'count': 69}, 'FA': {'count': 6}, 'nucl-th': {'count': 17}, 'q-bio': {'count': 80}, 'HE': {'count': 22}, 'HC': {'count': 2}, 'stat': {'count': 60}, 'ML': {'count': 16}, 'IV': {'count': 13}, 'stat-mech': {'count': 47}, 'DS': {'count': 14}, 'ME': {'count': 12}, 'CC': {'count': 2}, 'mtrl-sci': {'count': 22}, 'PE': {'count': 16}, 'NT': {'count': 11}, 'SC': {'count': 6}, 'AG': {'count': 13}, 'physics': {'count': 81}, 'ins-det': {'count': 9}, 'GA': {'count': 18}, 'BM': {'count': 6}, 'GN': {'count': 17}, 'NA': {'count': 15}, 'app-ph': {'count': 7}, 'RT': {'count': 6}, 'other': {'count': 37}, 'soft': {'count': 15}, 'CO': {'count': 33}, 'supr-con': {'count': 21}, 'chem-ph': {'count': 3}, 'DM': {'count': 2}, 'MN': {'count': 12}, 'q-fin': {'count': 27}, 'PM': {'count': 2}, 'AP': {'count': 27}, 'gr-qc': {'count': 15}, 'quant-gas': {'count': 8}, 'mes-hall': {'count': 33}, 'IT': {'count': 19}, 'SI': {'count': 6}, 'SG': {'count': 3}, 'bio-ph': {'count': 2}, 'SR': {'count': 16}, 'soc-ph': {'count': 5}, 'hep-ex': {'count': 15}, 'DG': {'count': 11}, 'NE': {'count': 5}, 'CR': {'count': 6}, 'CL': {'count': 12}, 'RM': {'count': 3}, 'econ': {'count': 17}, 'nlin': {'count': 5}, 'PS': {'count': 1}, 'LG': {'count': 26}, 'QA': {'count': 9}, 'str-el': {'count': 26}, 'CV': {'count': 34}, 'MF': {'count': 6}, 'IM': {'count': 7}, 'EM': {'count': 6}, 'TH': {'count': 5}, 'PR': {'count': 20}, 'AT': {'count': 4}, 'OA': {'count': 4}, 'CP': {'count': 6}, 'LO': {'count': 14}, 'flu-dyn': {'count': 6}, 'atom-ph': {'count': 8}, 'class-ph': {'count': 1}, 'SY': {'count': 20}, 'IR': {'count': 1}, 'plasm-ph': {'count': 8}, 'CE': {'count': 2}, 'AO': {'count': 1}, 'comp-ph': {'count': 3}, 'optics': {'count': 12}, 'MG': {'count': 4}, 'ST': {'count': 6}, 'nucl-ex': {'count': 6}, 'CY': {'count': 9}, 'ao-ph': {'count': 2}, 'DB': {'count': 1}, 'math-ph': {'count': 10}, 'NC': {'count': 13}, 'GT': {'count': 11}, 'TO': {'count': 2}, 'AI': {'count': 9}, 'NI': {'count': 2}, 'gen-ph': {'count': 4}, 'OT': {'count': 4}, 'SD': {'count': 2}, 'dis-nn': {'count': 4}, 'RO': {'count': 7}, 'CA': {'count': 6}, 'FL': {'count': 1}, 'SE': {'count': 5}, 'EP': {'count': 9}, 'hist-ph': {'count': 1}, 'QM': {'count': 9}, 'ed-ph': {'count': 2}, 'GR': {'count': 4}, 'MS': {'count': 1}, 'CD': {'count': 1}, 'ET': {'count': 1}, 'acc-ph': {'count': 5}, 'AC': {'count': 2}, 'OH': {'count': 1}, 'EC': {'count': 2}, 'DL': {'count': 1}, 'AS': {'count': 3}, 'geo-ph': {'count': 2}, 'CG': {'count': 3}, 'CB': {'count': 1}, 'AR': {'count': 1}, 'TR': {'count': 1}, 'atm-clus': {'count': 1}}}} | +| [ArXivHierarchicalClusteringS2S](https://www.kaggle.com/Cornell-University/arxiv) | ['eng'] | Clustering | p2p | [Academic, Written] | None | None | +| [ArguAna](http://argumentation.bplaced.net/arguana/data) (Boteva et al., 2016) | ['eng'] | Retrieval | s2p | [Medical, Written] | None | None | +| [ArguAna-PL](https://huggingface.co/datasets/clarin-knext/arguana-pl) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | | None | None | +| [ArmenianParaphrasePC](https://github.com/ivannikov-lab/arpa-paraphrase-corpus) (Arthur Malajyan, 2020) | ['hye'] | PairClassification | s2s | [News, Written] | None | None | +| [ArxivClassification](https://ieeexplore.ieee.org/document/8675939) (He et al., 2019) | ['eng'] | Classification | s2s | [Academic, Written] | None | None | +| [AskUbuntuDupQuestions](https://github.com/taolei87/askubuntu) | ['eng'] | Reranking | s2s | | {'test': 375} | {'test': {'num_samples': 375, 'number_of_characters': 413674, 'num_positive': 2255, 'num_negative': 5245, 'avg_query_len': 50.21, 'avg_positive_len': 52.54, 'avg_negative_len': 52.69}} | +| [Assin2RTE](https://link.springer.com/chapter/10.1007/978-3-030-41505-1_39) (Real et al., 2020) | ['por'] | PairClassification | s2s | [Written] | None | None | +| [Assin2STS](https://link.springer.com/chapter/10.1007/978-3-030-41505-1_39) (Real et al., 2020) | ['por'] | STS | s2s | [Written] | None | None | | [BIOSSES](https://tabilab.cmpe.boun.edu.tr/BIOSSES/DataSet.html) (Soğancıoğlu et al., 2017) | ['eng'] | STS | s2s | | None | None | | [BQ](https://aclanthology.org/2021.emnlp-main.357) (Shitao Xiao, 2024) | ['cmn'] | STS | s2s | | None | None | -| [BSARDRetrieval](https://huggingface.co/datasets/maastrichtlawtech/bsard) (Louis et al., 2022) | ['fra'] | Retrieval | s2p | [Legal, Spoken] | {'test': 222} | {'test': {'average_document_length': 880.2900631820793, 'average_query_length': 144.77027027027026, 'num_documents': 22633, 'num_queries': 222, 'average_relevant_docs_per_query': 1.0}} | -| [BUCC.v2](https://comparable.limsi.fr/bucc2018/bucc2018-task.html) | ['cmn', 'deu', 'eng', 'fra', 'rus'] | BitextMining | s2s | [Written] | {'test': 641684} | {'test': 101.3} | -| [Banking77Classification](https://arxiv.org/abs/2003.04807) | ['eng'] | Classification | s2s | [Written] | {'test': 3080} | {'test': 54.2} | -| [BelebeleRetrieval](https://arxiv.org/abs/2308.16884) (Lucas Bandarkar, 2023) | ['acm', 'afr', 'als', 'amh', 'apc', 'arb', 'ars', 'ary', 'arz', 'asm', 'azj', 'bam', 'ben', 'bod', 'bul', 'cat', 'ceb', 'ces', 'ckb', 'dan', 'deu', 'ell', 'eng', 'est', 'eus', 'fin', 'fra', 'fuv', 'gaz', 'grn', 'guj', 'hat', 'hau', 'heb', 'hin', 'hrv', 'hun', 'hye', 'ibo', 'ilo', 'ind', 'isl', 'ita', 'jav', 'jpn', 'kac', 'kan', 'kat', 'kaz', 'kea', 'khk', 'khm', 'kin', 'kir', 'kor', 'lao', 'lin', 'lit', 'lug', 'luo', 'lvs', 'mal', 'mar', 'mkd', 'mlt', 'mri', 'mya', 'nld', 'nob', 'npi', 'nso', 'nya', 'ory', 'pan', 'pbt', 'pes', 'plt', 'pol', 'por', 'ron', 'rus', 'shn', 'sin', 'slk', 'slv', 'sna', 'snd', 'som', 'sot', 'spa', 'srp', 'ssw', 'sun', 'swe', 'swh', 'tam', 'tel', 'tgk', 'tgl', 'tha', 'tir', 'tsn', 'tso', 'tur', 'ukr', 'urd', 'uzn', 'vie', 'war', 'wol', 'xho', 'yor', 'zho', 'zsm', 'zul'] | Retrieval | s2p | [Web, News, Written] | {'test': 103500} | {'test': {'average_document_length': 487.3975028339728, 'average_query_length': 74.49551684802204, 'num_documents': 183488, 'num_queries': 338378, 'average_relevant_docs_per_query': 1.0, 'hf_subset_descriptive_stats': {'acm_Arab-acm_Arab': {'average_document_length': 416.4733606557377, 'average_query_length': 55.84, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'acm_Arab-eng_Latn': {'average_document_length': 416.4733606557377, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-acm_Arab': {'average_document_length': 475.51024590163934, 'average_query_length': 55.84, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'afr_Latn-afr_Latn': {'average_document_length': 503.6659836065574, 'average_query_length': 78.04555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'afr_Latn-eng_Latn': {'average_document_length': 503.6659836065574, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-afr_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 78.04555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'als_Latn-als_Latn': {'average_document_length': 534.016393442623, 'average_query_length': 76.13555555555556, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'als_Latn-eng_Latn': {'average_document_length': 534.016393442623, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-als_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 76.13555555555556, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'amh_Ethi-amh_Ethi': {'average_document_length': 319.8688524590164, 'average_query_length': 49.16111111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'amh_Ethi-eng_Latn': {'average_document_length': 319.8688524590164, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-amh_Ethi': {'average_document_length': 475.51024590163934, 'average_query_length': 49.16111111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'apc_Arab-apc_Arab': {'average_document_length': 393.0553278688525, 'average_query_length': 55.85777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'apc_Arab-eng_Latn': {'average_document_length': 393.0553278688525, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-apc_Arab': {'average_document_length': 475.51024590163934, 'average_query_length': 55.85777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'arb_Arab-arb_Arab': {'average_document_length': 421.96311475409834, 'average_query_length': 58.55, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'arb_Arab-eng_Latn': {'average_document_length': 421.96311475409834, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-arb_Arab': {'average_document_length': 475.51024590163934, 'average_query_length': 58.55, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'arb_Latn-arb_Latn': {'average_document_length': 555.6188524590164, 'average_query_length': 67.02444444444444, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'arb_Latn-eng_Latn': {'average_document_length': 555.6188524590164, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-arb_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 67.02444444444444, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ars_Arab-ars_Arab': {'average_document_length': 422.5553278688525, 'average_query_length': 56.43222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ars_Arab-eng_Latn': {'average_document_length': 422.5553278688525, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ars_Arab': {'average_document_length': 475.51024590163934, 'average_query_length': 56.43222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ary_Arab-ary_Arab': {'average_document_length': 411.1475409836066, 'average_query_length': 66.01893095768374, 'num_documents': 488, 'num_queries': 898, 'average_relevant_docs_per_query': 1.0}, 'ary_Arab-eng_Latn': {'average_document_length': 411.1475409836066, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ary_Arab': {'average_document_length': 475.51024590163934, 'average_query_length': 66.01893095768374, 'num_documents': 488, 'num_queries': 898, 'average_relevant_docs_per_query': 1.0}, 'arz_Arab-arz_Arab': {'average_document_length': 412.05122950819674, 'average_query_length': 57.14111111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'arz_Arab-eng_Latn': {'average_document_length': 412.05122950819674, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-arz_Arab': {'average_document_length': 475.51024590163934, 'average_query_length': 57.14111111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'asm_Beng-asm_Beng': {'average_document_length': 458.5983606557377, 'average_query_length': 68.26, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'asm_Beng-eng_Latn': {'average_document_length': 458.5983606557377, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-asm_Beng': {'average_document_length': 475.51024590163934, 'average_query_length': 68.26, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'azj_Latn-azj_Latn': {'average_document_length': 519.6127049180328, 'average_query_length': 73.51222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'azj_Latn-eng_Latn': {'average_document_length': 519.6127049180328, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-azj_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 73.51222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'bam_Latn-bam_Latn': {'average_document_length': 457.3114754098361, 'average_query_length': 72.34222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'bam_Latn-eng_Latn': {'average_document_length': 457.3114754098361, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-bam_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 72.34222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ben_Beng-ben_Beng': {'average_document_length': 467.7745901639344, 'average_query_length': 69.48444444444445, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ben_Beng-eng_Latn': {'average_document_length': 467.7745901639344, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ben_Beng': {'average_document_length': 475.51024590163934, 'average_query_length': 69.48444444444445, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ben_Latn-ben_Latn': {'average_document_length': 522.8934426229508, 'average_query_length': 74.78777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ben_Latn-eng_Latn': {'average_document_length': 522.8934426229508, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ben_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 74.78777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'bod_Tibt-bod_Tibt': {'average_document_length': 533.3872950819672, 'average_query_length': 86.90222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'bod_Tibt-eng_Latn': {'average_document_length': 533.3872950819672, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-bod_Tibt': {'average_document_length': 475.51024590163934, 'average_query_length': 86.90222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'bul_Cyrl-bul_Cyrl': {'average_document_length': 496.97131147540983, 'average_query_length': 72.89, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'bul_Cyrl-eng_Latn': {'average_document_length': 496.97131147540983, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-bul_Cyrl': {'average_document_length': 475.51024590163934, 'average_query_length': 72.89, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'cat_Latn-cat_Latn': {'average_document_length': 525.4467213114754, 'average_query_length': 75.40666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'cat_Latn-eng_Latn': {'average_document_length': 525.4467213114754, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-cat_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 75.40666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ceb_Latn-ceb_Latn': {'average_document_length': 570.8483606557377, 'average_query_length': 81.19666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ceb_Latn-eng_Latn': {'average_document_length': 570.8483606557377, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ceb_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 81.19666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ces_Latn-ces_Latn': {'average_document_length': 461.0061475409836, 'average_query_length': 67.73333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ces_Latn-eng_Latn': {'average_document_length': 461.0061475409836, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ces_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 67.73333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ckb_Arab-ckb_Arab': {'average_document_length': 462.98770491803276, 'average_query_length': 71.04555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ckb_Arab-eng_Latn': {'average_document_length': 462.98770491803276, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ckb_Arab': {'average_document_length': 475.51024590163934, 'average_query_length': 71.04555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'dan_Latn-dan_Latn': {'average_document_length': 489.4856557377049, 'average_query_length': 72.96888888888888, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'dan_Latn-eng_Latn': {'average_document_length': 489.4856557377049, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-dan_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 72.96888888888888, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'deu_Latn-deu_Latn': {'average_document_length': 555.1659836065573, 'average_query_length': 75.32444444444444, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'deu_Latn-eng_Latn': {'average_document_length': 555.1659836065573, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-deu_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 75.32444444444444, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ell_Grek-ell_Grek': {'average_document_length': 568.3872950819672, 'average_query_length': 86.92666666666666, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ell_Grek-eng_Latn': {'average_document_length': 568.3872950819672, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ell_Grek': {'average_document_length': 475.51024590163934, 'average_query_length': 86.92666666666666, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-eng_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'est_Latn-est_Latn': {'average_document_length': 467.1475409836066, 'average_query_length': 67.55888888888889, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'est_Latn-eng_Latn': {'average_document_length': 467.1475409836066, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-est_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 67.55888888888889, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eus_Latn-eus_Latn': {'average_document_length': 506.19262295081967, 'average_query_length': 74.44777777777777, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eus_Latn-eng_Latn': {'average_document_length': 506.19262295081967, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-eus_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 74.44777777777777, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'fin_Latn-fin_Latn': {'average_document_length': 507.5, 'average_query_length': 72.50888888888889, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'fin_Latn-eng_Latn': {'average_document_length': 507.5, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-fin_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 72.50888888888889, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'fra_Latn-fra_Latn': {'average_document_length': 564.8401639344262, 'average_query_length': 90.54222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'fra_Latn-eng_Latn': {'average_document_length': 564.8401639344262, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-fra_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 90.54222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'fuv_Latn-fuv_Latn': {'average_document_length': 443.4733606557377, 'average_query_length': 58.42111111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'fuv_Latn-eng_Latn': {'average_document_length': 443.4733606557377, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-fuv_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 58.42111111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'gaz_Latn-gaz_Latn': {'average_document_length': 563.5389344262295, 'average_query_length': 85.93222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'gaz_Latn-eng_Latn': {'average_document_length': 563.5389344262295, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-gaz_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 85.93222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'grn_Latn-grn_Latn': {'average_document_length': 480.3299180327869, 'average_query_length': 75.10666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'grn_Latn-eng_Latn': {'average_document_length': 480.3299180327869, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-grn_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 75.10666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'guj_Gujr-guj_Gujr': {'average_document_length': 458.1885245901639, 'average_query_length': 62.25666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'guj_Gujr-eng_Latn': {'average_document_length': 458.1885245901639, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-guj_Gujr': {'average_document_length': 475.51024590163934, 'average_query_length': 62.25666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'hat_Latn-hat_Latn': {'average_document_length': 438.6700819672131, 'average_query_length': 70.64666666666666, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'hat_Latn-eng_Latn': {'average_document_length': 438.6700819672131, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-hat_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 70.64666666666666, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'hau_Latn-hau_Latn': {'average_document_length': 507.24590163934425, 'average_query_length': 85.8488888888889, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'hau_Latn-eng_Latn': {'average_document_length': 507.24590163934425, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-hau_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 85.8488888888889, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'heb_Hebr-heb_Hebr': {'average_document_length': 371.36270491803276, 'average_query_length': 55.135555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'heb_Hebr-eng_Latn': {'average_document_length': 371.36270491803276, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-heb_Hebr': {'average_document_length': 475.51024590163934, 'average_query_length': 55.135555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'hin_Deva-hin_Deva': {'average_document_length': 473.55737704918033, 'average_query_length': 72.61777777777777, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'hin_Deva-eng_Latn': {'average_document_length': 473.55737704918033, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-hin_Deva': {'average_document_length': 475.51024590163934, 'average_query_length': 72.61777777777777, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'hin_Latn-hin_Latn': {'average_document_length': 541.7315573770492, 'average_query_length': 74.81222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'hin_Latn-eng_Latn': {'average_document_length': 541.7315573770492, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-hin_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 74.81222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'hrv_Latn-hrv_Latn': {'average_document_length': 469.202868852459, 'average_query_length': 68.83555555555556, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'hrv_Latn-eng_Latn': {'average_document_length': 469.202868852459, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-hrv_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 68.83555555555556, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'hun_Latn-hun_Latn': {'average_document_length': 501.1946721311475, 'average_query_length': 74.40555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'hun_Latn-eng_Latn': {'average_document_length': 501.1946721311475, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-hun_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 74.40555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'hye_Armn-hye_Armn': {'average_document_length': 527.5102459016393, 'average_query_length': 75.42555555555556, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'hye_Armn-eng_Latn': {'average_document_length': 527.5102459016393, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-hye_Armn': {'average_document_length': 475.51024590163934, 'average_query_length': 75.42555555555556, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ibo_Latn-ibo_Latn': {'average_document_length': 482.3483606557377, 'average_query_length': 72.51501668520578, 'num_documents': 488, 'num_queries': 899, 'average_relevant_docs_per_query': 1.0011123470522802}, 'ibo_Latn-eng_Latn': {'average_document_length': 482.3483606557377, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ibo_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 72.51501668520578, 'num_documents': 488, 'num_queries': 899, 'average_relevant_docs_per_query': 1.0011123470522802}, 'ilo_Latn-ilo_Latn': {'average_document_length': 574.6987704918033, 'average_query_length': 85.7611111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ilo_Latn-eng_Latn': {'average_document_length': 574.6987704918033, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ilo_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 85.7611111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ind_Latn-ind_Latn': {'average_document_length': 516.0573770491803, 'average_query_length': 82.10555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ind_Latn-eng_Latn': {'average_document_length': 516.0573770491803, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ind_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 82.10555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'isl_Latn-isl_Latn': {'average_document_length': 470.73975409836066, 'average_query_length': 77.27333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'isl_Latn-eng_Latn': {'average_document_length': 470.73975409836066, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-isl_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 77.27333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ita_Latn-ita_Latn': {'average_document_length': 560.9344262295082, 'average_query_length': 83.49777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ita_Latn-eng_Latn': {'average_document_length': 560.9344262295082, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ita_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 83.49777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'jav_Latn-jav_Latn': {'average_document_length': 494.1803278688525, 'average_query_length': 78.60666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'jav_Latn-eng_Latn': {'average_document_length': 494.1803278688525, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-jav_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 78.60666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'jpn_Jpan-jpn_Jpan': {'average_document_length': 207.74795081967213, 'average_query_length': 35.79, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'jpn_Jpan-eng_Latn': {'average_document_length': 207.74795081967213, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-jpn_Jpan': {'average_document_length': 475.51024590163934, 'average_query_length': 35.79, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'kac_Latn-kac_Latn': {'average_document_length': 605.2889344262295, 'average_query_length': 98.64182424916574, 'num_documents': 488, 'num_queries': 899, 'average_relevant_docs_per_query': 1.0}, 'kac_Latn-eng_Latn': {'average_document_length': 605.2889344262295, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-kac_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 98.64182424916574, 'num_documents': 488, 'num_queries': 899, 'average_relevant_docs_per_query': 1.0}, 'kan_Knda-kan_Knda': {'average_document_length': 498.9077868852459, 'average_query_length': 72.13666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'kan_Knda-eng_Latn': {'average_document_length': 498.9077868852459, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-kan_Knda': {'average_document_length': 475.51024590163934, 'average_query_length': 72.13666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'kat_Geor-kat_Geor': {'average_document_length': 521.7766393442623, 'average_query_length': 74.81444444444445, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'kat_Geor-eng_Latn': {'average_document_length': 521.7766393442623, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-kat_Geor': {'average_document_length': 475.51024590163934, 'average_query_length': 74.81444444444445, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'kaz_Cyrl-kaz_Cyrl': {'average_document_length': 488.2110655737705, 'average_query_length': 70.75666666666666, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'kaz_Cyrl-eng_Latn': {'average_document_length': 488.2110655737705, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-kaz_Cyrl': {'average_document_length': 475.51024590163934, 'average_query_length': 70.75666666666666, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'kea_Latn-kea_Latn': {'average_document_length': 471.5594262295082, 'average_query_length': 75.94111111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'kea_Latn-eng_Latn': {'average_document_length': 471.5594262295082, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-kea_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 75.94111111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'khk_Cyrl-khk_Cyrl': {'average_document_length': 496.655737704918, 'average_query_length': 73.33444444444444, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'khk_Cyrl-eng_Latn': {'average_document_length': 496.655737704918, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-khk_Cyrl': {'average_document_length': 475.51024590163934, 'average_query_length': 73.33444444444444, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'khm_Khmr-khm_Khmr': {'average_document_length': 562.4139344262295, 'average_query_length': 75.74888888888889, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'khm_Khmr-eng_Latn': {'average_document_length': 562.4139344262295, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-khm_Khmr': {'average_document_length': 475.51024590163934, 'average_query_length': 75.74888888888889, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'kin_Latn-kin_Latn': {'average_document_length': 529.2520491803278, 'average_query_length': 79.89655172413794, 'num_documents': 488, 'num_queries': 899, 'average_relevant_docs_per_query': 1.0011123470522802}, 'kin_Latn-eng_Latn': {'average_document_length': 529.2520491803278, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-kin_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 79.89655172413794, 'num_documents': 488, 'num_queries': 899, 'average_relevant_docs_per_query': 1.0011123470522802}, 'kir_Cyrl-kir_Cyrl': {'average_document_length': 487.80737704918033, 'average_query_length': 74.42333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'kir_Cyrl-eng_Latn': {'average_document_length': 487.80737704918033, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-kir_Cyrl': {'average_document_length': 475.51024590163934, 'average_query_length': 74.42333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'kor_Hang-kor_Hang': {'average_document_length': 241.32991803278688, 'average_query_length': 35.257777777777775, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'kor_Hang-eng_Latn': {'average_document_length': 241.32991803278688, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-kor_Hang': {'average_document_length': 475.51024590163934, 'average_query_length': 35.257777777777775, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'lao_Laoo-lao_Laoo': {'average_document_length': 471.6495901639344, 'average_query_length': 63.31333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'lao_Laoo-eng_Latn': {'average_document_length': 471.6495901639344, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-lao_Laoo': {'average_document_length': 475.51024590163934, 'average_query_length': 63.31333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'lin_Latn-lin_Latn': {'average_document_length': 512.9016393442623, 'average_query_length': 81.56681514476615, 'num_documents': 488, 'num_queries': 898, 'average_relevant_docs_per_query': 1.0022271714922049}, 'lin_Latn-eng_Latn': {'average_document_length': 512.9016393442623, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-lin_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 81.56681514476615, 'num_documents': 488, 'num_queries': 898, 'average_relevant_docs_per_query': 1.0022271714922049}, 'lit_Latn-lit_Latn': {'average_document_length': 474.0553278688525, 'average_query_length': 68.69888888888889, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'lit_Latn-eng_Latn': {'average_document_length': 474.0553278688525, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-lit_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 68.69888888888889, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'lug_Latn-lug_Latn': {'average_document_length': 485.73975409836066, 'average_query_length': 78.52057842046719, 'num_documents': 488, 'num_queries': 899, 'average_relevant_docs_per_query': 1.0011123470522802}, 'lug_Latn-eng_Latn': {'average_document_length': 485.73975409836066, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-lug_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 78.52057842046719, 'num_documents': 488, 'num_queries': 899, 'average_relevant_docs_per_query': 1.0011123470522802}, 'luo_Latn-luo_Latn': {'average_document_length': 497.53688524590166, 'average_query_length': 73.14333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'luo_Latn-eng_Latn': {'average_document_length': 497.53688524590166, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-luo_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 73.14333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'lvs_Latn-lvs_Latn': {'average_document_length': 487.21311475409834, 'average_query_length': 69.97888888888889, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'lvs_Latn-eng_Latn': {'average_document_length': 487.21311475409834, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-lvs_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 69.97888888888889, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'mal_Mlym-mal_Mlym': {'average_document_length': 539.2827868852459, 'average_query_length': 80.69222222222223, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'mal_Mlym-eng_Latn': {'average_document_length': 539.2827868852459, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-mal_Mlym': {'average_document_length': 475.51024590163934, 'average_query_length': 80.69222222222223, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'mar_Deva-mar_Deva': {'average_document_length': 478.67418032786884, 'average_query_length': 68.62625139043382, 'num_documents': 488, 'num_queries': 899, 'average_relevant_docs_per_query': 1.0011123470522802}, 'mar_Deva-eng_Latn': {'average_document_length': 478.67418032786884, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-mar_Deva': {'average_document_length': 475.51024590163934, 'average_query_length': 68.62625139043382, 'num_documents': 488, 'num_queries': 899, 'average_relevant_docs_per_query': 1.0011123470522802}, 'mkd_Cyrl-mkd_Cyrl': {'average_document_length': 495.77868852459017, 'average_query_length': 74.01333333333334, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'mkd_Cyrl-eng_Latn': {'average_document_length': 495.77868852459017, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-mkd_Cyrl': {'average_document_length': 475.51024590163934, 'average_query_length': 74.01333333333334, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'mlt_Latn-mlt_Latn': {'average_document_length': 525.8995901639345, 'average_query_length': 75.00444444444445, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'mlt_Latn-eng_Latn': {'average_document_length': 525.8995901639345, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-mlt_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 75.00444444444445, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'mri_Latn-mri_Latn': {'average_document_length': 526.0860655737705, 'average_query_length': 81.71444444444444, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'mri_Latn-eng_Latn': {'average_document_length': 526.0860655737705, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-mri_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 81.71444444444444, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'mya_Mymr-mya_Mymr': {'average_document_length': 590.389344262295, 'average_query_length': 89.28333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'mya_Mymr-eng_Latn': {'average_document_length': 590.389344262295, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-mya_Mymr': {'average_document_length': 475.51024590163934, 'average_query_length': 89.28333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'nld_Latn-nld_Latn': {'average_document_length': 529.1434426229508, 'average_query_length': 75.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'nld_Latn-eng_Latn': {'average_document_length': 529.1434426229508, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-nld_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 75.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'nob_Latn-nob_Latn': {'average_document_length': 479.13729508196724, 'average_query_length': 71.04555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'nob_Latn-eng_Latn': {'average_document_length': 479.13729508196724, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-nob_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 71.04555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'npi_Deva-npi_Deva': {'average_document_length': 456.9590163934426, 'average_query_length': 66.89666666666666, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'npi_Deva-eng_Latn': {'average_document_length': 456.9590163934426, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-npi_Deva': {'average_document_length': 475.51024590163934, 'average_query_length': 66.89666666666666, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'npi_Latn-npi_Latn': {'average_document_length': 515.9815573770492, 'average_query_length': 71.89666666666666, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'npi_Latn-eng_Latn': {'average_document_length': 515.9815573770492, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-npi_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 71.89666666666666, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'nso_Latn-nso_Latn': {'average_document_length': 548.0225409836065, 'average_query_length': 86.77444444444444, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'nso_Latn-eng_Latn': {'average_document_length': 548.0225409836065, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-nso_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 86.77444444444444, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'nya_Latn-nya_Latn': {'average_document_length': 532.3934426229508, 'average_query_length': 90.78777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'nya_Latn-eng_Latn': {'average_document_length': 532.3934426229508, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-nya_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 90.78777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ory_Orya-ory_Orya': {'average_document_length': 487.78483606557376, 'average_query_length': 72.95777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ory_Orya-eng_Latn': {'average_document_length': 487.78483606557376, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ory_Orya': {'average_document_length': 475.51024590163934, 'average_query_length': 72.95777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'pan_Guru-pan_Guru': {'average_document_length': 480.2438524590164, 'average_query_length': 73.29777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'pan_Guru-eng_Latn': {'average_document_length': 480.2438524590164, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-pan_Guru': {'average_document_length': 475.51024590163934, 'average_query_length': 73.29777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'pbt_Arab-pbt_Arab': {'average_document_length': 453.3299180327869, 'average_query_length': 67.67111111111112, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'pbt_Arab-eng_Latn': {'average_document_length': 453.3299180327869, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-pbt_Arab': {'average_document_length': 475.51024590163934, 'average_query_length': 67.67111111111112, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'pes_Arab-pes_Arab': {'average_document_length': 448.84631147540983, 'average_query_length': 64.75111111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'pes_Arab-eng_Latn': {'average_document_length': 448.84631147540983, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-pes_Arab': {'average_document_length': 475.51024590163934, 'average_query_length': 64.75111111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'plt_Latn-plt_Latn': {'average_document_length': 581.2745901639345, 'average_query_length': 94.99555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'plt_Latn-eng_Latn': {'average_document_length': 581.2745901639345, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-plt_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 94.99555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'pol_Latn-pol_Latn': {'average_document_length': 504.0409836065574, 'average_query_length': 74.09777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'pol_Latn-eng_Latn': {'average_document_length': 504.0409836065574, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-pol_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 74.09777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'por_Latn-por_Latn': {'average_document_length': 517.2827868852459, 'average_query_length': 78.11666666666666, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'por_Latn-eng_Latn': {'average_document_length': 517.2827868852459, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-por_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 78.11666666666666, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ron_Latn-ron_Latn': {'average_document_length': 534.8668032786885, 'average_query_length': 78.74222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ron_Latn-eng_Latn': {'average_document_length': 534.8668032786885, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ron_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 78.74222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'rus_Cyrl-rus_Cyrl': {'average_document_length': 520.1700819672132, 'average_query_length': 83.16333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'rus_Cyrl-eng_Latn': {'average_document_length': 520.1700819672132, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-rus_Cyrl': {'average_document_length': 475.51024590163934, 'average_query_length': 83.16333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'shn_Mymr-shn_Mymr': {'average_document_length': 676.172131147541, 'average_query_length': 75.90222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'shn_Mymr-eng_Latn': {'average_document_length': 676.172131147541, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-shn_Mymr': {'average_document_length': 475.51024590163934, 'average_query_length': 75.90222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'sin_Latn-sin_Latn': {'average_document_length': 590.7889344262295, 'average_query_length': 94.46666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'sin_Latn-eng_Latn': {'average_document_length': 590.7889344262295, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-sin_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 94.46666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'sin_Sinh-sin_Sinh': {'average_document_length': 478.66803278688525, 'average_query_length': 69.91777777777777, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'sin_Sinh-eng_Latn': {'average_document_length': 478.66803278688525, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-sin_Sinh': {'average_document_length': 475.51024590163934, 'average_query_length': 69.91777777777777, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'slk_Latn-slk_Latn': {'average_document_length': 476.7766393442623, 'average_query_length': 68.5411111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'slk_Latn-eng_Latn': {'average_document_length': 476.7766393442623, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-slk_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 68.5411111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'slv_Latn-slv_Latn': {'average_document_length': 474.84631147540983, 'average_query_length': 68.79888888888888, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'slv_Latn-eng_Latn': {'average_document_length': 474.84631147540983, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-slv_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 68.79888888888888, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'sna_Latn-sna_Latn': {'average_document_length': 532.5860655737705, 'average_query_length': 81.30700778642937, 'num_documents': 488, 'num_queries': 899, 'average_relevant_docs_per_query': 1.0}, 'sna_Latn-eng_Latn': {'average_document_length': 532.5860655737705, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-sna_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 81.30700778642937, 'num_documents': 488, 'num_queries': 899, 'average_relevant_docs_per_query': 1.0}, 'snd_Arab-snd_Arab': {'average_document_length': 431.48770491803276, 'average_query_length': 63.42333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'snd_Arab-eng_Latn': {'average_document_length': 431.48770491803276, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-snd_Arab': {'average_document_length': 475.51024590163934, 'average_query_length': 63.42333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'som_Latn-som_Latn': {'average_document_length': 542.0737704918033, 'average_query_length': 90.95777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'som_Latn-eng_Latn': {'average_document_length': 542.0737704918033, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-som_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 90.95777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'sot_Latn-sot_Latn': {'average_document_length': 573.3258196721312, 'average_query_length': 83.13111111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'sot_Latn-eng_Latn': {'average_document_length': 573.3258196721312, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-sot_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 83.13111111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'spa_Latn-spa_Latn': {'average_document_length': 564.3319672131148, 'average_query_length': 82.16, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'spa_Latn-eng_Latn': {'average_document_length': 564.3319672131148, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-spa_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 82.16, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'srp_Cyrl-srp_Cyrl': {'average_document_length': 471.84631147540983, 'average_query_length': 67.49833147942158, 'num_documents': 488, 'num_queries': 899, 'average_relevant_docs_per_query': 1.0011123470522802}, 'srp_Cyrl-eng_Latn': {'average_document_length': 471.84631147540983, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-srp_Cyrl': {'average_document_length': 475.51024590163934, 'average_query_length': 67.49833147942158, 'num_documents': 488, 'num_queries': 899, 'average_relevant_docs_per_query': 1.0011123470522802}, 'ssw_Latn-ssw_Latn': {'average_document_length': 535.0901639344262, 'average_query_length': 81.09777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ssw_Latn-eng_Latn': {'average_document_length': 535.0901639344262, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ssw_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 81.09777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'sun_Latn-sun_Latn': {'average_document_length': 495.3032786885246, 'average_query_length': 78.16, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'sun_Latn-eng_Latn': {'average_document_length': 495.3032786885246, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-sun_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 78.16, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'swe_Latn-swe_Latn': {'average_document_length': 480.6803278688525, 'average_query_length': 68.67666666666666, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'swe_Latn-eng_Latn': {'average_document_length': 480.6803278688525, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-swe_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 68.67666666666666, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'swh_Latn-swh_Latn': {'average_document_length': 499.0983606557377, 'average_query_length': 80.56, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'swh_Latn-eng_Latn': {'average_document_length': 499.0983606557377, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-swh_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 80.56, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tam_Taml-tam_Taml': {'average_document_length': 555.5286885245902, 'average_query_length': 81.12777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tam_Taml-eng_Latn': {'average_document_length': 555.5286885245902, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tam_Taml': {'average_document_length': 475.51024590163934, 'average_query_length': 81.12777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tel_Telu-tel_Telu': {'average_document_length': 481.5245901639344, 'average_query_length': 72.18777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tel_Telu-eng_Latn': {'average_document_length': 481.5245901639344, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tel_Telu': {'average_document_length': 475.51024590163934, 'average_query_length': 72.18777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tgk_Cyrl-tgk_Cyrl': {'average_document_length': 528.516393442623, 'average_query_length': 74.28111111111112, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tgk_Cyrl-eng_Latn': {'average_document_length': 528.516393442623, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tgk_Cyrl': {'average_document_length': 475.51024590163934, 'average_query_length': 74.28111111111112, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tgl_Latn-tgl_Latn': {'average_document_length': 597.6270491803278, 'average_query_length': 82.34555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tgl_Latn-eng_Latn': {'average_document_length': 597.6270491803278, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tgl_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 82.34555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tha_Thai-tha_Thai': {'average_document_length': 456.1659836065574, 'average_query_length': 59.46666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tha_Thai-eng_Latn': {'average_document_length': 456.1659836065574, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tha_Thai': {'average_document_length': 475.51024590163934, 'average_query_length': 59.46666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tir_Ethi-tir_Ethi': {'average_document_length': 327.6967213114754, 'average_query_length': 51.99888888888889, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tir_Ethi-eng_Latn': {'average_document_length': 327.6967213114754, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tir_Ethi': {'average_document_length': 475.51024590163934, 'average_query_length': 51.99888888888889, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tsn_Latn-tsn_Latn': {'average_document_length': 591.7131147540983, 'average_query_length': 87.12777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tsn_Latn-eng_Latn': {'average_document_length': 591.7131147540983, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tsn_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 87.12777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tso_Latn-tso_Latn': {'average_document_length': 569.6475409836065, 'average_query_length': 91.69444444444444, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tso_Latn-eng_Latn': {'average_document_length': 569.6475409836065, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tso_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 91.69444444444444, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tur_Latn-tur_Latn': {'average_document_length': 489.0409836065574, 'average_query_length': 71.56222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'tur_Latn-eng_Latn': {'average_document_length': 489.0409836065574, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tur_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 71.56222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ukr_Cyrl-ukr_Cyrl': {'average_document_length': 488.11475409836066, 'average_query_length': 72.08222222222223, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ukr_Cyrl-eng_Latn': {'average_document_length': 488.11475409836066, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ukr_Cyrl': {'average_document_length': 475.51024590163934, 'average_query_length': 72.08222222222223, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'urd_Arab-urd_Arab': {'average_document_length': 470.452868852459, 'average_query_length': 70.52666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'urd_Arab-eng_Latn': {'average_document_length': 470.452868852459, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-urd_Arab': {'average_document_length': 475.51024590163934, 'average_query_length': 70.52666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'urd_Latn-urd_Latn': {'average_document_length': 590.5348360655738, 'average_query_length': 90.07, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'urd_Latn-eng_Latn': {'average_document_length': 590.5348360655738, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-urd_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 90.07, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'uzn_Latn-uzn_Latn': {'average_document_length': 539.2418032786885, 'average_query_length': 77.61333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'uzn_Latn-eng_Latn': {'average_document_length': 539.2418032786885, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-uzn_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 77.61333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'vie_Latn-vie_Latn': {'average_document_length': 499.8360655737705, 'average_query_length': 73.05333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'vie_Latn-eng_Latn': {'average_document_length': 499.8360655737705, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-vie_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 73.05333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'war_Latn-war_Latn': {'average_document_length': 592.8688524590164, 'average_query_length': 86.07555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'war_Latn-eng_Latn': {'average_document_length': 592.8688524590164, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-war_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 86.07555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'wol_Latn-wol_Latn': {'average_document_length': 456.9795081967213, 'average_query_length': 70.60555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'wol_Latn-eng_Latn': {'average_document_length': 456.9795081967213, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-wol_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 70.60555555555555, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'xho_Latn-xho_Latn': {'average_document_length': 505.0655737704918, 'average_query_length': 78.50333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'xho_Latn-eng_Latn': {'average_document_length': 505.0655737704918, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-xho_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 78.50333333333333, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'yor_Latn-yor_Latn': {'average_document_length': 459.5204918032787, 'average_query_length': 68.64, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'yor_Latn-eng_Latn': {'average_document_length': 459.5204918032787, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-yor_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 68.64, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'zho_Hans-zho_Hans': {'average_document_length': 159.76024590163934, 'average_query_length': 21.747777777777777, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'zho_Hans-eng_Latn': {'average_document_length': 159.76024590163934, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-zho_Hans': {'average_document_length': 475.51024590163934, 'average_query_length': 21.747777777777777, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'zho_Hant-zho_Hant': {'average_document_length': 149.77254098360655, 'average_query_length': 21.07888888888889, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'zho_Hant-eng_Latn': {'average_document_length': 149.77254098360655, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-zho_Hant': {'average_document_length': 475.51024590163934, 'average_query_length': 21.07888888888889, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'zsm_Latn-zsm_Latn': {'average_document_length': 528.9139344262295, 'average_query_length': 78.92444444444445, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'zsm_Latn-eng_Latn': {'average_document_length': 528.9139344262295, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-zsm_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 78.92444444444445, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'zul_Latn-zul_Latn': {'average_document_length': 532.9713114754098, 'average_query_length': 76.0411111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'zul_Latn-eng_Latn': {'average_document_length': 532.9713114754098, 'average_query_length': 77.34777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-zul_Latn': {'average_document_length': 475.51024590163934, 'average_query_length': 76.0411111111111, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'arb_Arab-arb_Latn': {'average_document_length': 421.96311475409834, 'average_query_length': 67.02444444444444, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'arb_Latn-arb_Arab': {'average_document_length': 555.6188524590164, 'average_query_length': 58.55, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ben_Beng-ben_Latn': {'average_document_length': 467.7745901639344, 'average_query_length': 74.78777777777778, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'ben_Latn-ben_Beng': {'average_document_length': 522.8934426229508, 'average_query_length': 69.48444444444445, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'hin_Deva-hin_Latn': {'average_document_length': 473.55737704918033, 'average_query_length': 74.81222222222222, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'hin_Latn-hin_Deva': {'average_document_length': 541.7315573770492, 'average_query_length': 72.61777777777777, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'npi_Deva-npi_Latn': {'average_document_length': 456.9590163934426, 'average_query_length': 71.89666666666666, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'npi_Latn-npi_Deva': {'average_document_length': 515.9815573770492, 'average_query_length': 66.89666666666666, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'sin_Sinh-sin_Latn': {'average_document_length': 478.66803278688525, 'average_query_length': 94.46666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'sin_Latn-sin_Sinh': {'average_document_length': 590.7889344262295, 'average_query_length': 69.91777777777777, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'urd_Arab-urd_Latn': {'average_document_length': 470.452868852459, 'average_query_length': 90.07, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}, 'urd_Latn-urd_Arab': {'average_document_length': 590.5348360655738, 'average_query_length': 70.52666666666667, 'num_documents': 488, 'num_queries': 900, 'average_relevant_docs_per_query': 1.0}}}} | -| [BengaliDocumentClassification](https://aclanthology.org/2023.eacl-main.4) | ['ben'] | Classification | s2s | [News, Written] | {'test': 2048} | {'test': 1658.1} | -| [BengaliHateSpeechClassification](https://huggingface.co/datasets/bn_hate_speech) (Karim et al., 2020) | ['ben'] | Classification | s2s | [News, Written] | {'train': 3418} | {'train': 103.42} | -| [BengaliSentimentAnalysis](https://data.mendeley.com/datasets/p6zc7krs37/4) (Sazzed et al., 2020) | ['ben'] | Classification | s2s | [Reviews, Written] | {'train': 11807} | {'train': 69.66} | -| [BibleNLPBitextMining](https://arxiv.org/abs/2304.09919) (Akerman et al., 2023) | ['aai', 'aak', 'aau', 'aaz', 'abt', 'abx', 'aby', 'acf', 'acr', 'acu', 'adz', 'aer', 'aey', 'agd', 'agg', 'agm', 'agn', 'agr', 'agt', 'agu', 'aia', 'aii', 'aka', 'ake', 'alp', 'alq', 'als', 'aly', 'ame', 'amf', 'amk', 'amm', 'amn', 'amo', 'amp', 'amr', 'amu', 'amx', 'anh', 'anv', 'aoi', 'aoj', 'aom', 'aon', 'apb', 'ape', 'apn', 'apr', 'apu', 'apw', 'apz', 'arb', 'are', 'arl', 'arn', 'arp', 'asm', 'aso', 'ata', 'atb', 'atd', 'atg', 'att', 'auc', 'aui', 'auy', 'avt', 'awb', 'awk', 'awx', 'azb', 'azg', 'azz', 'bao', 'bba', 'bbb', 'bbr', 'bch', 'bco', 'bdd', 'bea', 'bef', 'bel', 'ben', 'beo', 'beu', 'bgs', 'bgt', 'bhg', 'bhl', 'big', 'bjk', 'bjp', 'bjr', 'bjv', 'bjz', 'bkd', 'bki', 'bkq', 'bkx', 'blw', 'blz', 'bmh', 'bmk', 'bmr', 'bmu', 'bnp', 'boa', 'boj', 'bon', 'box', 'bpr', 'bps', 'bqc', 'bqp', 'bre', 'bsj', 'bsn', 'bsp', 'bss', 'buk', 'bus', 'bvd', 'bvr', 'bxh', 'byr', 'byx', 'bzd', 'bzh', 'bzj', 'caa', 'cab', 'cac', 'caf', 'cak', 'cao', 'cap', 'car', 'cav', 'cax', 'cbc', 'cbi', 'cbk', 'cbr', 'cbs', 'cbt', 'cbu', 'cbv', 'cco', 'ceb', 'cek', 'ces', 'cgc', 'cha', 'chd', 'chf', 'chk', 'chq', 'chz', 'cjo', 'cjv', 'ckb', 'cle', 'clu', 'cme', 'cmn', 'cni', 'cnl', 'cnt', 'cof', 'con', 'cop', 'cot', 'cpa', 'cpb', 'cpc', 'cpu', 'cpy', 'crn', 'crx', 'cso', 'csy', 'cta', 'cth', 'ctp', 'ctu', 'cub', 'cuc', 'cui', 'cuk', 'cut', 'cux', 'cwe', 'cya', 'daa', 'dad', 'dah', 'dan', 'ded', 'deu', 'dgc', 'dgr', 'dgz', 'dhg', 'dif', 'dik', 'dji', 'djk', 'djr', 'dob', 'dop', 'dov', 'dwr', 'dww', 'dwy', 'ebk', 'eko', 'emi', 'emp', 'eng', 'enq', 'epo', 'eri', 'ese', 'esk', 'etr', 'ewe', 'faa', 'fai', 'far', 'ffm', 'for', 'fra', 'fue', 'fuf', 'fuh', 'gah', 'gai', 'gam', 'gaw', 'gdn', 'gdr', 'geb', 'gfk', 'ghs', 'glk', 'gmv', 'gng', 'gnn', 'gnw', 'gof', 'grc', 'gub', 'guh', 'gui', 'guj', 'gul', 'gum', 'gun', 'guo', 'gup', 'gux', 'gvc', 'gvf', 'gvn', 'gvs', 'gwi', 'gym', 'gyr', 'hat', 'hau', 'haw', 'hbo', 'hch', 'heb', 'heg', 'hin', 'hix', 'hla', 'hlt', 'hmo', 'hns', 'hop', 'hot', 'hrv', 'hto', 'hub', 'hui', 'hun', 'hus', 'huu', 'huv', 'hvn', 'ian', 'ign', 'ikk', 'ikw', 'ilo', 'imo', 'inb', 'ind', 'ino', 'iou', 'ipi', 'isn', 'ita', 'iws', 'ixl', 'jac', 'jae', 'jao', 'jic', 'jid', 'jiv', 'jni', 'jpn', 'jvn', 'kan', 'kaq', 'kbc', 'kbh', 'kbm', 'kbq', 'kdc', 'kde', 'kdl', 'kek', 'ken', 'kew', 'kgf', 'kgk', 'kgp', 'khs', 'khz', 'kik', 'kiw', 'kiz', 'kje', 'kjs', 'kkc', 'kkl', 'klt', 'klv', 'kmg', 'kmh', 'kmk', 'kmo', 'kms', 'kmu', 'kne', 'knf', 'knj', 'knv', 'kos', 'kpf', 'kpg', 'kpj', 'kpr', 'kpw', 'kpx', 'kqa', 'kqc', 'kqf', 'kql', 'kqw', 'ksd', 'ksj', 'ksr', 'ktm', 'kto', 'kud', 'kue', 'kup', 'kvg', 'kvn', 'kwd', 'kwf', 'kwi', 'kwj', 'kyc', 'kyf', 'kyg', 'kyq', 'kyz', 'kze', 'lac', 'lat', 'lbb', 'lbk', 'lcm', 'leu', 'lex', 'lgl', 'lid', 'lif', 'lin', 'lit', 'llg', 'lug', 'luo', 'lww', 'maa', 'maj', 'mal', 'mam', 'maq', 'mar', 'mau', 'mav', 'maz', 'mbb', 'mbc', 'mbh', 'mbj', 'mbl', 'mbs', 'mbt', 'mca', 'mcb', 'mcd', 'mcf', 'mco', 'mcp', 'mcq', 'mcr', 'mdy', 'med', 'mee', 'mek', 'meq', 'met', 'meu', 'mgc', 'mgh', 'mgw', 'mhl', 'mib', 'mic', 'mie', 'mig', 'mih', 'mil', 'mio', 'mir', 'mit', 'miz', 'mjc', 'mkj', 'mkl', 'mkn', 'mks', 'mle', 'mlh', 'mlp', 'mmo', 'mmx', 'mna', 'mop', 'mox', 'mph', 'mpj', 'mpm', 'mpp', 'mps', 'mpt', 'mpx', 'mqb', 'mqj', 'msb', 'msc', 'msk', 'msm', 'msy', 'mti', 'mto', 'mux', 'muy', 'mva', 'mvn', 'mwc', 'mwe', 'mwf', 'mwp', 'mxb', 'mxp', 'mxq', 'mxt', 'mya', 'myk', 'myu', 'myw', 'myy', 'mzz', 'nab', 'naf', 'nak', 'nas', 'nbq', 'nca', 'nch', 'ncj', 'ncl', 'ncu', 'ndg', 'ndj', 'nfa', 'ngp', 'ngu', 'nhe', 'nhg', 'nhi', 'nho', 'nhr', 'nhu', 'nhw', 'nhy', 'nif', 'nii', 'nin', 'nko', 'nld', 'nlg', 'nna', 'nnq', 'noa', 'nop', 'not', 'nou', 'npi', 'npl', 'nsn', 'nss', 'ntj', 'ntp', 'ntu', 'nuy', 'nvm', 'nwi', 'nya', 'nys', 'nyu', 'obo', 'okv', 'omw', 'ong', 'ons', 'ood', 'opm', 'ory', 'ote', 'otm', 'otn', 'otq', 'ots', 'pab', 'pad', 'pah', 'pan', 'pao', 'pes', 'pib', 'pio', 'pir', 'piu', 'pjt', 'pls', 'plu', 'pma', 'poe', 'poh', 'poi', 'pol', 'pon', 'por', 'poy', 'ppo', 'prf', 'pri', 'ptp', 'ptu', 'pwg', 'qub', 'quc', 'quf', 'quh', 'qul', 'qup', 'qvc', 'qve', 'qvh', 'qvm', 'qvn', 'qvs', 'qvw', 'qvz', 'qwh', 'qxh', 'qxn', 'qxo', 'rai', 'reg', 'rgu', 'rkb', 'rmc', 'rmy', 'ron', 'roo', 'rop', 'row', 'rro', 'ruf', 'rug', 'rus', 'rwo', 'sab', 'san', 'sbe', 'sbk', 'sbs', 'seh', 'sey', 'sgb', 'sgz', 'shj', 'shp', 'sim', 'sja', 'sll', 'smk', 'snc', 'snn', 'snp', 'snx', 'sny', 'som', 'soq', 'soy', 'spa', 'spl', 'spm', 'spp', 'sps', 'spy', 'sri', 'srm', 'srn', 'srp', 'srq', 'ssd', 'ssg', 'ssx', 'stp', 'sua', 'sue', 'sus', 'suz', 'swe', 'swh', 'swp', 'sxb', 'tac', 'taj', 'tam', 'tav', 'taw', 'tbc', 'tbf', 'tbg', 'tbo', 'tbz', 'tca', 'tcs', 'tcz', 'tdt', 'tee', 'tel', 'ter', 'tet', 'tew', 'tfr', 'tgk', 'tgl', 'tgo', 'tgp', 'tha', 'tif', 'tim', 'tiw', 'tiy', 'tke', 'tku', 'tlf', 'tmd', 'tna', 'tnc', 'tnk', 'tnn', 'tnp', 'toc', 'tod', 'tof', 'toj', 'ton', 'too', 'top', 'tos', 'tpa', 'tpi', 'tpt', 'tpz', 'trc', 'tsw', 'ttc', 'tte', 'tuc', 'tue', 'tuf', 'tuo', 'tur', 'tvk', 'twi', 'txq', 'txu', 'tzj', 'tzo', 'ubr', 'ubu', 'udu', 'uig', 'ukr', 'uli', 'ulk', 'upv', 'ura', 'urb', 'urd', 'uri', 'urt', 'urw', 'usa', 'usp', 'uvh', 'uvl', 'vid', 'vie', 'viv', 'vmy', 'waj', 'wal', 'wap', 'wat', 'wbi', 'wbp', 'wed', 'wer', 'wim', 'wiu', 'wiv', 'wmt', 'wmw', 'wnc', 'wnu', 'wol', 'wos', 'wrk', 'wro', 'wrs', 'wsk', 'wuv', 'xav', 'xbi', 'xed', 'xla', 'xnn', 'xon', 'xsi', 'xtd', 'xtm', 'yaa', 'yad', 'yal', 'yap', 'yaq', 'yby', 'ycn', 'yka', 'yle', 'yml', 'yon', 'yor', 'yrb', 'yre', 'yss', 'yuj', 'yut', 'yuw', 'yva', 'zaa', 'zab', 'zac', 'zad', 'zai', 'zaj', 'zam', 'zao', 'zap', 'zar', 'zas', 'zat', 'zav', 'zaw', 'zca', 'zga', 'zia', 'ziw', 'zlm', 'zos', 'zpc', 'zpl', 'zpm', 'zpo', 'zpq', 'zpu', 'zpv', 'zpz', 'zsr', 'ztq', 'zty', 'zyp'] | BitextMining | s2s | [Religious, Written] | {'train': 256} | {'train': 120} | -| [BigPatentClustering.v2](https://huggingface.co/datasets/NortheasternUniversity/big_patent) (Eva Sharma and Chen Li and Lu Wang, 2019) | ['eng'] | Clustering | p2p | [Legal, Written] | {'test': 2048} | {'test': 30995.5} | -| [BiorxivClusteringP2P.v2](https://api.biorxiv.org/) | ['eng'] | Clustering | p2p | [Academic, Written] | {'test': 2151} | {'test': 1664.0} | -| [BiorxivClusteringS2S.v2](https://api.biorxiv.org/) | ['eng'] | Clustering | s2s | [Academic, Written] | {'test': 2151} | {'test': 101.7} | -| [BlurbsClusteringP2P.v2](https://www.inf.uni-hamburg.de/en/inst/ab/lt/resources/data/germeval-2019-hmc.html) (Steffen Remus, 2019) | ['deu'] | Clustering | p2p | [Fiction, Written] | {'test': 2048} | {'test': 664.09} | -| [BlurbsClusteringS2S.v2](https://www.inf.uni-hamburg.de/en/inst/ab/lt/resources/data/germeval-2019-hmc.html) (Steffen Remus, 2019) | ['deu'] | Clustering | s2s | [Fiction, Written] | {'test': 2048} | {'test': 23.02} | -| [BornholmBitextMining](https://aclanthology.org/W19-6138/) | ['dan'] | BitextMining | s2s | [Web, Social, Fiction, Written] | {'test': 500} | {'test': {'average_sentence1_length': 49.834, 'average_sentence2_length': 38.888, 'num_samples': 500}} | -| [BrazilianToxicTweetsClassification](https://paperswithcode.com/dataset/told-br) (Joao Augusto Leite and Diego F. Silva and Kalina Bontcheva and Carolina Scarton, 2020) | ['por'] | MultilabelClassification | s2s | [Constructed, Written] | {'test': 2048} | {'test': 85.05} | -| [BrightRetrieval](https://huggingface.co/datasets/xlangai/BRIGHT) (Hongjin Su, 2024) | ['eng'] | Retrieval | s2p | [Non-fiction] | {'standard': 1334914, 'long': 7048} | {'standard': 800.3994729248476, 'long': 46527.35839954597} | -| [BulgarianStoreReviewSentimentClassfication](https://doi.org/10.7910/DVN/TXIK9P) (Georgieva-Trifonova et al., 2018) | ['bul'] | Classification | s2s | [Reviews, Written] | {'test': 182} | {'test': 316.7} | -| [CBD](http://2019.poleval.pl/files/poleval2019.pdf) | ['pol'] | Classification | s2s | [Written, Social] | {'test': 1000} | {'test': 93.2} | +| [BSARDRetrieval](https://huggingface.co/datasets/maastrichtlawtech/bsard) (Louis et al., 2022) | ['fra'] | Retrieval | s2p | [Legal, Spoken] | None | None | +| [BUCC.v2](https://comparable.limsi.fr/bucc2018/bucc2018-task.html) | ['cmn', 'deu', 'eng', 'fra', 'rus'] | BitextMining | s2s | [Written] | None | None | +| [Banking77Classification](https://arxiv.org/abs/2003.04807) | ['eng'] | Classification | s2s | [Written] | None | None | +| [BelebeleRetrieval](https://arxiv.org/abs/2308.16884) (Lucas Bandarkar, 2023) | ['acm', 'afr', 'als', 'amh', 'apc', 'arb', 'ars', 'ary', 'arz', 'asm', 'azj', 'bam', 'ben', 'bod', 'bul', 'cat', 'ceb', 'ces', 'ckb', 'dan', 'deu', 'ell', 'eng', 'est', 'eus', 'fin', 'fra', 'fuv', 'gaz', 'grn', 'guj', 'hat', 'hau', 'heb', 'hin', 'hrv', 'hun', 'hye', 'ibo', 'ilo', 'ind', 'isl', 'ita', 'jav', 'jpn', 'kac', 'kan', 'kat', 'kaz', 'kea', 'khk', 'khm', 'kin', 'kir', 'kor', 'lao', 'lin', 'lit', 'lug', 'luo', 'lvs', 'mal', 'mar', 'mkd', 'mlt', 'mri', 'mya', 'nld', 'nob', 'npi', 'nso', 'nya', 'ory', 'pan', 'pbt', 'pes', 'plt', 'pol', 'por', 'ron', 'rus', 'shn', 'sin', 'slk', 'slv', 'sna', 'snd', 'som', 'sot', 'spa', 'srp', 'ssw', 'sun', 'swe', 'swh', 'tam', 'tel', 'tgk', 'tgl', 'tha', 'tir', 'tsn', 'tso', 'tur', 'ukr', 'urd', 'uzn', 'vie', 'war', 'wol', 'xho', 'yor', 'zho', 'zsm', 'zul'] | Retrieval | s2p | [Web, News, Written] | {'test': 521866} | {'test': {'number_of_characters': 76.5, 'num_samples': 521866, 'num_queries': 338378, 'num_documents': 183488, 'average_document_length': 0.0, 'average_query_length': 0.0, 'average_relevant_docs_per_query': 1.0, 'hf_subset_descriptive_stats': {'acm_Arab-acm_Arab': {'number_of_characters': 57.84, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'acm_Arab-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-acm_Arab': {'number_of_characters': 57.84, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'afr_Latn-afr_Latn': {'number_of_characters': 80.05, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'afr_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-afr_Latn': {'number_of_characters': 80.05, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'als_Latn-als_Latn': {'number_of_characters': 78.14, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'als_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-als_Latn': {'number_of_characters': 78.14, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'amh_Ethi-amh_Ethi': {'number_of_characters': 51.16, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.05, 'average_relevant_docs_per_query': 1.0}, 'amh_Ethi-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-amh_Ethi': {'number_of_characters': 51.16, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.05, 'average_relevant_docs_per_query': 1.0}, 'apc_Arab-apc_Arab': {'number_of_characters': 57.86, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'apc_Arab-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-apc_Arab': {'number_of_characters': 57.86, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'arb_Arab-arb_Arab': {'number_of_characters': 60.55, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'arb_Arab-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-arb_Arab': {'number_of_characters': 60.55, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'arb_Latn-arb_Latn': {'number_of_characters': 69.02, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'arb_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-arb_Latn': {'number_of_characters': 69.02, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'ars_Arab-ars_Arab': {'number_of_characters': 58.43, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'ars_Arab-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ars_Arab': {'number_of_characters': 58.43, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'ary_Arab-ary_Arab': {'number_of_characters': 68.02, 'num_samples': 1386, 'num_queries': 898, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'ary_Arab-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ary_Arab': {'number_of_characters': 68.02, 'num_samples': 1386, 'num_queries': 898, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'arz_Arab-arz_Arab': {'number_of_characters': 59.14, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'arz_Arab-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-arz_Arab': {'number_of_characters': 59.14, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'asm_Beng-asm_Beng': {'number_of_characters': 70.26, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'asm_Beng-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-asm_Beng': {'number_of_characters': 70.26, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'azj_Latn-azj_Latn': {'number_of_characters': 75.51, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'azj_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-azj_Latn': {'number_of_characters': 75.51, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'bam_Latn-bam_Latn': {'number_of_characters': 74.34, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'bam_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-bam_Latn': {'number_of_characters': 74.34, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ben_Beng-ben_Beng': {'number_of_characters': 71.48, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ben_Beng-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ben_Beng': {'number_of_characters': 71.48, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ben_Latn-ben_Latn': {'number_of_characters': 76.79, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ben_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ben_Latn': {'number_of_characters': 76.79, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'bod_Tibt-bod_Tibt': {'number_of_characters': 88.9, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'bod_Tibt-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-bod_Tibt': {'number_of_characters': 88.9, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'bul_Cyrl-bul_Cyrl': {'number_of_characters': 74.89, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'bul_Cyrl-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-bul_Cyrl': {'number_of_characters': 74.89, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'cat_Latn-cat_Latn': {'number_of_characters': 77.41, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'cat_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-cat_Latn': {'number_of_characters': 77.41, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ceb_Latn-ceb_Latn': {'number_of_characters': 83.2, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'ceb_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ceb_Latn': {'number_of_characters': 83.2, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'ces_Latn-ces_Latn': {'number_of_characters': 69.73, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ces_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ces_Latn': {'number_of_characters': 69.73, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ckb_Arab-ckb_Arab': {'number_of_characters': 73.05, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ckb_Arab-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ckb_Arab': {'number_of_characters': 73.05, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'dan_Latn-dan_Latn': {'number_of_characters': 74.97, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'dan_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-dan_Latn': {'number_of_characters': 74.97, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'deu_Latn-deu_Latn': {'number_of_characters': 77.32, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'deu_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-deu_Latn': {'number_of_characters': 77.32, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ell_Grek-ell_Grek': {'number_of_characters': 88.93, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'ell_Grek-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ell_Grek': {'number_of_characters': 88.93, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'est_Latn-est_Latn': {'number_of_characters': 69.56, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'est_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-est_Latn': {'number_of_characters': 69.56, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'eus_Latn-eus_Latn': {'number_of_characters': 76.45, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'eus_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-eus_Latn': {'number_of_characters': 76.45, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'fin_Latn-fin_Latn': {'number_of_characters': 74.51, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'fin_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-fin_Latn': {'number_of_characters': 74.51, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'fra_Latn-fra_Latn': {'number_of_characters': 92.54, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'fra_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-fra_Latn': {'number_of_characters': 92.54, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'fuv_Latn-fuv_Latn': {'number_of_characters': 60.42, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'fuv_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-fuv_Latn': {'number_of_characters': 60.42, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'gaz_Latn-gaz_Latn': {'number_of_characters': 87.93, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'gaz_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-gaz_Latn': {'number_of_characters': 87.93, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'grn_Latn-grn_Latn': {'number_of_characters': 77.11, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'grn_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-grn_Latn': {'number_of_characters': 77.11, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'guj_Gujr-guj_Gujr': {'number_of_characters': 64.26, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'guj_Gujr-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-guj_Gujr': {'number_of_characters': 64.26, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'hat_Latn-hat_Latn': {'number_of_characters': 72.65, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'hat_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-hat_Latn': {'number_of_characters': 72.65, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'hau_Latn-hau_Latn': {'number_of_characters': 87.85, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'hau_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-hau_Latn': {'number_of_characters': 87.85, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'heb_Hebr-heb_Hebr': {'number_of_characters': 57.14, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'heb_Hebr-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-heb_Hebr': {'number_of_characters': 57.14, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'hin_Deva-hin_Deva': {'number_of_characters': 74.62, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'hin_Deva-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-hin_Deva': {'number_of_characters': 74.62, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'hin_Latn-hin_Latn': {'number_of_characters': 76.81, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'hin_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-hin_Latn': {'number_of_characters': 76.81, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'hrv_Latn-hrv_Latn': {'number_of_characters': 70.84, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'hrv_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-hrv_Latn': {'number_of_characters': 70.84, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'hun_Latn-hun_Latn': {'number_of_characters': 76.41, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'hun_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-hun_Latn': {'number_of_characters': 76.41, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'hye_Armn-hye_Armn': {'number_of_characters': 77.43, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'hye_Armn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-hye_Armn': {'number_of_characters': 77.43, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ibo_Latn-ibo_Latn': {'number_of_characters': 74.52, 'num_samples': 1387, 'num_queries': 899, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ibo_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ibo_Latn': {'number_of_characters': 74.52, 'num_samples': 1387, 'num_queries': 899, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ilo_Latn-ilo_Latn': {'number_of_characters': 87.76, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'ilo_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ilo_Latn': {'number_of_characters': 87.76, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'ind_Latn-ind_Latn': {'number_of_characters': 84.11, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'ind_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ind_Latn': {'number_of_characters': 84.11, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'isl_Latn-isl_Latn': {'number_of_characters': 79.27, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'isl_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-isl_Latn': {'number_of_characters': 79.27, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'ita_Latn-ita_Latn': {'number_of_characters': 85.5, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'ita_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ita_Latn': {'number_of_characters': 85.5, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'jav_Latn-jav_Latn': {'number_of_characters': 80.61, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'jav_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-jav_Latn': {'number_of_characters': 80.61, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'jpn_Jpan-jpn_Jpan': {'number_of_characters': 37.79, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.04, 'average_relevant_docs_per_query': 1.0}, 'jpn_Jpan-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-jpn_Jpan': {'number_of_characters': 37.79, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.04, 'average_relevant_docs_per_query': 1.0}, 'kac_Latn-kac_Latn': {'number_of_characters': 100.64, 'num_samples': 1387, 'num_queries': 899, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.11, 'average_relevant_docs_per_query': 1.0}, 'kac_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-kac_Latn': {'number_of_characters': 100.64, 'num_samples': 1387, 'num_queries': 899, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.11, 'average_relevant_docs_per_query': 1.0}, 'kan_Knda-kan_Knda': {'number_of_characters': 74.14, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'kan_Knda-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-kan_Knda': {'number_of_characters': 74.14, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'kat_Geor-kat_Geor': {'number_of_characters': 76.81, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'kat_Geor-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-kat_Geor': {'number_of_characters': 76.81, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'kaz_Cyrl-kaz_Cyrl': {'number_of_characters': 72.76, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'kaz_Cyrl-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-kaz_Cyrl': {'number_of_characters': 72.76, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'kea_Latn-kea_Latn': {'number_of_characters': 77.94, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'kea_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-kea_Latn': {'number_of_characters': 77.94, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'khk_Cyrl-khk_Cyrl': {'number_of_characters': 75.33, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'khk_Cyrl-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-khk_Cyrl': {'number_of_characters': 75.33, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'khm_Khmr-khm_Khmr': {'number_of_characters': 77.75, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'khm_Khmr-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-khm_Khmr': {'number_of_characters': 77.75, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'kin_Latn-kin_Latn': {'number_of_characters': 81.9, 'num_samples': 1387, 'num_queries': 899, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'kin_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-kin_Latn': {'number_of_characters': 81.9, 'num_samples': 1387, 'num_queries': 899, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'kir_Cyrl-kir_Cyrl': {'number_of_characters': 76.42, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'kir_Cyrl-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-kir_Cyrl': {'number_of_characters': 76.42, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'kor_Hang-kor_Hang': {'number_of_characters': 37.26, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.04, 'average_relevant_docs_per_query': 1.0}, 'kor_Hang-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-kor_Hang': {'number_of_characters': 37.26, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.04, 'average_relevant_docs_per_query': 1.0}, 'lao_Laoo-lao_Laoo': {'number_of_characters': 65.31, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'lao_Laoo-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-lao_Laoo': {'number_of_characters': 65.31, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'lin_Latn-lin_Latn': {'number_of_characters': 83.57, 'num_samples': 1386, 'num_queries': 898, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'lin_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-lin_Latn': {'number_of_characters': 83.57, 'num_samples': 1386, 'num_queries': 898, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'lit_Latn-lit_Latn': {'number_of_characters': 70.7, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'lit_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-lit_Latn': {'number_of_characters': 70.7, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'lug_Latn-lug_Latn': {'number_of_characters': 80.52, 'num_samples': 1387, 'num_queries': 899, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'lug_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-lug_Latn': {'number_of_characters': 80.52, 'num_samples': 1387, 'num_queries': 899, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'luo_Latn-luo_Latn': {'number_of_characters': 75.14, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'luo_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-luo_Latn': {'number_of_characters': 75.14, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'lvs_Latn-lvs_Latn': {'number_of_characters': 71.98, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'lvs_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-lvs_Latn': {'number_of_characters': 71.98, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'mal_Mlym-mal_Mlym': {'number_of_characters': 82.69, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'mal_Mlym-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-mal_Mlym': {'number_of_characters': 82.69, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'mar_Deva-mar_Deva': {'number_of_characters': 70.63, 'num_samples': 1387, 'num_queries': 899, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'mar_Deva-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-mar_Deva': {'number_of_characters': 70.63, 'num_samples': 1387, 'num_queries': 899, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'mkd_Cyrl-mkd_Cyrl': {'number_of_characters': 76.01, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'mkd_Cyrl-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-mkd_Cyrl': {'number_of_characters': 76.01, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'mlt_Latn-mlt_Latn': {'number_of_characters': 77.0, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'mlt_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-mlt_Latn': {'number_of_characters': 77.0, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'mri_Latn-mri_Latn': {'number_of_characters': 83.71, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'mri_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-mri_Latn': {'number_of_characters': 83.71, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'mya_Mymr-mya_Mymr': {'number_of_characters': 91.28, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'mya_Mymr-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-mya_Mymr': {'number_of_characters': 91.28, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'nld_Latn-nld_Latn': {'number_of_characters': 77.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'nld_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-nld_Latn': {'number_of_characters': 77.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'nob_Latn-nob_Latn': {'number_of_characters': 73.05, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'nob_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-nob_Latn': {'number_of_characters': 73.05, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'npi_Deva-npi_Deva': {'number_of_characters': 68.9, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'npi_Deva-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-npi_Deva': {'number_of_characters': 68.9, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'npi_Latn-npi_Latn': {'number_of_characters': 73.9, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'npi_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-npi_Latn': {'number_of_characters': 73.9, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'nso_Latn-nso_Latn': {'number_of_characters': 88.77, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'nso_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-nso_Latn': {'number_of_characters': 88.77, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'nya_Latn-nya_Latn': {'number_of_characters': 92.79, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'nya_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-nya_Latn': {'number_of_characters': 92.79, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'ory_Orya-ory_Orya': {'number_of_characters': 74.96, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ory_Orya-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ory_Orya': {'number_of_characters': 74.96, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'pan_Guru-pan_Guru': {'number_of_characters': 75.3, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'pan_Guru-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-pan_Guru': {'number_of_characters': 75.3, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'pbt_Arab-pbt_Arab': {'number_of_characters': 69.67, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'pbt_Arab-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-pbt_Arab': {'number_of_characters': 69.67, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'pes_Arab-pes_Arab': {'number_of_characters': 66.75, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'pes_Arab-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-pes_Arab': {'number_of_characters': 66.75, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'plt_Latn-plt_Latn': {'number_of_characters': 97.0, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.11, 'average_relevant_docs_per_query': 1.0}, 'plt_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-plt_Latn': {'number_of_characters': 97.0, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.11, 'average_relevant_docs_per_query': 1.0}, 'pol_Latn-pol_Latn': {'number_of_characters': 76.1, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'pol_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-pol_Latn': {'number_of_characters': 76.1, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'por_Latn-por_Latn': {'number_of_characters': 80.12, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'por_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-por_Latn': {'number_of_characters': 80.12, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'ron_Latn-ron_Latn': {'number_of_characters': 80.74, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'ron_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ron_Latn': {'number_of_characters': 80.74, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'rus_Cyrl-rus_Cyrl': {'number_of_characters': 85.16, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'rus_Cyrl-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-rus_Cyrl': {'number_of_characters': 85.16, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'shn_Mymr-shn_Mymr': {'number_of_characters': 77.9, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'shn_Mymr-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-shn_Mymr': {'number_of_characters': 77.9, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'sin_Latn-sin_Latn': {'number_of_characters': 96.47, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'sin_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-sin_Latn': {'number_of_characters': 96.47, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'sin_Sinh-sin_Sinh': {'number_of_characters': 71.92, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'sin_Sinh-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-sin_Sinh': {'number_of_characters': 71.92, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'slk_Latn-slk_Latn': {'number_of_characters': 70.54, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'slk_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-slk_Latn': {'number_of_characters': 70.54, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'slv_Latn-slv_Latn': {'number_of_characters': 70.8, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'slv_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-slv_Latn': {'number_of_characters': 70.8, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'sna_Latn-sna_Latn': {'number_of_characters': 83.31, 'num_samples': 1387, 'num_queries': 899, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'sna_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-sna_Latn': {'number_of_characters': 83.31, 'num_samples': 1387, 'num_queries': 899, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'snd_Arab-snd_Arab': {'number_of_characters': 65.42, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'snd_Arab-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-snd_Arab': {'number_of_characters': 65.42, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'som_Latn-som_Latn': {'number_of_characters': 92.96, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'som_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-som_Latn': {'number_of_characters': 92.96, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'sot_Latn-sot_Latn': {'number_of_characters': 85.13, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'sot_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-sot_Latn': {'number_of_characters': 85.13, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'spa_Latn-spa_Latn': {'number_of_characters': 84.16, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'spa_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-spa_Latn': {'number_of_characters': 84.16, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'srp_Cyrl-srp_Cyrl': {'number_of_characters': 69.5, 'num_samples': 1387, 'num_queries': 899, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'srp_Cyrl-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-srp_Cyrl': {'number_of_characters': 69.5, 'num_samples': 1387, 'num_queries': 899, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ssw_Latn-ssw_Latn': {'number_of_characters': 83.1, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'ssw_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ssw_Latn': {'number_of_characters': 83.1, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'sun_Latn-sun_Latn': {'number_of_characters': 80.16, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'sun_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-sun_Latn': {'number_of_characters': 80.16, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'swe_Latn-swe_Latn': {'number_of_characters': 70.68, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'swe_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-swe_Latn': {'number_of_characters': 70.68, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'swh_Latn-swh_Latn': {'number_of_characters': 82.56, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'swh_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-swh_Latn': {'number_of_characters': 82.56, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'tam_Taml-tam_Taml': {'number_of_characters': 83.13, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'tam_Taml-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tam_Taml': {'number_of_characters': 83.13, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'tel_Telu-tel_Telu': {'number_of_characters': 74.19, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'tel_Telu-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tel_Telu': {'number_of_characters': 74.19, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'tgk_Cyrl-tgk_Cyrl': {'number_of_characters': 76.28, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'tgk_Cyrl-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tgk_Cyrl': {'number_of_characters': 76.28, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'tgl_Latn-tgl_Latn': {'number_of_characters': 84.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'tgl_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tgl_Latn': {'number_of_characters': 84.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'tha_Thai-tha_Thai': {'number_of_characters': 61.47, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'tha_Thai-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tha_Thai': {'number_of_characters': 61.47, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'tir_Ethi-tir_Ethi': {'number_of_characters': 54.0, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'tir_Ethi-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tir_Ethi': {'number_of_characters': 54.0, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'tsn_Latn-tsn_Latn': {'number_of_characters': 89.13, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'tsn_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tsn_Latn': {'number_of_characters': 89.13, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'tso_Latn-tso_Latn': {'number_of_characters': 93.69, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'tso_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tso_Latn': {'number_of_characters': 93.69, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'tur_Latn-tur_Latn': {'number_of_characters': 73.56, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'tur_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-tur_Latn': {'number_of_characters': 73.56, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ukr_Cyrl-ukr_Cyrl': {'number_of_characters': 74.08, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ukr_Cyrl-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-ukr_Cyrl': {'number_of_characters': 74.08, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'urd_Arab-urd_Arab': {'number_of_characters': 72.53, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'urd_Arab-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-urd_Arab': {'number_of_characters': 72.53, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'urd_Latn-urd_Latn': {'number_of_characters': 92.07, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'urd_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-urd_Latn': {'number_of_characters': 92.07, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'uzn_Latn-uzn_Latn': {'number_of_characters': 79.61, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'uzn_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-uzn_Latn': {'number_of_characters': 79.61, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'vie_Latn-vie_Latn': {'number_of_characters': 75.05, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'vie_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-vie_Latn': {'number_of_characters': 75.05, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'war_Latn-war_Latn': {'number_of_characters': 88.08, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'war_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-war_Latn': {'number_of_characters': 88.08, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'wol_Latn-wol_Latn': {'number_of_characters': 72.61, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'wol_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-wol_Latn': {'number_of_characters': 72.61, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'xho_Latn-xho_Latn': {'number_of_characters': 80.5, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'xho_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-xho_Latn': {'number_of_characters': 80.5, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'yor_Latn-yor_Latn': {'number_of_characters': 70.64, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'yor_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-yor_Latn': {'number_of_characters': 70.64, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'zho_Hans-zho_Hans': {'number_of_characters': 23.75, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.02, 'average_relevant_docs_per_query': 1.0}, 'zho_Hans-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-zho_Hans': {'number_of_characters': 23.75, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.02, 'average_relevant_docs_per_query': 1.0}, 'zho_Hant-zho_Hant': {'number_of_characters': 23.08, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.02, 'average_relevant_docs_per_query': 1.0}, 'zho_Hant-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-zho_Hant': {'number_of_characters': 23.08, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.02, 'average_relevant_docs_per_query': 1.0}, 'zsm_Latn-zsm_Latn': {'number_of_characters': 80.92, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'zsm_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-zsm_Latn': {'number_of_characters': 80.92, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'zul_Latn-zul_Latn': {'number_of_characters': 78.04, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'zul_Latn-eng_Latn': {'number_of_characters': 79.35, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.09, 'average_relevant_docs_per_query': 1.0}, 'eng_Latn-zul_Latn': {'number_of_characters': 78.04, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'arb_Arab-arb_Latn': {'number_of_characters': 69.02, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'arb_Latn-arb_Arab': {'number_of_characters': 60.55, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'ben_Beng-ben_Latn': {'number_of_characters': 76.79, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'ben_Latn-ben_Beng': {'number_of_characters': 71.48, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'hin_Deva-hin_Latn': {'number_of_characters': 76.81, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'hin_Latn-hin_Deva': {'number_of_characters': 74.62, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'npi_Deva-npi_Latn': {'number_of_characters': 73.9, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'npi_Latn-npi_Deva': {'number_of_characters': 68.9, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'sin_Sinh-sin_Latn': {'number_of_characters': 96.47, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'sin_Latn-sin_Sinh': {'number_of_characters': 71.92, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}, 'urd_Arab-urd_Latn': {'number_of_characters': 92.07, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'urd_Latn-urd_Arab': {'number_of_characters': 72.53, 'num_samples': 1388, 'num_queries': 900, 'num_documents': 488, 'average_document_length': 0.0, 'average_query_length': 0.08, 'average_relevant_docs_per_query': 1.0}}}} | +| [BengaliDocumentClassification](https://aclanthology.org/2023.eacl-main.4) | ['ben'] | Classification | s2s | [News, Written] | None | None | +| [BengaliHateSpeechClassification](https://huggingface.co/datasets/bn_hate_speech) (Karim et al., 2020) | ['ben'] | Classification | s2s | [News, Written] | None | None | +| [BengaliSentimentAnalysis](https://data.mendeley.com/datasets/p6zc7krs37/4) (Sazzed et al., 2020) | ['ben'] | Classification | s2s | [Reviews, Written] | None | None | +| [BibleNLPBitextMining](https://arxiv.org/abs/2304.09919) (Akerman et al., 2023) | ['aai', 'aak', 'aau', 'aaz', 'abt', 'abx', 'aby', 'acf', 'acr', 'acu', 'adz', 'aer', 'aey', 'agd', 'agg', 'agm', 'agn', 'agr', 'agt', 'agu', 'aia', 'aii', 'aka', 'ake', 'alp', 'alq', 'als', 'aly', 'ame', 'amf', 'amk', 'amm', 'amn', 'amo', 'amp', 'amr', 'amu', 'amx', 'anh', 'anv', 'aoi', 'aoj', 'aom', 'aon', 'apb', 'ape', 'apn', 'apr', 'apu', 'apw', 'apz', 'arb', 'are', 'arl', 'arn', 'arp', 'asm', 'aso', 'ata', 'atb', 'atd', 'atg', 'att', 'auc', 'aui', 'auy', 'avt', 'awb', 'awk', 'awx', 'azb', 'azg', 'azz', 'bao', 'bba', 'bbb', 'bbr', 'bch', 'bco', 'bdd', 'bea', 'bef', 'bel', 'ben', 'beo', 'beu', 'bgs', 'bgt', 'bhg', 'bhl', 'big', 'bjk', 'bjp', 'bjr', 'bjv', 'bjz', 'bkd', 'bki', 'bkq', 'bkx', 'blw', 'blz', 'bmh', 'bmk', 'bmr', 'bmu', 'bnp', 'boa', 'boj', 'bon', 'box', 'bpr', 'bps', 'bqc', 'bqp', 'bre', 'bsj', 'bsn', 'bsp', 'bss', 'buk', 'bus', 'bvd', 'bvr', 'bxh', 'byr', 'byx', 'bzd', 'bzh', 'bzj', 'caa', 'cab', 'cac', 'caf', 'cak', 'cao', 'cap', 'car', 'cav', 'cax', 'cbc', 'cbi', 'cbk', 'cbr', 'cbs', 'cbt', 'cbu', 'cbv', 'cco', 'ceb', 'cek', 'ces', 'cgc', 'cha', 'chd', 'chf', 'chk', 'chq', 'chz', 'cjo', 'cjv', 'ckb', 'cle', 'clu', 'cme', 'cmn', 'cni', 'cnl', 'cnt', 'cof', 'con', 'cop', 'cot', 'cpa', 'cpb', 'cpc', 'cpu', 'cpy', 'crn', 'crx', 'cso', 'csy', 'cta', 'cth', 'ctp', 'ctu', 'cub', 'cuc', 'cui', 'cuk', 'cut', 'cux', 'cwe', 'cya', 'daa', 'dad', 'dah', 'dan', 'ded', 'deu', 'dgc', 'dgr', 'dgz', 'dhg', 'dif', 'dik', 'dji', 'djk', 'djr', 'dob', 'dop', 'dov', 'dwr', 'dww', 'dwy', 'ebk', 'eko', 'emi', 'emp', 'eng', 'enq', 'epo', 'eri', 'ese', 'esk', 'etr', 'ewe', 'faa', 'fai', 'far', 'ffm', 'for', 'fra', 'fue', 'fuf', 'fuh', 'gah', 'gai', 'gam', 'gaw', 'gdn', 'gdr', 'geb', 'gfk', 'ghs', 'glk', 'gmv', 'gng', 'gnn', 'gnw', 'gof', 'grc', 'gub', 'guh', 'gui', 'guj', 'gul', 'gum', 'gun', 'guo', 'gup', 'gux', 'gvc', 'gvf', 'gvn', 'gvs', 'gwi', 'gym', 'gyr', 'hat', 'hau', 'haw', 'hbo', 'hch', 'heb', 'heg', 'hin', 'hix', 'hla', 'hlt', 'hmo', 'hns', 'hop', 'hot', 'hrv', 'hto', 'hub', 'hui', 'hun', 'hus', 'huu', 'huv', 'hvn', 'ian', 'ign', 'ikk', 'ikw', 'ilo', 'imo', 'inb', 'ind', 'ino', 'iou', 'ipi', 'isn', 'ita', 'iws', 'ixl', 'jac', 'jae', 'jao', 'jic', 'jid', 'jiv', 'jni', 'jpn', 'jvn', 'kan', 'kaq', 'kbc', 'kbh', 'kbm', 'kbq', 'kdc', 'kde', 'kdl', 'kek', 'ken', 'kew', 'kgf', 'kgk', 'kgp', 'khs', 'khz', 'kik', 'kiw', 'kiz', 'kje', 'kjs', 'kkc', 'kkl', 'klt', 'klv', 'kmg', 'kmh', 'kmk', 'kmo', 'kms', 'kmu', 'kne', 'knf', 'knj', 'knv', 'kos', 'kpf', 'kpg', 'kpj', 'kpr', 'kpw', 'kpx', 'kqa', 'kqc', 'kqf', 'kql', 'kqw', 'ksd', 'ksj', 'ksr', 'ktm', 'kto', 'kud', 'kue', 'kup', 'kvg', 'kvn', 'kwd', 'kwf', 'kwi', 'kwj', 'kyc', 'kyf', 'kyg', 'kyq', 'kyz', 'kze', 'lac', 'lat', 'lbb', 'lbk', 'lcm', 'leu', 'lex', 'lgl', 'lid', 'lif', 'lin', 'lit', 'llg', 'lug', 'luo', 'lww', 'maa', 'maj', 'mal', 'mam', 'maq', 'mar', 'mau', 'mav', 'maz', 'mbb', 'mbc', 'mbh', 'mbj', 'mbl', 'mbs', 'mbt', 'mca', 'mcb', 'mcd', 'mcf', 'mco', 'mcp', 'mcq', 'mcr', 'mdy', 'med', 'mee', 'mek', 'meq', 'met', 'meu', 'mgc', 'mgh', 'mgw', 'mhl', 'mib', 'mic', 'mie', 'mig', 'mih', 'mil', 'mio', 'mir', 'mit', 'miz', 'mjc', 'mkj', 'mkl', 'mkn', 'mks', 'mle', 'mlh', 'mlp', 'mmo', 'mmx', 'mna', 'mop', 'mox', 'mph', 'mpj', 'mpm', 'mpp', 'mps', 'mpt', 'mpx', 'mqb', 'mqj', 'msb', 'msc', 'msk', 'msm', 'msy', 'mti', 'mto', 'mux', 'muy', 'mva', 'mvn', 'mwc', 'mwe', 'mwf', 'mwp', 'mxb', 'mxp', 'mxq', 'mxt', 'mya', 'myk', 'myu', 'myw', 'myy', 'mzz', 'nab', 'naf', 'nak', 'nas', 'nbq', 'nca', 'nch', 'ncj', 'ncl', 'ncu', 'ndg', 'ndj', 'nfa', 'ngp', 'ngu', 'nhe', 'nhg', 'nhi', 'nho', 'nhr', 'nhu', 'nhw', 'nhy', 'nif', 'nii', 'nin', 'nko', 'nld', 'nlg', 'nna', 'nnq', 'noa', 'nop', 'not', 'nou', 'npi', 'npl', 'nsn', 'nss', 'ntj', 'ntp', 'ntu', 'nuy', 'nvm', 'nwi', 'nya', 'nys', 'nyu', 'obo', 'okv', 'omw', 'ong', 'ons', 'ood', 'opm', 'ory', 'ote', 'otm', 'otn', 'otq', 'ots', 'pab', 'pad', 'pah', 'pan', 'pao', 'pes', 'pib', 'pio', 'pir', 'piu', 'pjt', 'pls', 'plu', 'pma', 'poe', 'poh', 'poi', 'pol', 'pon', 'por', 'poy', 'ppo', 'prf', 'pri', 'ptp', 'ptu', 'pwg', 'qub', 'quc', 'quf', 'quh', 'qul', 'qup', 'qvc', 'qve', 'qvh', 'qvm', 'qvn', 'qvs', 'qvw', 'qvz', 'qwh', 'qxh', 'qxn', 'qxo', 'rai', 'reg', 'rgu', 'rkb', 'rmc', 'rmy', 'ron', 'roo', 'rop', 'row', 'rro', 'ruf', 'rug', 'rus', 'rwo', 'sab', 'san', 'sbe', 'sbk', 'sbs', 'seh', 'sey', 'sgb', 'sgz', 'shj', 'shp', 'sim', 'sja', 'sll', 'smk', 'snc', 'snn', 'snp', 'snx', 'sny', 'som', 'soq', 'soy', 'spa', 'spl', 'spm', 'spp', 'sps', 'spy', 'sri', 'srm', 'srn', 'srp', 'srq', 'ssd', 'ssg', 'ssx', 'stp', 'sua', 'sue', 'sus', 'suz', 'swe', 'swh', 'swp', 'sxb', 'tac', 'taj', 'tam', 'tav', 'taw', 'tbc', 'tbf', 'tbg', 'tbo', 'tbz', 'tca', 'tcs', 'tcz', 'tdt', 'tee', 'tel', 'ter', 'tet', 'tew', 'tfr', 'tgk', 'tgl', 'tgo', 'tgp', 'tha', 'tif', 'tim', 'tiw', 'tiy', 'tke', 'tku', 'tlf', 'tmd', 'tna', 'tnc', 'tnk', 'tnn', 'tnp', 'toc', 'tod', 'tof', 'toj', 'ton', 'too', 'top', 'tos', 'tpa', 'tpi', 'tpt', 'tpz', 'trc', 'tsw', 'ttc', 'tte', 'tuc', 'tue', 'tuf', 'tuo', 'tur', 'tvk', 'twi', 'txq', 'txu', 'tzj', 'tzo', 'ubr', 'ubu', 'udu', 'uig', 'ukr', 'uli', 'ulk', 'upv', 'ura', 'urb', 'urd', 'uri', 'urt', 'urw', 'usa', 'usp', 'uvh', 'uvl', 'vid', 'vie', 'viv', 'vmy', 'waj', 'wal', 'wap', 'wat', 'wbi', 'wbp', 'wed', 'wer', 'wim', 'wiu', 'wiv', 'wmt', 'wmw', 'wnc', 'wnu', 'wol', 'wos', 'wrk', 'wro', 'wrs', 'wsk', 'wuv', 'xav', 'xbi', 'xed', 'xla', 'xnn', 'xon', 'xsi', 'xtd', 'xtm', 'yaa', 'yad', 'yal', 'yap', 'yaq', 'yby', 'ycn', 'yka', 'yle', 'yml', 'yon', 'yor', 'yrb', 'yre', 'yss', 'yuj', 'yut', 'yuw', 'yva', 'zaa', 'zab', 'zac', 'zad', 'zai', 'zaj', 'zam', 'zao', 'zap', 'zar', 'zas', 'zat', 'zav', 'zaw', 'zca', 'zga', 'zia', 'ziw', 'zlm', 'zos', 'zpc', 'zpl', 'zpm', 'zpo', 'zpq', 'zpu', 'zpv', 'zpz', 'zsr', 'ztq', 'zty', 'zyp'] | BitextMining | s2s | [Religious, Written] | None | None | +| [BigPatentClustering.v2](https://huggingface.co/datasets/NortheasternUniversity/big_patent) (Eva Sharma and Chen Li and Lu Wang, 2019) | ['eng'] | Clustering | p2p | [Legal, Written] | None | None | +| [BiorxivClusteringP2P.v2](https://api.biorxiv.org/) | ['eng'] | Clustering | p2p | [Academic, Written] | None | None | +| [BiorxivClusteringS2S.v2](https://api.biorxiv.org/) | ['eng'] | Clustering | s2s | [Academic, Written] | None | None | +| [BlurbsClusteringP2P.v2](https://www.inf.uni-hamburg.de/en/inst/ab/lt/resources/data/germeval-2019-hmc.html) (Steffen Remus, 2019) | ['deu'] | Clustering | p2p | [Fiction, Written] | None | None | +| [BlurbsClusteringS2S.v2](https://www.inf.uni-hamburg.de/en/inst/ab/lt/resources/data/germeval-2019-hmc.html) (Steffen Remus, 2019) | ['deu'] | Clustering | s2s | [Fiction, Written] | None | None | +| [BornholmBitextMining](https://aclanthology.org/W19-6138/) | ['dan'] | BitextMining | s2s | [Web, Social, Fiction, Written] | {'test': 500} | {'test': {'average_sentence1_length': 49.83, 'average_sentence2_length': 38.89, 'num_samples': 500, 'number_of_characters': 44361}} | +| [BrazilianToxicTweetsClassification](https://paperswithcode.com/dataset/told-br) (Joao Augusto Leite and Diego F. Silva and Kalina Bontcheva and Carolina Scarton, 2020) | ['por'] | MultilabelClassification | s2s | [Constructed, Written] | None | None | +| [BrightRetrieval](https://huggingface.co/datasets/xlangai/BRIGHT) (Hongjin Su, 2024) | ['eng'] | Retrieval | s2p | [Non-fiction] | None | None | +| [BulgarianStoreReviewSentimentClassfication](https://doi.org/10.7910/DVN/TXIK9P) (Georgieva-Trifonova et al., 2018) | ['bul'] | Classification | s2s | [Reviews, Written] | None | None | +| [CBD](http://2019.poleval.pl/files/poleval2019.pdf) | ['pol'] | Classification | s2s | [Written, Social] | None | None | | [CDSC-E](https://aclanthology.org/P17-1073.pdf) | ['pol'] | PairClassification | s2s | [Written] | None | None | -| [CDSC-R](https://aclanthology.org/P17-1073.pdf) | ['pol'] | STS | s2s | [Web, Written] | {'test': 1000} | {'test': 75.24} | -| [CEDRClassification](https://www.sciencedirect.com/science/article/pii/S1877050921013247) (Sboev et al., 2021) | ['rus'] | MultilabelClassification | s2s | [Web, Social, Blog, Written] | {'test': 1882} | {'test': {'average_text_length': 91.20563230605738, 'average_label_per_text': 0.620616365568544, 'num_samples': 1882, 'unique_labels': 6, 'labels': {'null': {'count': 734}, '3': {'count': 141}, '2': {'count': 170}, '1': {'count': 379}, '0': {'count': 353}, '4': {'count': 125}}}} | -| [CLSClusteringP2P.v2](https://arxiv.org/abs/2209.05034) (Yudong Li, 2022) | ['cmn'] | Clustering | p2p | [Academic, Written] | {'test': 2048} | {} | -| [CLSClusteringS2S.v2](https://arxiv.org/abs/2209.05034) (Yudong Li, 2022) | ['cmn'] | Clustering | s2s | [Academic, Written] | {'test': 2048} | {} | -| [CMedQAv1-reranking](https://github.com/zhangsheng93/cMedQA) (Zhang et al., 2017) | ['cmn'] | Reranking | s2s | [Medical, Written] | {'test': 2000} | {'test': 165} | +| [CDSC-R](https://aclanthology.org/P17-1073.pdf) | ['pol'] | STS | s2s | [Web, Written] | None | None | +| [CEDRClassification](https://www.sciencedirect.com/science/article/pii/S1877050921013247) (Sboev et al., 2021) | ['rus'] | MultilabelClassification | s2s | [Web, Social, Blog, Written] | {'test': 1882} | {'test': {'average_text_length': 91.21, 'number_of_characters': 171649, 'average_label_per_text': 0.62, 'num_samples': 1882, 'unique_labels': 6, 'labels': {'None': {'count': 734}, '3': {'count': 141}, '2': {'count': 170}, '1': {'count': 379}, '0': {'count': 353}, '4': {'count': 125}}}} | +| [CLSClusteringP2P.v2](https://arxiv.org/abs/2209.05034) (Yudong Li, 2022) | ['cmn'] | Clustering | p2p | [Academic, Written] | None | None | +| [CLSClusteringS2S.v2](https://arxiv.org/abs/2209.05034) (Yudong Li, 2022) | ['cmn'] | Clustering | s2s | [Academic, Written] | None | None | +| [CMedQAv1-reranking](https://github.com/zhangsheng93/cMedQA) (Zhang et al., 2017) | ['cmn'] | Reranking | s2s | [Medical, Written] | None | None | | [CMedQAv2-reranking](https://github.com/zhangsheng93/cMedQA2) (S. Zhang, 2018) | ['cmn'] | Reranking | s2s | | None | None | -| [COIRCodeSearchNetRetrieval](https://huggingface.co/datasets/code_search_net/) (Husain et al., 2019) | ['go', 'java', 'javascript', 'php', 'python', 'ruby'] | Retrieval | p2p | [Programming, Written] | {'test': 1000} | {'test': {'python': {'average_document_length': 466.546, 'average_query_length': 862.842, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'javascript': {'average_document_length': 186.018, 'average_query_length': 1415.632, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'go': {'average_document_length': 125.213, 'average_query_length': 563.729, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'ruby': {'average_document_length': 313.818, 'average_query_length': 577.634, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'java': {'average_document_length': 420.287, 'average_query_length': 690.36, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'php': {'average_document_length': 162.119, 'average_query_length': 712.129, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}}} | -| [CPUSpeedTask](https://github.com/KennethEnevoldsen/scandinavian-embedding-benchmark/blob/c8376f967d1294419be1d3eb41217d04cd3a65d3/src/seb/registered_tasks/speed.py#L83-L96) | ['eng'] | Speed | s2s | [Fiction, Written] | {'test': 1} | {'test': 3591} | -| [CQADupstackAndroidRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | {'test': {'average_document_length': 593.701974084703, 'average_query_length': 51.76680972818312, 'num_documents': 22998, 'num_queries': 699, 'average_relevant_docs_per_query': 2.4263233190271816}} | -| [CQADupstackEnglishRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | {'test': {'average_document_length': 482.4710971880361, 'average_query_length': 48.32993630573248, 'num_documents': 40221, 'num_queries': 1570, 'average_relevant_docs_per_query': 2.3980891719745223}} | -| [CQADupstackGamingRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | {'test': {'average_document_length': 488.74152888457206, 'average_query_length': 48.772413793103446, 'num_documents': 45301, 'num_queries': 1595, 'average_relevant_docs_per_query': 1.418808777429467}} | -| [CQADupstackGisRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | {'test': {'average_document_length': 1012.167813587693, 'average_query_length': 52.2, 'num_documents': 37637, 'num_queries': 885, 'average_relevant_docs_per_query': 1.2587570621468926}} | -| [CQADupstackMathematicaRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | {'test': {'average_document_length': 1153.4967375037413, 'average_query_length': 48.90547263681592, 'num_documents': 16705, 'num_queries': 804, 'average_relevant_docs_per_query': 1.6890547263681592}} | -| [CQADupstackPhysicsRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | {'test': {'average_document_length': 818.6476145735463, 'average_query_length': 53.36477382098171, 'num_documents': 38316, 'num_queries': 1039, 'average_relevant_docs_per_query': 1.8604427333974976}} | -| [CQADupstackProgrammersRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | [Programming, Written, Non-fiction] | None | {'test': {'average_document_length': 1055.7033814022875, 'average_query_length': 55.1837899543379, 'num_documents': 32176, 'num_queries': 876, 'average_relevant_docs_per_query': 1.9121004566210045}} | -| [CQADupstackStatsRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | {'test': {'average_document_length': 1055.1668598736662, 'average_query_length': 56.31748466257669, 'num_documents': 42269, 'num_queries': 652, 'average_relevant_docs_per_query': 1.4003067484662577}} | -| [CQADupstackTexRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | {'test': {'average_document_length': 1297.09043177285, 'average_query_length': 46.935306262904334, 'num_documents': 68184, 'num_queries': 2906, 'average_relevant_docs_per_query': 1.7735719201651754}} | -| [CQADupstackUnixRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | {'test': {'average_document_length': 1004.8120383267908, 'average_query_length': 50.32369402985075, 'num_documents': 47382, 'num_queries': 1072, 'average_relevant_docs_per_query': 1.5792910447761195}} | -| [CQADupstackWebmastersRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | {'test': {'average_document_length': 707.3635736857225, 'average_query_length': 51.93478260869565, 'num_documents': 17405, 'num_queries': 506, 'average_relevant_docs_per_query': 2.7569169960474307}} | -| [CQADupstackWordpressRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | {'test': {'average_document_length': 1122.7690155333814, 'average_query_length': 48.7264325323475, 'num_documents': 48605, 'num_queries': 541, 'average_relevant_docs_per_query': 1.3752310536044363}} | -| [CSFDCZMovieReviewSentimentClassification](https://arxiv.org/abs/2304.01922) (Michal Štefánik, 2023) | ['ces'] | Classification | s2s | [Reviews, Written] | {'test': 2048} | {'test': 386.5} | -| [CSFDSKMovieReviewSentimentClassification](https://arxiv.org/abs/2304.01922) (Michal Štefánik, 2023) | ['slk'] | Classification | s2s | [Reviews, Written] | {'test': 2048} | {'test': 366.2} | -| [CTKFactsNLI](https://arxiv.org/abs/2201.11115) (Ullrich et al., 2023) | ['ces'] | PairClassification | s2s | [News, Written] | {'test': 375, 'validation': 305} | {'test': 225.62, 'validation': 219.32} | -| [CUADAffiliateLicenseLicenseeLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 198} | {'test': 484.11} | -| [CUADAffiliateLicenseLicensorLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 88} | {'test': 633.4} | -| [CUADAntiAssignmentLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 1172} | {'test': 340.81} | -| [CUADAuditRightsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 1216} | {'test': 337.14} | -| [CUADCapOnLiabilityLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 1246} | {'test': 375.74} | -| [CUADChangeOfControlLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 416} | {'test': 391.96} | -| [CUADCompetitiveRestrictionExceptionLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 220} | {'test': 433.04} | -| [CUADCovenantNotToSueLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 308} | {'test': 402.97} | -| [CUADEffectiveDateLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 236} | {'test': 277.62} | -| [CUADExclusivityLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 762} | {'test': 369.17} | -| [CUADExpirationDateLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 876} | {'test': 309.27} | -| [CUADGoverningLawLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 876} | {'test': 289.87} | -| [CUADIPOwnershipAssignmentLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 576} | {'test': 414.0} | -| [CUADInsuranceLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 1030} | {'test': 365.54} | -| [CUADIrrevocableOrPerpetualLicenseLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 280} | {'test': 473.4} | -| [CUADJointIPOwnershipLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 192} | {'test': 374.17} | -| [CUADLicenseGrantLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 1396} | {'test': 409.89} | -| [CUADLiquidatedDamagesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 220} | {'test': 351.76} | -| [CUADMinimumCommitmentLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 772} | {'test': 364.16} | -| [CUADMostFavoredNationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 64} | {'test': 418.75} | -| [CUADNoSolicitOfCustomersLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 84} | {'test': 392.89} | -| [CUADNoSolicitOfEmployeesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 142} | {'test': 417.94} | -| [CUADNonCompeteLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 442} | {'test': 383.2} | -| [CUADNonDisparagementLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 100} | {'test': 403.08} | -| [CUADNonTransferableLicenseLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 542} | {'test': 399.16} | -| [CUADNoticePeriodToTerminateRenewalLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 222} | {'test': 354.85} | -| [CUADPostTerminationServicesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 808} | {'test': 422.53} | -| [CUADPriceRestrictionsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 46} | {'test': 324.71} | -| [CUADRenewalTermLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 386} | {'test': 340.87} | -| [CUADRevenueProfitSharingLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 774} | {'test': 371.55} | -| [CUADRofrRofoRofnLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 690} | {'test': 395.46} | -| [CUADSourceCodeEscrowLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 118} | {'test': 399.18} | -| [CUADTerminationForConvenienceLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 430} | {'test': 326.3} | -| [CUADThirdPartyBeneficiaryLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 68} | {'test': 261.04} | -| [CUADUncappedLiabilityLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 294} | {'test': 441.04} | -| [CUADUnlimitedAllYouCanEatLicenseLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 48} | {'test': 368.08} | -| [CUADVolumeRestrictionLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 322} | {'test': 306.27} | -| [CUADWarrantyDurationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 320} | {'test': 352.27} | -| [CanadaTaxCourtOutcomesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 244} | {'test': 622.6} | -| [CataloniaTweetClassification](https://aclanthology.org/2020.lrec-1.171/) | ['cat', 'spa'] | Classification | s2s | [Social, Government, Written] | {'validation': 2000, 'test': 2000} | {'validation': 202.61, 'test': 200.49} | -| [ClimateFEVER](https://www.sustainablefinance.uzh.ch/en/research/climate-fever.html) (Thomas Diggelmann, 2021) | ['eng'] | Retrieval | s2p | | None | {'test': {'average_document_length': 538.241873443325, 'average_query_length': 123.39934853420195, 'num_documents': 5416593, 'num_queries': 1535, 'average_relevant_docs_per_query': 3.0495114006514656}} | -| [ClimateFEVERHardNegatives](https://www.sustainablefinance.uzh.ch/en/research/climate-fever.html) (Thomas Diggelmann, 2021) | ['eng'] | Retrieval | s2p | | {'test': 1000} | {'test': {'average_document_length': 1245.4236333727013, 'average_query_length': 121.879, 'num_documents': 47416, 'num_queries': 1000, 'average_relevant_docs_per_query': 3.048}} | -| [CmedqaRetrieval](https://aclanthology.org/2022.emnlp-main.357.pdf) | ['cmn'] | Retrieval | s2p | | None | {'dev': {'average_document_length': 307.7710222897771, 'average_query_length': 48.470367591897976, 'num_documents': 100001, 'num_queries': 3999, 'average_relevant_docs_per_query': 1.86271567891973}} | +| [COIRCodeSearchNetRetrieval](https://huggingface.co/datasets/code_search_net/) (Husain et al., 2019) | ['go', 'java', 'javascript', 'php', 'python', 'ruby'] | Retrieval | p2p | [Programming, Written] | {'test': 1056326} | {'test': {'number_of_characters': 664.77, 'num_samples': 1056326, 'num_queries': 52561, 'num_documents': 1003765, 'average_document_length': 0.0, 'average_query_length': 0.01, 'average_relevant_docs_per_query': 1.0, 'hf_subset_descriptive_stats': {'python': {'number_of_characters': 941.4, 'num_samples': 295228, 'num_queries': 14918, 'num_documents': 280310, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'javascript': {'number_of_characters': 748.83, 'num_samples': 68145, 'num_queries': 3291, 'num_documents': 64854, 'average_document_length': 0.0, 'average_query_length': 0.23, 'average_relevant_docs_per_query': 1.0}, 'go': {'number_of_characters': 405.38, 'num_samples': 190562, 'num_queries': 8122, 'num_documents': 182440, 'average_document_length': 0.0, 'average_query_length': 0.05, 'average_relevant_docs_per_query': 1.0}, 'ruby': {'number_of_characters': 457.44, 'num_samples': 28831, 'num_queries': 1261, 'num_documents': 27570, 'average_document_length': 0.0, 'average_query_length': 0.36, 'average_relevant_docs_per_query': 1.0}, 'java': {'number_of_characters': 588.89, 'num_samples': 191821, 'num_queries': 10955, 'num_documents': 180866, 'average_document_length': 0.0, 'average_query_length': 0.05, 'average_relevant_docs_per_query': 1.0}, 'php': {'number_of_characters': 578.85, 'num_samples': 281739, 'num_queries': 14014, 'num_documents': 267725, 'average_document_length': 0.0, 'average_query_length': 0.04, 'average_relevant_docs_per_query': 1.0}}}} | +| [CPUSpeedTask](https://github.com/KennethEnevoldsen/scandinavian-embedding-benchmark/blob/c8376f967d1294419be1d3eb41217d04cd3a65d3/src/seb/registered_tasks/speed.py#L83-L96) | ['eng'] | Speed | s2s | [Fiction, Written] | None | None | +| [CQADupstackAndroidRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | None | +| [CQADupstackEnglishRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | None | +| [CQADupstackGamingRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | None | +| [CQADupstackGisRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | None | +| [CQADupstackMathematicaRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | None | +| [CQADupstackPhysicsRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | None | +| [CQADupstackProgrammersRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | [Programming, Written, Non-fiction] | None | None | +| [CQADupstackStatsRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | None | +| [CQADupstackTexRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | None | +| [CQADupstackUnixRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | None | +| [CQADupstackWebmastersRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | None | +| [CQADupstackWordpressRetrieval](http://nlp.cis.unimelb.edu.au/resources/cqadupstack/) (Hoogeveen et al., 2015) | ['eng'] | Retrieval | s2p | | None | None | +| [CSFDCZMovieReviewSentimentClassification](https://arxiv.org/abs/2304.01922) (Michal Štefánik, 2023) | ['ces'] | Classification | s2s | [Reviews, Written] | None | None | +| [CSFDSKMovieReviewSentimentClassification](https://arxiv.org/abs/2304.01922) (Michal Štefánik, 2023) | ['slk'] | Classification | s2s | [Reviews, Written] | None | None | +| [CTKFactsNLI](https://arxiv.org/abs/2201.11115) (Ullrich et al., 2023) | ['ces'] | PairClassification | s2s | [News, Written] | None | None | +| [CUADAffiliateLicenseLicenseeLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADAffiliateLicenseLicensorLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADAntiAssignmentLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADAuditRightsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADCapOnLiabilityLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADChangeOfControlLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADCompetitiveRestrictionExceptionLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADCovenantNotToSueLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADEffectiveDateLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADExclusivityLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADExpirationDateLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADGoverningLawLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADIPOwnershipAssignmentLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADInsuranceLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADIrrevocableOrPerpetualLicenseLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADJointIPOwnershipLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADLicenseGrantLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADLiquidatedDamagesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADMinimumCommitmentLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADMostFavoredNationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADNoSolicitOfCustomersLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADNoSolicitOfEmployeesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADNonCompeteLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADNonDisparagementLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADNonTransferableLicenseLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADNoticePeriodToTerminateRenewalLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADPostTerminationServicesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADPriceRestrictionsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADRenewalTermLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADRevenueProfitSharingLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADRofrRofoRofnLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADSourceCodeEscrowLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADTerminationForConvenienceLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADThirdPartyBeneficiaryLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADUncappedLiabilityLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADUnlimitedAllYouCanEatLicenseLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADVolumeRestrictionLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CUADWarrantyDurationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CanadaTaxCourtOutcomesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CataloniaTweetClassification](https://aclanthology.org/2020.lrec-1.171/) | ['cat', 'spa'] | Classification | s2s | [Social, Government, Written] | None | None | +| [ClimateFEVER](https://www.sustainablefinance.uzh.ch/en/research/climate-fever.html) (Thomas Diggelmann, 2021) | ['eng'] | Retrieval | s2p | | None | None | +| [ClimateFEVERHardNegatives](https://www.sustainablefinance.uzh.ch/en/research/climate-fever.html) (Thomas Diggelmann, 2021) | ['eng'] | Retrieval | s2p | | None | None | +| [CmedqaRetrieval](https://aclanthology.org/2022.emnlp-main.357.pdf) | ['cmn'] | Retrieval | s2p | | None | None | | [Cmnli](https://huggingface.co/datasets/clue/viewer/cmnli) | ['cmn'] | PairClassification | s2s | | None | None | -| [CodeEditSearchRetrieval](https://huggingface.co/datasets/cassanof/CodeEditSearch/viewer) (Niklas Muennighoff, 2023) | ['c', 'c++', 'go', 'java', 'javascript', 'php', 'python', 'ruby', 'rust', 'scala', 'shell', 'swift', 'typescript'] | Retrieval | p2p | [Programming, Written] | {'train': 13000} | {'train': {'python': {'average_document_length': 597.592, 'average_query_length': 69.519, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'javascript': {'average_document_length': 582.554, 'average_query_length': 56.88, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'typescript': {'average_document_length': 580.877, 'average_query_length': 60.092, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'go': {'average_document_length': 548.498, 'average_query_length': 70.797, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'ruby': {'average_document_length': 518.895, 'average_query_length': 66.9, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'java': {'average_document_length': 620.332, 'average_query_length': 62.984, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'php': {'average_document_length': 545.452, 'average_query_length': 61.927, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'c': {'average_document_length': 475.868, 'average_query_length': 97.588, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'c++': {'average_document_length': 544.446, 'average_query_length': 114.48, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'rust': {'average_document_length': 609.548, 'average_query_length': 67.503, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'swift': {'average_document_length': 574.62, 'average_query_length': 57.279, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'scala': {'average_document_length': 495.485, 'average_query_length': 64.833, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'shell': {'average_document_length': 486.519, 'average_query_length': 72.059, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}}} | -| [CodeFeedbackMT](https://arxiv.org/abs/2402.14658) (Tianyu Zheng, 2024) | ['eng'] | Retrieval | p2p | [Programming, Written] | {'test': 1000} | {'test': {'average_document_length': 1467.879728243677, 'average_query_length': 4425.522256533855, 'num_documents': 66383, 'num_queries': 13277, 'average_relevant_docs_per_query': 1.0}} | -| [CodeFeedbackST](https://arxiv.org/abs/2407.02883) (Xiangyang Li, 2024) | ['eng'] | Retrieval | p2p | [Programming, Written] | {'test': 1000} | {'test': {'average_document_length': 1521.3317148588733, 'average_query_length': 724.2441704465598, 'num_documents': 156526, 'num_queries': 31306, 'average_relevant_docs_per_query': 1.0}} | -| [CodeSearchNetCCRetrieval](https://arxiv.org/abs/2407.02883) (Xiangyang Li, 2024) | ['go', 'java', 'javascript', 'php', 'python', 'ruby'] | Retrieval | p2p | [Programming, Written] | {'test': 1000} | {'test': {'python': {'average_document_length': 388.31577184555965, 'average_query_length': 551.7934039415471, 'num_documents': 280652, 'num_queries': 14918, 'average_relevant_docs_per_query': 1.0}, 'javascript': {'average_document_length': 276.0730050152605, 'average_query_length': 443.70707991491946, 'num_documents': 65201, 'num_queries': 3291, 'average_relevant_docs_per_query': 1.0}, 'go': {'average_document_length': 185.0307932251621, 'average_query_length': 233.76803742920464, 'num_documents': 182735, 'num_queries': 8122, 'average_relevant_docs_per_query': 1.0}, 'ruby': {'average_document_length': 214.86204146730464, 'average_query_length': 266.8731165741475, 'num_documents': 27588, 'num_queries': 1261, 'average_relevant_docs_per_query': 1.0}, 'java': {'average_document_length': 281.96280259139183, 'average_query_length': 342.5341853035144, 'num_documents': 181061, 'num_queries': 10955, 'average_relevant_docs_per_query': 1.0}, 'php': {'average_document_length': 268.9752569556027, 'average_query_length': 336.62194947909234, 'num_documents': 268237, 'num_queries': 14014, 'average_relevant_docs_per_query': 1.0}}} | -| [CodeSearchNetRetrieval](https://huggingface.co/datasets/code_search_net/) (Husain et al., 2019) | ['go', 'java', 'javascript', 'php', 'python', 'ruby'] | Retrieval | p2p | [Programming, Written] | {'test': 1000} | {'test': {'python': {'average_document_length': 862.842, 'average_query_length': 466.546, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'javascript': {'average_document_length': 1415.632, 'average_query_length': 186.018, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'go': {'average_document_length': 563.729, 'average_query_length': 125.213, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'ruby': {'average_document_length': 577.634, 'average_query_length': 313.818, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'java': {'average_document_length': 420.287, 'average_query_length': 690.36, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}, 'php': {'average_document_length': 712.129, 'average_query_length': 162.119, 'num_documents': 1000, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}}} | -| [CodeTransOceanContest](https://arxiv.org/abs/2310.04951) (Weixiang Yan, 2023) | ['c++', 'python'] | Retrieval | p2p | [Programming, Written] | | {'test': {'average_document_length': 1528.9156746031747, 'average_query_length': 1012.1131221719457, 'num_documents': 1008, 'num_queries': 221, 'average_relevant_docs_per_query': 1.0}} | -| [CodeTransOceanDL](https://arxiv.org/abs/2310.04951) (Weixiang Yan, 2023) | ['python'] | Retrieval | p2p | [Programming, Written] | | {'test': {'average_document_length': 1479.0735294117646, 'average_query_length': 1867.6222222222223, 'num_documents': 816, 'num_queries': 180, 'average_relevant_docs_per_query': 1.0}} | -| [ContractNLIConfidentialityOfAgreementLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 82} | {'test': 473.17} | -| [ContractNLIExplicitIdentificationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 109} | {'test': 506.12} | -| [ContractNLIInclusionOfVerballyConveyedInformationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 139} | {'test': 525.75} | -| [ContractNLILimitedUseLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 208} | {'test': 407.51} | -| [ContractNLINoLicensingLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 162} | {'test': 419.42} | -| [ContractNLINoticeOnCompelledDisclosureLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 142} | {'test': 503.45} | -| [ContractNLIPermissibleAcquirementOfSimilarInformationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 178} | {'test': 427.4} | -| [ContractNLIPermissibleCopyLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 87} | {'test': 386.84} | -| [ContractNLIPermissibleDevelopmentOfSimilarInformationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 136} | {'test': 396.4} | -| [ContractNLIPermissiblePostAgreementPossessionLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 111} | {'test': 529.09} | -| [ContractNLIReturnOfConfidentialInformationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 66} | {'test': 478.29} | -| [ContractNLISharingWithEmployeesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 170} | {'test': 548.63} | -| [ContractNLISharingWithThirdPartiesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 180} | {'test': 517.29} | -| [ContractNLISurvivalOfObligationsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 157} | {'test': 417.64} | -| [Core17InstructionRetrieval](https://arxiv.org/abs/2403.15246) (Orion Weller, 2024) | ['eng'] | InstructionRetrieval | s2p | [News, Written] | {'eng': 39838} | {'test': {'num_docs': 19899, 'num_queries': 20, 'average_document_length': 2233.0329664807277, 'average_query_length': 109.75, 'average_instruction_length': 295.55, 'average_changed_instruction_length': 355.2, 'average_relevant_docs_per_query': 32.7, 'average_top_ranked_per_query': 1000.0}} | -| [CorporateLobbyingLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 490} | {'test': 6039.85} | -| [CosQA](https://arxiv.org/abs/2105.13239) (Junjie Huang, 2021) | ['eng', 'python'] | Retrieval | p2p | [Programming, Written] | | {'test': {'average_document_length': 276.132741215298, 'average_query_length': 36.814, 'num_documents': 20604, 'num_queries': 500, 'average_relevant_docs_per_query': 1.0}} | -| [CovidRetrieval](https://arxiv.org/abs/2203.03367) | ['cmn'] | Retrieval | s2p | | None | {'dev': {'average_document_length': 332.4152658473415, 'average_query_length': 25.9304531085353, 'num_documents': 100001, 'num_queries': 949, 'average_relevant_docs_per_query': 1.0105374077976819}} | -| [CrossLingualSemanticDiscriminationWMT19](https://huggingface.co/datasets/Andrianos/clsd_wmt19_21) | ['deu', 'fra'] | Retrieval | s2s | [News, Written] | {'test': 2946} | {'test': {'deu-fra': {'average_document_length': 147.49857433808555, 'average_query_length': 152.95587236931433, 'num_documents': 7365, 'num_queries': 1473, 'average_relevant_docs_per_query': 1.0}, 'fra-deu': {'average_document_length': 154.21968771215208, 'average_query_length': 145.877800407332, 'num_documents': 7365, 'num_queries': 1473, 'average_relevant_docs_per_query': 1.0}}} | -| [CrossLingualSemanticDiscriminationWMT21](https://huggingface.co/datasets/Andrianos/clsd_wmt19_21) | ['deu', 'fra'] | Retrieval | s2s | [News, Written] | {'test': 1786} | {'test': {'deu-fra': {'average_document_length': 177.26270996640537, 'average_query_length': 171.73012318029114, 'num_documents': 4465, 'num_queries': 893, 'average_relevant_docs_per_query': 1.0}, 'fra-deu': {'average_document_length': 174.45061590145576, 'average_query_length': 176.99216125419932, 'num_documents': 4465, 'num_queries': 893, 'average_relevant_docs_per_query': 1.0}}} | -| [CyrillicTurkicLangClassification](https://huggingface.co/datasets/tatiana-merz/cyrillic_turkic_langs) (Goldhahn et al., 2012) | ['bak', 'chv', 'kaz', 'kir', 'krc', 'rus', 'sah', 'tat', 'tyv'] | Classification | s2s | [Web, Written] | {'test': 2048} | {'test': 92.22} | -| [CzechProductReviewSentimentClassification](https://aclanthology.org/W13-1609/) | ['ces'] | Classification | s2s | [Reviews, Written] | {'test': 2048} | {'test': 153.26} | -| [CzechSoMeSentimentClassification](https://aclanthology.org/W13-1609/) | ['ces'] | Classification | s2s | [Reviews, Written] | {'test': 1000} | {'test': 59.89} | -| [CzechSubjectivityClassification](https://arxiv.org/abs/2009.08712) | ['ces'] | Classification | s2s | [Reviews, Written] | {'validation': 500, 'test': 2000} | {'validation': 108.2, 'test': 108.3} | -| [DBPedia](https://github.com/iai-group/DBpedia-Entity/) (Hasibi et al., 2017) | ['eng'] | Retrieval | s2p | [Written, Encyclopaedic] | None | {'test': {'average_document_length': 1122.7690155333814, 'average_query_length': 48.7264325323475, 'num_documents': 48605, 'num_queries': 541, 'average_relevant_docs_per_query': 1.3752310536044363}} | -| [DBPedia-PL](https://github.com/iai-group/DBpedia-Entity/) (Hasibi et al., 2017) | ['pol'] | Retrieval | s2p | [Written, Encyclopaedic] | None | {'test': {'average_document_length': 311.7007956561823, 'average_query_length': 35.45, 'num_documents': 4635922, 'num_queries': 400, 'average_relevant_docs_per_query': 38.215}} | -| [DBPedia-PLHardNegatives](https://github.com/iai-group/DBpedia-Entity/) (Hasibi et al., 2017) | ['pol'] | Retrieval | s2p | [Written, Encyclopaedic] | {'test': 400} | {'test': {'average_document_length': 363.468546000768, 'average_query_length': 35.45, 'num_documents': 88542, 'num_queries': 400, 'average_relevant_docs_per_query': 38.215}} | -| [DBPediaHardNegatives](https://github.com/iai-group/DBpedia-Entity/) (Hasibi et al., 2017) | ['eng'] | Retrieval | s2p | [Written, Encyclopaedic] | {'test': 400} | {'test': {'average_document_length': 338.58561119129564, 'average_query_length': 34.085, 'num_documents': 90070, 'num_queries': 400, 'average_relevant_docs_per_query': 38.215}} | -| [DBpediaClassification](https://arxiv.org/abs/1509.01626) (Zhang et al., 2015) | ['eng'] | Classification | s2s | [Encyclopaedic, Written] | {'test': 70000} | {'test': 281.4} | -| [DKHateClassification](https://aclanthology.org/2020.lrec-1.430/) | ['dan'] | Classification | s2s | [Social, Written] | {'test': 329} | {'test': 104.0} | -| [DalajClassification](https://spraakbanken.gu.se/en/resources/superlim) | ['swe'] | Classification | s2s | [Non-fiction, Written] | {'test': 444} | {'test': 243.8} | -| [DanFeverRetrieval](https://aclanthology.org/2021.nodalida-main.47/) | ['dan'] | Retrieval | p2p | [Encyclopaedic, Non-fiction, Spoken] | {'train': 8897} | {'train': {'average_document_length': 312.1117274167987, 'average_query_length': 50.26957476855484, 'num_documents': 2524, 'num_queries': 6373, 'average_relevant_docs_per_query': 0.48721167425074535}} | -| [DanishPoliticalCommentsClassification](https://huggingface.co/datasets/danish_political_comments) (Mads Guldborg Kjeldgaard Kongsbak, 2019) | ['dan'] | Classification | s2s | [Social, Written] | {'train': 9010} | {'train': 69.9} | -| [DefinitionClassificationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 1337} | {'test': 253.72} | -| [DiaBlaBitextMining](https://inria.hal.science/hal-03021633) (González et al., 2019) | ['eng', 'fra'] | BitextMining | s2s | [Social, Written] | {} | {} | -| [Diversity1LegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 300} | {'test': 103.21} | -| [Diversity2LegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 300} | {'test': 0} | -| [Diversity3LegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 300} | {'test': 135.46} | -| [Diversity4LegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 300} | {'test': 144.52} | -| [Diversity5LegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 300} | {'test': 174.77} | -| [Diversity6LegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 300} | {'test': 301.01} | -| [DuRetrieval](https://aclanthology.org/2022.emnlp-main.357.pdf) (Yifu Qiu, 2022) | ['cmn'] | Retrieval | s2p | | None | {'dev': {'average_document_length': 331.3219967800322, 'average_query_length': 9.289, 'num_documents': 100001, 'num_queries': 2000, 'average_relevant_docs_per_query': 4.9195}} | -| [DutchBookReviewSentimentClassification](https://github.com/benjaminvdb/DBRD) (Benjamin et al., 2019) | ['nld'] | Classification | s2s | [Reviews, Written] | {'test': 2224} | {'test': 1443.0} | -| [ESCIReranking](https://github.com/amazon-science/esci-data/) (Chandan K. Reddy, 2022) | ['eng', 'jpn', 'spa'] | Reranking | s2p | [Written] | | | -| [EcomRetrieval](https://arxiv.org/abs/2203.03367) | ['cmn'] | Retrieval | s2p | | None | {'dev': {'average_document_length': 32.98041664189015, 'average_query_length': 6.798, 'num_documents': 100902, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}} | -| [EightTagsClustering.v2](https://aclanthology.org/2020.lrec-1.207.pdf) | ['pol'] | Clustering | s2s | [Social, Written] | {'test': 2048} | {'test': 78.73} | -| [EmotionClassification](https://www.aclweb.org/anthology/D18-1404) | ['eng'] | Classification | s2s | [Social, Written] | {'validation': 2000, 'test': 2000} | {'validation': 95.3, 'test': 95.6} | -| [EstQA](https://www.semanticscholar.org/paper/Extractive-Question-Answering-for-Estonian-Language-182912IAPM-Alum%C3%A4e/ea4f60ab36cadca059c880678bc4c51e293a85d6?utm_source=direct_link) | ['est'] | Retrieval | s2p | [Encyclopaedic, Written] | {'test': 603} | {'test': {'average_document_length': 785.595041322314, 'average_query_length': 55.32006633499171, 'num_documents': 121, 'num_queries': 603, 'average_relevant_docs_per_query': 1.0}} | -| [EstonianValenceClassification](https://figshare.com/articles/dataset/Estonian_Valence_Corpus_Eesti_valentsikorpus/24517054) | ['est'] | Classification | s2s | [News, Written] | {'train': 3270, 'test': 818} | {'train': 226.70642201834863, 'test': 231.5085574572127} | -| [FEVER](https://fever.ai/) | ['eng'] | Retrieval | s2p | | None | {'train': {'average_document_length': 538.2340070317589, 'average_query_length': 47.56034058828886, 'num_documents': 5416568, 'num_queries': 109810, 'average_relevant_docs_per_query': 1.2757034878426372}, 'dev': {'average_document_length': 538.2340070317589, 'average_query_length': 47.326282628262824, 'num_documents': 5416568, 'num_queries': 6666, 'average_relevant_docs_per_query': 1.211971197119712}, 'test': {'average_document_length': 538.2340070317589, 'average_query_length': 49.60546054605461, 'num_documents': 5416568, 'num_queries': 6666, 'average_relevant_docs_per_query': 1.1906690669066906}} | -| [FEVERHardNegatives](https://fever.ai/) | ['eng'] | Retrieval | s2p | | {'test': 1000} | {'test': {'average_document_length': 695.4370242764114, 'average_query_length': 49.62, 'num_documents': 163698, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.171}} | -| [FQuADRetrieval](https://huggingface.co/datasets/manu/fquad2_test) | ['fra'] | Retrieval | s2p | [Encyclopaedic, Written] | {'test': 400, 'validation': 100} | {'test': {'average_document_length': 896.3308550185874, 'average_query_length': 58.52, 'num_documents': 269, 'num_queries': 400, 'average_relevant_docs_per_query': 1.0}, 'validation': {'average_document_length': 895.1340206185567, 'average_query_length': 54.13, 'num_documents': 97, 'num_queries': 100, 'average_relevant_docs_per_query': 1.0}} | -| [FaithDial](https://mcgill-nlp.github.io/FaithDial) (Dziri et al., 2022) | ['eng'] | Retrieval | s2p | [Encyclopaedic, Written] | {'test': 2042} | {'test': {'average_document_length': 140.61062447018932, 'average_query_length': 4.926542605288932, 'num_documents': 3539, 'num_queries': 2042, 'average_relevant_docs_per_query': 1.0}} | -| [FalseFriendsGermanEnglish](https://drive.google.com/file/d/1jgq0nBnV-UiYNxbKNrrr2gxDEHm-DMKH/view?usp=share_link) | ['deu'] | PairClassification | s2s | [Written] | {'test': 1524} | {'test': 40.3} | -| [FaroeseSTS](https://aclanthology.org/2023.nodalida-1.74.pdf) | ['fao'] | STS | s2s | [News, Web, Written] | {'train': 729} | {'train': 43.6} | -| [FarsTail](https://link.springer.com/article/10.1007/s00500-023-08959-3) (Amirkhani et al., 2023) | ['fas'] | PairClassification | s2s | [Academic, Written] | {'test': 1029} | {'test': 125.84} | -| [FeedbackQARetrieval](https://arxiv.org/abs/2204.03025) | ['eng'] | Retrieval | s2p | [Web, Government, Medical, Written] | {'test': 1992} | {'test': {'average_document_length': 1174.7986463620982, 'average_query_length': 72.33182730923694, 'num_documents': 2364, 'num_queries': 1992, 'average_relevant_docs_per_query': 1.0}} | -| [FiQA-PL](https://sites.google.com/view/fiqa/) (Nandan Thakur, 2021) | ['pol'] | Retrieval | s2p | | None | {'test': {'average_document_length': 795.2371699226205, 'average_query_length': 70.00771604938272, 'num_documents': 57638, 'num_queries': 648, 'average_relevant_docs_per_query': 2.632716049382716}} | -| [FiQA2018](https://sites.google.com/view/fiqa/) (Nandan Thakur, 2021) | ['eng'] | Retrieval | s2p | | None | {'train': {'average_document_length': 767.2108157812554, 'average_query_length': 61.49763636363636, 'num_documents': 57638, 'num_queries': 5500, 'average_relevant_docs_per_query': 2.5756363636363635}, 'dev': {'average_document_length': 767.2108157812554, 'average_query_length': 62.756, 'num_documents': 57638, 'num_queries': 500, 'average_relevant_docs_per_query': 2.476}, 'test': {'average_document_length': 767.2108157812554, 'average_query_length': 62.7037037037037, 'num_documents': 57638, 'num_queries': 648, 'average_relevant_docs_per_query': 2.632716049382716}} | -| [FilipinoHateSpeechClassification](https://pcj.csp.org.ph/index.php/pcj/issue/download/29/PCJ%20V14%20N1%20pp1-14%202019) (Neil Vicente Cabasag et al., 2019) | ['fil'] | Classification | s2s | [Social, Written] | {'validation': 2048, 'test': 2048} | {'validation': 88.1, 'test': 87.4} | -| [FilipinoShopeeReviewsClassification](https://uijrt.com/articles/v4/i8/UIJRTV4I80009.pdf) | ['fil'] | Classification | s2s | [Social, Written] | {'validation': 2250, 'test': 2250} | {'validation': 143.8, 'test': 145.1} | -| [FinParaSTS](https://huggingface.co/datasets/TurkuNLP/turku_paraphrase_corpus) | ['fin'] | STS | s2s | [News, Subtitles, Written] | {'test': 1000, 'validation': 1000} | {'test': 59.0, 'validation': 58.8} | -| [FinToxicityClassification](https://aclanthology.org/2023.nodalida-1.68) | ['fin'] | Classification | s2s | [News, Written] | {'train': 2048, 'test': 2048} | {'train': 432.63, 'test': 401.03} | -| [FinancialPhrasebankClassification](https://arxiv.org/abs/1307.5336) (P. Malo, 2014) | ['eng'] | Classification | s2s | [News, Written] | {'train': 4840} | {'train': 121.96} | -| [FloresBitextMining](https://huggingface.co/datasets/facebook/flores) (Goyal et al., 2022) | ['ace', 'acm', 'acq', 'aeb', 'afr', 'ajp', 'aka', 'als', 'amh', 'apc', 'arb', 'ars', 'ary', 'arz', 'asm', 'ast', 'awa', 'ayr', 'azb', 'azj', 'bak', 'bam', 'ban', 'bel', 'bem', 'ben', 'bho', 'bjn', 'bod', 'bos', 'bug', 'bul', 'cat', 'ceb', 'ces', 'cjk', 'ckb', 'crh', 'cym', 'dan', 'deu', 'dik', 'dyu', 'dzo', 'ell', 'eng', 'epo', 'est', 'eus', 'ewe', 'fao', 'fij', 'fin', 'fon', 'fra', 'fur', 'fuv', 'gaz', 'gla', 'gle', 'glg', 'grn', 'guj', 'hat', 'hau', 'heb', 'hin', 'hne', 'hrv', 'hun', 'hye', 'ibo', 'ilo', 'ind', 'isl', 'ita', 'jav', 'jpn', 'kab', 'kac', 'kam', 'kan', 'kas', 'kat', 'kaz', 'kbp', 'kea', 'khk', 'khm', 'kik', 'kin', 'kir', 'kmb', 'kmr', 'knc', 'kon', 'kor', 'lao', 'lij', 'lim', 'lin', 'lit', 'lmo', 'ltg', 'ltz', 'lua', 'lug', 'luo', 'lus', 'lvs', 'mag', 'mai', 'mal', 'mar', 'min', 'mkd', 'mlt', 'mni', 'mos', 'mri', 'mya', 'nld', 'nno', 'nob', 'npi', 'nso', 'nus', 'nya', 'oci', 'ory', 'pag', 'pan', 'pap', 'pbt', 'pes', 'plt', 'pol', 'por', 'prs', 'quy', 'ron', 'run', 'rus', 'sag', 'san', 'sat', 'scn', 'shn', 'sin', 'slk', 'slv', 'smo', 'sna', 'snd', 'som', 'sot', 'spa', 'srd', 'srp', 'ssw', 'sun', 'swe', 'swh', 'szl', 'tam', 'taq', 'tat', 'tel', 'tgk', 'tgl', 'tha', 'tir', 'tpi', 'tsn', 'tso', 'tuk', 'tum', 'tur', 'twi', 'tzm', 'uig', 'ukr', 'umb', 'urd', 'uzn', 'vec', 'vie', 'war', 'wol', 'xho', 'ydd', 'yor', 'yue', 'zho', 'zsm', 'zul'] | BitextMining | s2s | [Non-fiction, Encyclopaedic, Written] | {'dev': 997, 'devtest': 1012} | {} | -| [FrenchBookReviews](https://huggingface.co/datasets/Abirate/french_book_reviews) | ['fra'] | Classification | s2s | [Reviews, Written] | {'train': 2048} | {'train': 311.5} | -| [FrenkEnClassification](https://arxiv.org/abs/1906.02045) (Nikola Ljubešić, 2019) | ['eng'] | Classification | s2s | [Social, Written] | {'test': 2300} | {'test': 188.75} | -| [FrenkHrClassification](https://arxiv.org/abs/1906.02045) (Nikola Ljubešić, 2019) | ['hrv'] | Classification | s2s | [Social, Written] | {'test': 2120} | {'test': 89.86} | -| [FrenkSlClassification](https://arxiv.org/pdf/1906.02045) (Nikola Ljubešić, 2019) | ['slv'] | Classification | s2s | [Social, Written] | {'test': 2177} | {'test': 136.61} | -| [FunctionOfDecisionSectionLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 367} | {'test': 551.07} | -| [GPUSpeedTask](https://github.com/KennethEnevoldsen/scandinavian-embedding-benchmark/blob/c8376f967d1294419be1d3eb41217d04cd3a65d3/src/seb/registered_tasks/speed.py#L83-L96) | ['eng'] | Speed | s2s | [Fiction, Written] | {'test': 1} | {'test': 3591} | -| [GeoreviewClassification](https://github.com/yandex/geo-reviews-dataset-2023) | ['rus'] | Classification | p2p | [Reviews, Written] | {'test': 2048} | {'test': 409.0} | -| [GeoreviewClusteringP2P](https://github.com/yandex/geo-reviews-dataset-2023) | ['rus'] | Clustering | p2p | [Reviews, Written] | {'test': 2000} | {'test': 384.5} | -| [GeorgianFAQRetrieval](https://huggingface.co/datasets/jupyterjazz/georgian-faq) | ['kat'] | Retrieval | s2p | [Web, Written] | {'test': 2566} | {'test': {'average_document_length': 511.24668745128605, 'average_query_length': 61.69551656920078, 'num_documents': 2566, 'num_queries': 2565, 'average_relevant_docs_per_query': 1.0003898635477584}} | -| [GerDaLIR](https://github.com/lavis-nlp/GerDaLIR) | ['deu'] | Retrieval | s2p | | None | {'test': {'average_document_length': 15483.237726805888, 'average_query_length': 1027.3495690356156, 'num_documents': 131445, 'num_queries': 12298, 'average_relevant_docs_per_query': 1.1704342169458448}} | -| [GerDaLIRSmall](https://github.com/lavis-nlp/GerDaLIR) | ['deu'] | Retrieval | p2p | [Legal, Written] | None | {'test': {'average_document_length': 19706.823653325308, 'average_query_length': 1031.0680889324833, 'num_documents': 9969, 'num_queries': 12234, 'average_relevant_docs_per_query': 1.1705084191597188}} | -| [GermanDPR](https://huggingface.co/datasets/deepset/germandpr) (Timo Möller, 2021) | ['deu'] | Retrieval | s2p | | None | {'test': {'average_document_length': 1288.3410987482614, 'average_query_length': 64.38439024390244, 'num_documents': 2876, 'num_queries': 1025, 'average_relevant_docs_per_query': 1.0}} | -| [GermanGovServiceRetrieval](https://huggingface.co/datasets/it-at-m/LHM-Dienstleistungen-QA) | ['deu'] | Retrieval | s2p | [Government, Written] | {'test': 357} | {'test': {'average_document_length': 1246.4571428571428, 'average_query_length': 68.17977528089888, 'num_documents': 105, 'num_queries': 356, 'average_relevant_docs_per_query': 1.0}} | -| [GermanPoliticiansTwitterSentimentClassification](https://aclanthology.org/2022.konvens-1.9) | ['deu'] | Classification | s2s | [Social, Government, Written] | {'test': 357} | {'test': 302.48} | -| [GermanQuAD-Retrieval](https://www.kaggle.com/datasets/GermanQuAD) (Timo Möller, 2021) | ['deu'] | Retrieval | s2p | | None | {'test': {'average_document_length': 1941.090717299578, 'average_query_length': 56.74773139745916, 'num_documents': 474, 'num_queries': 2204, 'average_relevant_docs_per_query': 1.0}} | +| [CodeEditSearchRetrieval](https://huggingface.co/datasets/cassanof/CodeEditSearch/viewer) (Niklas Muennighoff, 2023) | ['c', 'c++', 'go', 'java', 'javascript', 'php', 'python', 'ruby', 'rust', 'scala', 'shell', 'swift', 'typescript'] | Retrieval | p2p | [Programming, Written] | {'train': 26000} | {'train': {'number_of_characters': 71.99, 'num_samples': 26000, 'num_queries': 13000, 'num_documents': 13000, 'average_document_length': 0.0, 'average_query_length': 0.01, 'average_relevant_docs_per_query': 1.0, 'hf_subset_descriptive_stats': {'python': {'number_of_characters': 70.52, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'javascript': {'number_of_characters': 57.88, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'typescript': {'number_of_characters': 61.09, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'go': {'number_of_characters': 71.8, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'ruby': {'number_of_characters': 67.9, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'java': {'number_of_characters': 63.98, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'php': {'number_of_characters': 62.93, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'c': {'number_of_characters': 98.59, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.1, 'average_relevant_docs_per_query': 1.0}, 'c++': {'number_of_characters': 115.48, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.11, 'average_relevant_docs_per_query': 1.0}, 'rust': {'number_of_characters': 68.5, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}, 'swift': {'number_of_characters': 58.28, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'scala': {'number_of_characters': 65.83, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.06, 'average_relevant_docs_per_query': 1.0}, 'shell': {'number_of_characters': 73.06, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}}}} | +| [CodeFeedbackMT](https://arxiv.org/abs/2402.14658) (Tianyu Zheng, 2024) | ['eng'] | Retrieval | p2p | [Programming, Written] | {'test': 79660} | {'test': {'number_of_characters': 5894.4, 'num_samples': 79660, 'num_queries': 13277, 'num_documents': 66383, 'average_document_length': 0.02, 'average_query_length': 0.33, 'average_relevant_docs_per_query': 1.0}} | +| [CodeFeedbackST](https://arxiv.org/abs/2407.02883) (Xiangyang Li, 2024) | ['eng'] | Retrieval | p2p | [Programming, Written] | {'test': 187832} | {'test': {'number_of_characters': 2246.58, 'num_samples': 187832, 'num_queries': 31306, 'num_documents': 156526, 'average_document_length': 0.01, 'average_query_length': 0.02, 'average_relevant_docs_per_query': 1.0}} | +| [CodeSearchNetCCRetrieval](https://arxiv.org/abs/2407.02883) (Xiangyang Li, 2024) | ['go', 'java', 'javascript', 'php', 'python', 'ruby'] | Retrieval | p2p | [Programming, Written] | {'test': 1058035} | {'test': {'number_of_characters': 390.06, 'num_samples': 1058035, 'num_queries': 52561, 'num_documents': 1005474, 'average_document_length': 0.0, 'average_query_length': 0.01, 'average_relevant_docs_per_query': 1.0, 'hf_subset_descriptive_stats': {'python': {'number_of_characters': 553.79, 'num_samples': 295570, 'num_queries': 14918, 'num_documents': 280652, 'average_document_length': 0.0, 'average_query_length': 0.04, 'average_relevant_docs_per_query': 1.0}, 'javascript': {'number_of_characters': 445.71, 'num_samples': 68492, 'num_queries': 3291, 'num_documents': 65201, 'average_document_length': 0.0, 'average_query_length': 0.13, 'average_relevant_docs_per_query': 1.0}, 'go': {'number_of_characters': 235.77, 'num_samples': 190857, 'num_queries': 8122, 'num_documents': 182735, 'average_document_length': 0.0, 'average_query_length': 0.03, 'average_relevant_docs_per_query': 1.0}, 'ruby': {'number_of_characters': 268.87, 'num_samples': 28849, 'num_queries': 1261, 'num_documents': 27588, 'average_document_length': 0.0, 'average_query_length': 0.21, 'average_relevant_docs_per_query': 1.0}, 'java': {'number_of_characters': 344.53, 'num_samples': 192016, 'num_queries': 10955, 'num_documents': 181061, 'average_document_length': 0.0, 'average_query_length': 0.03, 'average_relevant_docs_per_query': 1.0}, 'php': {'number_of_characters': 338.62, 'num_samples': 282251, 'num_queries': 14014, 'num_documents': 268237, 'average_document_length': 0.0, 'average_query_length': 0.02, 'average_relevant_docs_per_query': 1.0}}}} | +| [CodeSearchNetRetrieval](https://huggingface.co/datasets/code_search_net/) (Husain et al., 2019) | ['go', 'java', 'javascript', 'php', 'python', 'ruby'] | Retrieval | p2p | [Programming, Written] | {'test': 12000} | {'test': {'number_of_characters': 325.01, 'num_samples': 12000, 'num_queries': 6000, 'num_documents': 6000, 'average_document_length': 0.0, 'average_query_length': 0.05, 'average_relevant_docs_per_query': 1.0, 'hf_subset_descriptive_stats': {'python': {'number_of_characters': 467.55, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.47, 'average_relevant_docs_per_query': 1.0}, 'javascript': {'number_of_characters': 187.02, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.19, 'average_relevant_docs_per_query': 1.0}, 'go': {'number_of_characters': 126.21, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.13, 'average_relevant_docs_per_query': 1.0}, 'ruby': {'number_of_characters': 314.82, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.31, 'average_relevant_docs_per_query': 1.0}, 'java': {'number_of_characters': 691.36, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.69, 'average_relevant_docs_per_query': 1.0}, 'php': {'number_of_characters': 163.12, 'num_samples': 2000, 'num_queries': 1000, 'num_documents': 1000, 'average_document_length': 0.0, 'average_query_length': 0.16, 'average_relevant_docs_per_query': 1.0}}}} | +| [CodeTransOceanContest](https://arxiv.org/abs/2310.04951) (Weixiang Yan, 2023) | ['c++', 'python'] | Retrieval | p2p | [Programming, Written] | {'test': 1229} | {'test': {'number_of_characters': 2520.65, 'num_samples': 1229, 'num_queries': 221, 'num_documents': 1008, 'average_document_length': 1.5, 'average_query_length': 4.58, 'average_relevant_docs_per_query': 1.0}} | +| [CodeTransOceanDL](https://arxiv.org/abs/2310.04951) (Weixiang Yan, 2023) | ['python'] | Retrieval | p2p | [Programming, Written] | {'test': 996} | {'test': {'number_of_characters': 3347.7, 'num_samples': 996, 'num_queries': 180, 'num_documents': 816, 'average_document_length': 1.81, 'average_query_length': 10.38, 'average_relevant_docs_per_query': 1.0}} | +| [ContractNLIConfidentialityOfAgreementLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [ContractNLIExplicitIdentificationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [ContractNLIInclusionOfVerballyConveyedInformationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [ContractNLILimitedUseLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [ContractNLINoLicensingLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [ContractNLINoticeOnCompelledDisclosureLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [ContractNLIPermissibleAcquirementOfSimilarInformationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [ContractNLIPermissibleCopyLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [ContractNLIPermissibleDevelopmentOfSimilarInformationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [ContractNLIPermissiblePostAgreementPossessionLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [ContractNLIReturnOfConfidentialInformationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [ContractNLISharingWithEmployeesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [ContractNLISharingWithThirdPartiesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [ContractNLISurvivalOfObligationsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [Core17InstructionRetrieval](https://arxiv.org/abs/2403.15246) (Orion Weller, 2024) | ['eng'] | InstructionRetrieval | s2p | [News, Written] | {'test': 19919} | {'test': {'num_samples': 19919, 'num_docs': 19899, 'num_queries': 20, 'number_of_characters': 44450333, 'average_document_length': 2233.03, 'average_query_length': 109.75, 'average_instruction_length': 295.55, 'average_changed_instruction_length': 355.2, 'average_relevant_docs_per_query': 32.7, 'average_top_ranked_per_query': 1000.0}} | +| [CorporateLobbyingLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [CosQA](https://arxiv.org/abs/2105.13239) (Junjie Huang, 2021) | ['eng', 'python'] | Retrieval | p2p | [Programming, Written] | {'test': 21104} | {'test': {'number_of_characters': 313.95, 'num_samples': 21104, 'num_queries': 500, 'num_documents': 20604, 'average_document_length': 0.01, 'average_query_length': 0.07, 'average_relevant_docs_per_query': 1.0}} | +| [CovidRetrieval](https://arxiv.org/abs/2203.03367) | ['cmn'] | Retrieval | s2p | | None | None | +| [CrossLingualSemanticDiscriminationWMT19](https://huggingface.co/datasets/Andrianos/clsd_wmt19_21) | ['deu', 'fra'] | Retrieval | s2s | [News, Written] | None | None | +| [CrossLingualSemanticDiscriminationWMT21](https://huggingface.co/datasets/Andrianos/clsd_wmt19_21) | ['deu', 'fra'] | Retrieval | s2s | [News, Written] | None | None | +| [CyrillicTurkicLangClassification](https://huggingface.co/datasets/tatiana-merz/cyrillic_turkic_langs) (Goldhahn et al., 2012) | ['bak', 'chv', 'kaz', 'kir', 'krc', 'rus', 'sah', 'tat', 'tyv'] | Classification | s2s | [Web, Written] | None | None | +| [CzechProductReviewSentimentClassification](https://aclanthology.org/W13-1609/) | ['ces'] | Classification | s2s | [Reviews, Written] | None | None | +| [CzechSoMeSentimentClassification](https://aclanthology.org/W13-1609/) | ['ces'] | Classification | s2s | [Reviews, Written] | None | None | +| [CzechSubjectivityClassification](https://arxiv.org/abs/2009.08712) | ['ces'] | Classification | s2s | [Reviews, Written] | None | None | +| [DBPedia](https://github.com/iai-group/DBpedia-Entity/) (Hasibi et al., 2017) | ['eng'] | Retrieval | s2p | [Written, Encyclopaedic] | None | None | +| [DBPedia-PL](https://github.com/iai-group/DBpedia-Entity/) (Hasibi et al., 2017) | ['pol'] | Retrieval | s2p | [Written, Encyclopaedic] | None | None | +| [DBPedia-PLHardNegatives](https://github.com/iai-group/DBpedia-Entity/) (Hasibi et al., 2017) | ['pol'] | Retrieval | s2p | [Written, Encyclopaedic] | None | None | +| [DBPediaHardNegatives](https://github.com/iai-group/DBpedia-Entity/) (Hasibi et al., 2017) | ['eng'] | Retrieval | s2p | [Written, Encyclopaedic] | None | None | +| [DBpediaClassification](https://arxiv.org/abs/1509.01626) (Zhang et al., 2015) | ['eng'] | Classification | s2s | [Encyclopaedic, Written] | None | None | +| [DKHateClassification](https://aclanthology.org/2020.lrec-1.430/) | ['dan'] | Classification | s2s | [Social, Written] | None | None | +| [DalajClassification](https://spraakbanken.gu.se/en/resources/superlim) | ['swe'] | Classification | s2s | [Non-fiction, Written] | None | None | +| [DanFeverRetrieval](https://aclanthology.org/2021.nodalida-main.47/) | ['dan'] | Retrieval | p2p | [Encyclopaedic, Non-fiction, Spoken] | None | None | +| [DanishPoliticalCommentsClassification](https://huggingface.co/datasets/danish_political_comments) (Mads Guldborg Kjeldgaard Kongsbak, 2019) | ['dan'] | Classification | s2s | [Social, Written] | None | None | +| [DefinitionClassificationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [DiaBlaBitextMining](https://inria.hal.science/hal-03021633) (González et al., 2019) | ['eng', 'fra'] | BitextMining | s2s | [Social, Written] | None | None | +| [Diversity1LegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [Diversity2LegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [Diversity3LegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [Diversity4LegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [Diversity5LegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [Diversity6LegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [DuRetrieval](https://aclanthology.org/2022.emnlp-main.357.pdf) (Yifu Qiu, 2022) | ['cmn'] | Retrieval | s2p | | None | None | +| [DutchBookReviewSentimentClassification](https://github.com/benjaminvdb/DBRD) (Benjamin et al., 2019) | ['nld'] | Classification | s2s | [Reviews, Written] | None | None | +| [ESCIReranking](https://github.com/amazon-science/esci-data/) (Chandan K. Reddy, 2022) | ['eng', 'jpn', 'spa'] | Reranking | s2p | [Written] | {'test': 29285} | {'test': {'num_samples': 29285, 'number_of_characters': 254538331, 'num_positive': 271416, 'num_negative': 44235, 'avg_query_len': 19.69, 'avg_positive_len': 803.92, 'avg_negative_len': 808.5, 'hf_subset_descriptive_stats': {'us': {'num_samples': 21296, 'number_of_characters': 186915609, 'num_positive': 189375, 'num_negative': 25463, 'avg_query_len': 21.44, 'avg_positive_len': 868.37, 'avg_negative_len': 864.45}, 'es': {'num_samples': 3703, 'number_of_characters': 48861389, 'num_positive': 39110, 'num_negative': 10183, 'avg_query_len': 20.68, 'avg_positive_len': 980.96, 'avg_negative_len': 1023.22}, 'jp': {'num_samples': 4286, 'number_of_characters': 18761333, 'num_positive': 42931, 'num_negative': 8589, 'avg_query_len': 10.15, 'avg_positive_len': 358.36, 'avg_negative_len': 388.08}}}} | +| [EcomRetrieval](https://arxiv.org/abs/2203.03367) | ['cmn'] | Retrieval | s2p | | None | None | +| [EightTagsClustering.v2](https://aclanthology.org/2020.lrec-1.207.pdf) | ['pol'] | Clustering | s2s | [Social, Written] | None | None | +| [EmotionClassification](https://www.aclweb.org/anthology/D18-1404) | ['eng'] | Classification | s2s | [Social, Written] | None | None | +| [EstQA](https://www.semanticscholar.org/paper/Extractive-Question-Answering-for-Estonian-Language-182912IAPM-Alum%C3%A4e/ea4f60ab36cadca059c880678bc4c51e293a85d6?utm_source=direct_link) | ['est'] | Retrieval | s2p | [Encyclopaedic, Written] | None | None | +| [EstonianValenceClassification](https://figshare.com/articles/dataset/Estonian_Valence_Corpus_Eesti_valentsikorpus/24517054) | ['est'] | Classification | s2s | [News, Written] | None | None | +| [FEVER](https://fever.ai/) | ['eng'] | Retrieval | s2p | | None | None | +| [FEVERHardNegatives](https://fever.ai/) | ['eng'] | Retrieval | s2p | | None | None | +| [FQuADRetrieval](https://huggingface.co/datasets/manu/fquad2_test) | ['fra'] | Retrieval | s2p | [Encyclopaedic, Written] | None | None | +| [FaithDial](https://mcgill-nlp.github.io/FaithDial) (Dziri et al., 2022) | ['eng'] | Retrieval | s2p | [Encyclopaedic, Written] | None | None | +| [FalseFriendsGermanEnglish](https://drive.google.com/file/d/1jgq0nBnV-UiYNxbKNrrr2gxDEHm-DMKH/view?usp=share_link) | ['deu'] | PairClassification | s2s | [Written] | None | None | +| [FaroeseSTS](https://aclanthology.org/2023.nodalida-1.74.pdf) | ['fao'] | STS | s2s | [News, Web, Written] | None | None | +| [FarsTail](https://link.springer.com/article/10.1007/s00500-023-08959-3) (Amirkhani et al., 2023) | ['fas'] | PairClassification | s2s | [Academic, Written] | None | None | +| [FeedbackQARetrieval](https://arxiv.org/abs/2204.03025) | ['eng'] | Retrieval | s2p | [Web, Government, Medical, Written] | None | None | +| [FiQA-PL](https://sites.google.com/view/fiqa/) (Nandan Thakur, 2021) | ['pol'] | Retrieval | s2p | | None | None | +| [FiQA2018](https://sites.google.com/view/fiqa/) (Nandan Thakur, 2021) | ['eng'] | Retrieval | s2p | | None | None | +| [FilipinoHateSpeechClassification](https://pcj.csp.org.ph/index.php/pcj/issue/download/29/PCJ%20V14%20N1%20pp1-14%202019) (Neil Vicente Cabasag et al., 2019) | ['fil'] | Classification | s2s | [Social, Written] | None | None | +| [FilipinoShopeeReviewsClassification](https://uijrt.com/articles/v4/i8/UIJRTV4I80009.pdf) | ['fil'] | Classification | s2s | [Social, Written] | None | None | +| [FinParaSTS](https://huggingface.co/datasets/TurkuNLP/turku_paraphrase_corpus) | ['fin'] | STS | s2s | [News, Subtitles, Written] | None | None | +| [FinToxicityClassification](https://aclanthology.org/2023.nodalida-1.68) | ['fin'] | Classification | s2s | [News, Written] | None | None | +| [FinancialPhrasebankClassification](https://arxiv.org/abs/1307.5336) (P. Malo, 2014) | ['eng'] | Classification | s2s | [News, Written] | None | None | +| [FloresBitextMining](https://huggingface.co/datasets/facebook/flores) (Goyal et al., 2022) | ['ace', 'acm', 'acq', 'aeb', 'afr', 'ajp', 'aka', 'als', 'amh', 'apc', 'arb', 'ars', 'ary', 'arz', 'asm', 'ast', 'awa', 'ayr', 'azb', 'azj', 'bak', 'bam', 'ban', 'bel', 'bem', 'ben', 'bho', 'bjn', 'bod', 'bos', 'bug', 'bul', 'cat', 'ceb', 'ces', 'cjk', 'ckb', 'crh', 'cym', 'dan', 'deu', 'dik', 'dyu', 'dzo', 'ell', 'eng', 'epo', 'est', 'eus', 'ewe', 'fao', 'fij', 'fin', 'fon', 'fra', 'fur', 'fuv', 'gaz', 'gla', 'gle', 'glg', 'grn', 'guj', 'hat', 'hau', 'heb', 'hin', 'hne', 'hrv', 'hun', 'hye', 'ibo', 'ilo', 'ind', 'isl', 'ita', 'jav', 'jpn', 'kab', 'kac', 'kam', 'kan', 'kas', 'kat', 'kaz', 'kbp', 'kea', 'khk', 'khm', 'kik', 'kin', 'kir', 'kmb', 'kmr', 'knc', 'kon', 'kor', 'lao', 'lij', 'lim', 'lin', 'lit', 'lmo', 'ltg', 'ltz', 'lua', 'lug', 'luo', 'lus', 'lvs', 'mag', 'mai', 'mal', 'mar', 'min', 'mkd', 'mlt', 'mni', 'mos', 'mri', 'mya', 'nld', 'nno', 'nob', 'npi', 'nso', 'nus', 'nya', 'oci', 'ory', 'pag', 'pan', 'pap', 'pbt', 'pes', 'plt', 'pol', 'por', 'prs', 'quy', 'ron', 'run', 'rus', 'sag', 'san', 'sat', 'scn', 'shn', 'sin', 'slk', 'slv', 'smo', 'sna', 'snd', 'som', 'sot', 'spa', 'srd', 'srp', 'ssw', 'sun', 'swe', 'swh', 'szl', 'tam', 'taq', 'tat', 'tel', 'tgk', 'tgl', 'tha', 'tir', 'tpi', 'tsn', 'tso', 'tuk', 'tum', 'tur', 'twi', 'tzm', 'uig', 'ukr', 'umb', 'urd', 'uzn', 'vec', 'vie', 'war', 'wol', 'xho', 'ydd', 'yor', 'yue', 'zho', 'zsm', 'zul'] | BitextMining | s2s | [Non-fiction, Encyclopaedic, Written] | None | None | +| [FrenchBookReviews](https://huggingface.co/datasets/Abirate/french_book_reviews) | ['fra'] | Classification | s2s | [Reviews, Written] | None | None | +| [FrenkEnClassification](https://arxiv.org/abs/1906.02045) (Nikola Ljubešić, 2019) | ['eng'] | Classification | s2s | [Social, Written] | None | None | +| [FrenkHrClassification](https://arxiv.org/abs/1906.02045) (Nikola Ljubešić, 2019) | ['hrv'] | Classification | s2s | [Social, Written] | None | None | +| [FrenkSlClassification](https://arxiv.org/pdf/1906.02045) (Nikola Ljubešić, 2019) | ['slv'] | Classification | s2s | [Social, Written] | None | None | +| [FunctionOfDecisionSectionLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [GPUSpeedTask](https://github.com/KennethEnevoldsen/scandinavian-embedding-benchmark/blob/c8376f967d1294419be1d3eb41217d04cd3a65d3/src/seb/registered_tasks/speed.py#L83-L96) | ['eng'] | Speed | s2s | [Fiction, Written] | None | None | +| [GeoreviewClassification](https://github.com/yandex/geo-reviews-dataset-2023) | ['rus'] | Classification | p2p | [Reviews, Written] | None | None | +| [GeoreviewClusteringP2P](https://github.com/yandex/geo-reviews-dataset-2023) | ['rus'] | Clustering | p2p | [Reviews, Written] | None | None | +| [GeorgianFAQRetrieval](https://huggingface.co/datasets/jupyterjazz/georgian-faq) | ['kat'] | Retrieval | s2p | [Web, Written] | None | None | +| [GerDaLIR](https://github.com/lavis-nlp/GerDaLIR) | ['deu'] | Retrieval | s2p | | None | None | +| [GerDaLIRSmall](https://github.com/lavis-nlp/GerDaLIR) | ['deu'] | Retrieval | p2p | [Legal, Written] | None | None | +| [GermanDPR](https://huggingface.co/datasets/deepset/germandpr) (Timo Möller, 2021) | ['deu'] | Retrieval | s2p | | None | None | +| [GermanGovServiceRetrieval](https://huggingface.co/datasets/it-at-m/LHM-Dienstleistungen-QA) | ['deu'] | Retrieval | s2p | [Government, Written] | None | None | +| [GermanPoliticiansTwitterSentimentClassification](https://aclanthology.org/2022.konvens-1.9) | ['deu'] | Classification | s2s | [Social, Government, Written] | None | None | +| [GermanQuAD-Retrieval](https://www.kaggle.com/datasets/GermanQuAD) (Timo Möller, 2021) | ['deu'] | Retrieval | s2p | | None | None | | [GermanSTSBenchmark](https://github.com/t-systems-on-site-services-gmbh/german-STSbenchmark) (Philip May, 2021) | ['deu'] | STS | s2s | | None | None | -| [GreekCivicsQA](https://huggingface.co/datasets/antoinelb7/alloprof) | ['ell'] | Retrieval | s2p | [Academic, Written] | {'default': 407} | {'default': {'average_document_length': 1074.894348894349, 'average_query_length': 77.06142506142506, 'num_documents': 407, 'num_queries': 407, 'average_relevant_docs_per_query': 1.0}} | -| [GreekLegalCodeClassification](https://arxiv.org/abs/2109.15298) | ['ell'] | Classification | s2s | [Legal, Written] | {'validation': 2048, 'test': 2048} | {'validation': 4046.8, 'test': 4200.8} | -| [GujaratiNewsClassification](https://github.com/goru001/nlp-for-gujarati) | ['guj'] | Classification | s2s | [News, Written] | {'train': 5269, 'test': 1318} | {'train': 61.95, 'test': 61.91} | -| [HALClusteringS2S.v2](https://huggingface.co/datasets/lyon-nlp/clustering-hal-s2s) (Mathieu Ciancone, 2024) | ['fra'] | Clustering | s2s | [Academic, Written] | {'test': 2048} | {'test': 86.6} | -| [HagridRetrieval](https://github.com/project-miracl/hagrid) (Ehsan Kamalloo, 2023) | ['eng'] | Retrieval | s2p | [Encyclopaedic, Written] | {'train': 1922} | {'dev': {'average_document_length': 228.36693548387098, 'average_query_length': 40.064516129032256, 'num_documents': 496, 'num_queries': 496, 'average_relevant_docs_per_query': 1.0}} | -| [HateSpeechPortugueseClassification](https://aclanthology.org/W19-3510) | ['por'] | Classification | s2s | [Social, Written] | {'train': 2048} | {'train': 101.02} | -| [HeadlineClassification](https://aclanthology.org/2020.ngt-1.6/) | ['rus'] | Classification | s2s | [News, Written] | {'test': 2048} | {'test': 61.6} | -| [HebrewSentimentAnalysis](https://huggingface.co/datasets/hebrew_sentiment) | ['heb'] | Classification | s2s | [Reviews, Written] | {'test': 2048} | {'test': 113.57} | -| [HellaSwag](https://rowanzellers.com/hellaswag/) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | {'test': 10042} | {'test': {'average_document_length': 137.36519014671472, 'average_query_length': 224.53654650468033, 'num_documents': 199162, 'num_queries': 10042, 'average_relevant_docs_per_query': 1.0}} | -| [HinDialectClassification](https://lindat.mff.cuni.cz/repository/xmlui/handle/11234/1-4839) (Bafna et al., 2022) | ['anp', 'awa', 'ben', 'bgc', 'bhb', 'bhd', 'bho', 'bjj', 'bns', 'bra', 'gbm', 'guj', 'hne', 'kfg', 'kfy', 'mag', 'mar', 'mup', 'noe', 'pan', 'raj'] | Classification | s2s | [Social, Spoken, Written] | {'test': 1152} | {'test': 583.82} | -| [HindiDiscourseClassification](https://aclanthology.org/2020.lrec-1.149/) | ['hin'] | Classification | s2s | [Fiction, Social, Written] | {'train': 2048} | {'train': 79.23828125} | -| [HotelReviewSentimentClassification](https://link.springer.com/chapter/10.1007/978-3-319-67056-0_3) (Elnagar et al., 2018) | ['ara'] | Classification | s2s | [Reviews, Written] | {'train': 2048} | {'train': 137.2} | -| [HotpotQA](https://hotpotqa.github.io/) | ['eng'] | Retrieval | s2p | [Web, Written] | None | {'train': {'average_document_length': 287.9079517072212, 'average_query_length': 105.54965882352941, 'num_documents': 5233329, 'num_queries': 85000, 'average_relevant_docs_per_query': 2.0}, 'dev': {'average_document_length': 287.9079517072212, 'average_query_length': 105.35634294106848, 'num_documents': 5233329, 'num_queries': 5447, 'average_relevant_docs_per_query': 2.0}, 'test': {'average_document_length': 287.9079517072212, 'average_query_length': 92.17096556380824, 'num_documents': 5233329, 'num_queries': 7405, 'average_relevant_docs_per_query': 2.0}} | -| [HotpotQA-PL](https://hotpotqa.github.io/) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | [Web, Written] | None | {'test': {'average_document_length': 292.26835882093405, 'average_query_length': 94.64064821066847, 'num_documents': 5233329, 'num_queries': 7405, 'average_relevant_docs_per_query': 2.0}} | -| [HotpotQA-PLHardNegatives](https://hotpotqa.github.io/) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | [Web, Written] | {'test': 1000} | {'test': {'average_document_length': 438.3888210025661, 'average_query_length': 95.161, 'num_documents': 212774, 'num_queries': 1000, 'average_relevant_docs_per_query': 2.0}} | -| [HotpotQAHardNegatives](https://hotpotqa.github.io/) | ['eng'] | Retrieval | s2p | [Web, Written] | {'test': 1000} | {'test': {'average_document_length': 373.558822095461, 'average_query_length': 92.584, 'num_documents': 225621, 'num_queries': 1000, 'average_relevant_docs_per_query': 2.0}} | -| [HunSum2AbstractiveRetrieval](https://arxiv.org/abs/2404.03555) (Botond Barta, 2024) | ['hun'] | Retrieval | s2p | [News, Written] | {'test': 1998} | {'test': {'average_document_length': 2511.0315315315315, 'average_query_length': 201.2112112112112, 'num_documents': 1998, 'num_queries': 1998, 'average_relevant_docs_per_query': 1.0}} | +| [GreekCivicsQA](https://huggingface.co/datasets/antoinelb7/alloprof) | ['ell'] | Retrieval | s2p | [Academic, Written] | None | None | +| [GreekLegalCodeClassification](https://arxiv.org/abs/2109.15298) | ['ell'] | Classification | s2s | [Legal, Written] | None | None | +| [GujaratiNewsClassification](https://github.com/goru001/nlp-for-gujarati) | ['guj'] | Classification | s2s | [News, Written] | None | None | +| [HALClusteringS2S.v2](https://huggingface.co/datasets/lyon-nlp/clustering-hal-s2s) (Mathieu Ciancone, 2024) | ['fra'] | Clustering | s2s | [Academic, Written] | None | None | +| [HagridRetrieval](https://github.com/project-miracl/hagrid) (Ehsan Kamalloo, 2023) | ['eng'] | Retrieval | s2p | [Encyclopaedic, Written] | None | None | +| [HateSpeechPortugueseClassification](https://aclanthology.org/W19-3510) | ['por'] | Classification | s2s | [Social, Written] | None | None | +| [HeadlineClassification](https://aclanthology.org/2020.ngt-1.6/) | ['rus'] | Classification | s2s | [News, Written] | None | None | +| [HebrewSentimentAnalysis](https://huggingface.co/datasets/hebrew_sentiment) | ['heb'] | Classification | s2s | [Reviews, Written] | None | None | +| [HellaSwag](https://rowanzellers.com/hellaswag/) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | None | None | +| [HinDialectClassification](https://lindat.mff.cuni.cz/repository/xmlui/handle/11234/1-4839) (Bafna et al., 2022) | ['anp', 'awa', 'ben', 'bgc', 'bhb', 'bhd', 'bho', 'bjj', 'bns', 'bra', 'gbm', 'guj', 'hne', 'kfg', 'kfy', 'mag', 'mar', 'mup', 'noe', 'pan', 'raj'] | Classification | s2s | [Social, Spoken, Written] | None | None | +| [HindiDiscourseClassification](https://aclanthology.org/2020.lrec-1.149/) | ['hin'] | Classification | s2s | [Fiction, Social, Written] | None | None | +| [HotelReviewSentimentClassification](https://link.springer.com/chapter/10.1007/978-3-319-67056-0_3) (Elnagar et al., 2018) | ['ara'] | Classification | s2s | [Reviews, Written] | None | None | +| [HotpotQA](https://hotpotqa.github.io/) | ['eng'] | Retrieval | s2p | [Web, Written] | None | None | +| [HotpotQA-PL](https://hotpotqa.github.io/) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | [Web, Written] | None | None | +| [HotpotQA-PLHardNegatives](https://hotpotqa.github.io/) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | [Web, Written] | None | None | +| [HotpotQAHardNegatives](https://hotpotqa.github.io/) | ['eng'] | Retrieval | s2p | [Web, Written] | None | None | +| [HunSum2AbstractiveRetrieval](https://arxiv.org/abs/2404.03555) (Botond Barta, 2024) | ['hun'] | Retrieval | s2p | [News, Written] | None | None | | [IFlyTek](https://www.cluebenchmarks.com/introduce.html) | ['cmn'] | Classification | s2s | | None | None | -| [IN22ConvBitextMining](https://huggingface.co/datasets/ai4bharat/IN22-Conv) (Jay Gala, 2023) | ['asm', 'ben', 'brx', 'doi', 'eng', 'gom', 'guj', 'hin', 'kan', 'kas', 'mai', 'mal', 'mar', 'mni', 'npi', 'ory', 'pan', 'san', 'sat', 'snd', 'tam', 'tel', 'urd'] | BitextMining | s2s | [Social, Spoken, Fiction, Spoken] | | | -| [IN22GenBitextMining](https://huggingface.co/datasets/ai4bharat/IN22-Gen) (Jay Gala, 2023) | ['asm', 'ben', 'brx', 'doi', 'eng', 'gom', 'guj', 'hin', 'kan', 'kas', 'mai', 'mal', 'mar', 'mni', 'npi', 'ory', 'pan', 'san', 'sat', 'snd', 'tam', 'tel', 'urd'] | BitextMining | s2s | [Web, Legal, Government, News, Religious, Non-fiction, Written] | {'test': 1024} | {'test': 156.7} | -| [IWSLT2017BitextMining](https://aclanthology.org/2017.iwslt-1.1/) | ['ara', 'cmn', 'deu', 'eng', 'fra', 'ita', 'jpn', 'kor', 'nld', 'ron'] | BitextMining | s2s | [Non-fiction, Fiction, Written] | {'validation': 21928} | {'validation': 95.4} | -| [ImdbClassification](http://www.aclweb.org/anthology/P11-1015) | ['eng'] | Classification | p2p | [Reviews, Written] | {'test': 25000} | {'test': 1293.8} | -| [InappropriatenessClassification](https://aclanthology.org/2021.bsnlp-1.4) | ['rus'] | Classification | s2s | [Web, Social, Written] | {'test': 2048} | {'test': 97.7} | -| [IndicCrosslingualSTS](https://huggingface.co/datasets/jaygala24/indic_sts) (Ramesh et al., 2022) | ['asm', 'ben', 'eng', 'guj', 'hin', 'kan', 'mal', 'mar', 'ory', 'pan', 'tam', 'tel', 'urd'] | STS | s2s | [News, Non-fiction, Web, Spoken, Government, Written, Spoken] | {'test': 10020} | {'test': 76.22} | -| [IndicGenBenchFloresBitextMining](https://github.com/google-research-datasets/indic-gen-bench/) (Harman Singh, 2024) | ['asm', 'awa', 'ben', 'bgc', 'bho', 'bod', 'boy', 'eng', 'gbm', 'gom', 'guj', 'hin', 'hne', 'kan', 'mai', 'mal', 'mar', 'mni', 'mup', 'mwr', 'nep', 'ory', 'pan', 'pus', 'raj', 'san', 'sat', 'tam', 'tel', 'urd'] | BitextMining | s2s | [Web, News, Written] | {'validation': 997, 'test': 1012} | {'validation': 126.25, 'test': 130.84} | -| [IndicLangClassification](https://arxiv.org/abs/2305.15814) | ['asm', 'ben', 'brx', 'doi', 'gom', 'guj', 'hin', 'kan', 'kas', 'mai', 'mal', 'mar', 'mni', 'npi', 'ory', 'pan', 'san', 'sat', 'snd', 'tam', 'tel', 'urd'] | Classification | s2s | [Web, Non-fiction, Written] | {'test': 30418} | {'test': 106.5} | -| [IndicNLPNewsClassification](https://github.com/AI4Bharat/indicnlp_corpus#indicnlp-news-article-classification-dataset) (Anoop Kunchukuttan, 2020) | ['guj', 'kan', 'mal', 'mar', 'ori', 'pan', 'tam', 'tel'] | Classification | s2s | [News, Written] | {'test': 2048} | {'test': 1169.053974484789} | -| [IndicQARetrieval](https://arxiv.org/abs/2212.05409) (Sumanth Doddapaneni, 2022) | ['asm', 'ben', 'guj', 'hin', 'kan', 'mal', 'mar', 'ory', 'pan', 'tam', 'tel'] | Retrieval | s2p | [Web, Written] | {'test': 18586} | {'test': {'as': {'average_document_length': 1401.28, 'average_query_length': 56.60504201680672, 'num_documents': 250, 'num_queries': 1785, 'average_relevant_docs_per_query': 1.0016806722689076}, 'bn': {'average_document_length': 2196.012, 'average_query_length': 57.069239500567534, 'num_documents': 250, 'num_queries': 1762, 'average_relevant_docs_per_query': 1.0005675368898979}, 'gu': {'average_document_length': 960.4959677419355, 'average_query_length': 60.3712158808933, 'num_documents': 248, 'num_queries': 2015, 'average_relevant_docs_per_query': 1.0009925558312656}, 'hi': {'average_document_length': 2550.770114942529, 'average_query_length': 52.84909326424871, 'num_documents': 261, 'num_queries': 1544, 'average_relevant_docs_per_query': 1.0019430051813472}, 'kn': {'average_document_length': 882.7354085603113, 'average_query_length': 50.58734344100198, 'num_documents': 257, 'num_queries': 1517, 'average_relevant_docs_per_query': 1.0}, 'ml': {'average_document_length': 2522.6437246963565, 'average_query_length': 75.93635790800252, 'num_documents': 247, 'num_queries': 1587, 'average_relevant_docs_per_query': 1.0}, 'mr': {'average_document_length': 1711.74, 'average_query_length': 58.785, 'num_documents': 250, 'num_queries': 1600, 'average_relevant_docs_per_query': 1.0}, 'or': {'average_document_length': 801.9206349206349, 'average_query_length': 55.072792362768496, 'num_documents': 252, 'num_queries': 1676, 'average_relevant_docs_per_query': 1.0011933174224343}, 'pa': {'average_document_length': 1423.5062240663901, 'average_query_length': 58.394925178919976, 'num_documents': 241, 'num_queries': 1537, 'average_relevant_docs_per_query': 1.0013012361743656}, 'ta': {'average_document_length': 2288.2608695652175, 'average_query_length': 54.06211869107044, 'num_documents': 253, 'num_queries': 1803, 'average_relevant_docs_per_query': 1.0005546311702718}, 'te': {'average_document_length': 2936.176, 'average_query_length': 67.00634371395617, 'num_documents': 250, 'num_queries': 1734, 'average_relevant_docs_per_query': 1.0}}} | -| [IndicReviewsClusteringP2P](https://arxiv.org/abs/2212.05409) (Sumanth Doddapaneni, 2022) | ['asm', 'ben', 'brx', 'guj', 'hin', 'kan', 'mal', 'mar', 'ory', 'pan', 'tam', 'tel', 'urd'] | Clustering | p2p | [Reviews, Written] | {'test': 1000} | {'test': 137.6} | -| [IndicSentimentClassification](https://arxiv.org/abs/2212.05409) (Sumanth Doddapaneni, 2022) | ['asm', 'ben', 'brx', 'guj', 'hin', 'kan', 'mal', 'mar', 'ory', 'pan', 'tam', 'tel', 'urd'] | Classification | s2s | [Reviews, Written] | {'test': 1000} | {'test': 137.6} | -| [IndonesianIdClickbaitClassification](http://www.sciencedirect.com/science/article/pii/S2352340920311252) | ['ind'] | Classification | s2s | [News, Written] | {'train': 2048} | {'train': 64.28} | -| [IndonesianMongabayConservationClassification](https://aclanthology.org/2023.sealp-1.4/) | ['ind'] | Classification | s2s | [Web, Written] | {'validation': 984, 'test': 970} | {'validation': 1675.8, 'test': 1675.5} | -| [InsurancePolicyInterpretationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 133} | {'test': 521.88} | -| [InternationalCitizenshipQuestionsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 2048} | {'test': 206.18} | -| [IsiZuluNewsClassification](https://huggingface.co/datasets/dsfsi/za-isizulu-siswati-news) (Madodonga et al., 2023) | ['zul'] | Classification | s2s | [News, Written] | {'train': 752} | {'train': 43.1} | -| [ItaCaseholdClassification](https://doi.org/10.1145/3594536.3595177) (Licari et al., 2023) | ['ita'] | Classification | s2s | [Legal, Government, Written] | {'test': 221} | {'test': 4207.9} | -| [Itacola](https://aclanthology.org/2021.findings-emnlp.250/) | ['ita'] | Classification | s2s | [Non-fiction, Spoken, Written] | {'train': 7801, 'test': 975} | {'train': 35.95, 'test': 36.67} | -| [JCrewBlockerLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 54} | {'test': 1092.22} | +| [IN22ConvBitextMining](https://huggingface.co/datasets/ai4bharat/IN22-Conv) (Jay Gala, 2023) | ['asm', 'ben', 'brx', 'doi', 'eng', 'gom', 'guj', 'hin', 'kan', 'kas', 'mai', 'mal', 'mar', 'mni', 'npi', 'ory', 'pan', 'san', 'sat', 'snd', 'tam', 'tel', 'urd'] | BitextMining | s2s | [Social, Spoken, Fiction, Spoken] | {'test': 760518} | {'test': {'average_sentence1_length': 54.33, 'average_sentence2_length': 54.33, 'num_samples': 760518, 'number_of_characters': 82637104, 'hf_subset_descriptive_stats': {'asm_Beng-ben_Beng': {'average_sentence1_length': 53.75, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 155988}, 'asm_Beng-brx_Deva': {'average_sentence1_length': 53.75, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 162044}, 'asm_Beng-doi_Deva': {'average_sentence1_length': 53.75, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 167032}, 'asm_Beng-eng_Latn': {'average_sentence1_length': 53.75, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 160716}, 'asm_Beng-gom_Deva': {'average_sentence1_length': 53.75, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 156282}, 'asm_Beng-guj_Gujr': {'average_sentence1_length': 53.75, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 158269}, 'asm_Beng-hin_Deva': {'average_sentence1_length': 53.75, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 159964}, 'asm_Beng-kan_Knda': {'average_sentence1_length': 53.75, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 165177}, 'asm_Beng-kas_Arab': {'average_sentence1_length': 53.75, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 164681}, 'asm_Beng-mai_Deva': {'average_sentence1_length': 53.75, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 162408}, 'asm_Beng-mal_Mlym': {'average_sentence1_length': 53.75, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 172838}, 'asm_Beng-mar_Deva': {'average_sentence1_length': 53.75, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 162747}, 'asm_Beng-mni_Mtei': {'average_sentence1_length': 53.75, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 157316}, 'asm_Beng-npi_Deva': {'average_sentence1_length': 53.75, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 160906}, 'asm_Beng-ory_Orya': {'average_sentence1_length': 53.75, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 164223}, 'asm_Beng-pan_Guru': {'average_sentence1_length': 53.75, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 160201}, 'asm_Beng-san_Deva': {'average_sentence1_length': 53.75, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 158093}, 'asm_Beng-sat_Olck': {'average_sentence1_length': 53.75, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 169379}, 'asm_Beng-snd_Deva': {'average_sentence1_length': 53.75, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 162623}, 'asm_Beng-tam_Taml': {'average_sentence1_length': 53.75, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 174866}, 'asm_Beng-tel_Telu': {'average_sentence1_length': 53.75, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 157690}, 'asm_Beng-urd_Arab': {'average_sentence1_length': 53.75, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 161305}, 'ben_Beng-asm_Beng': {'average_sentence1_length': 50.03, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 155988}, 'ben_Beng-brx_Deva': {'average_sentence1_length': 50.03, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 156448}, 'ben_Beng-doi_Deva': {'average_sentence1_length': 50.03, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 161436}, 'ben_Beng-eng_Latn': {'average_sentence1_length': 50.03, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 155120}, 'ben_Beng-gom_Deva': {'average_sentence1_length': 50.03, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 150686}, 'ben_Beng-guj_Gujr': {'average_sentence1_length': 50.03, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 152673}, 'ben_Beng-hin_Deva': {'average_sentence1_length': 50.03, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 154368}, 'ben_Beng-kan_Knda': {'average_sentence1_length': 50.03, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 159581}, 'ben_Beng-kas_Arab': {'average_sentence1_length': 50.03, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 159085}, 'ben_Beng-mai_Deva': {'average_sentence1_length': 50.03, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 156812}, 'ben_Beng-mal_Mlym': {'average_sentence1_length': 50.03, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 167242}, 'ben_Beng-mar_Deva': {'average_sentence1_length': 50.03, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 157151}, 'ben_Beng-mni_Mtei': {'average_sentence1_length': 50.03, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 151720}, 'ben_Beng-npi_Deva': {'average_sentence1_length': 50.03, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 155310}, 'ben_Beng-ory_Orya': {'average_sentence1_length': 50.03, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 158627}, 'ben_Beng-pan_Guru': {'average_sentence1_length': 50.03, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 154605}, 'ben_Beng-san_Deva': {'average_sentence1_length': 50.03, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 152497}, 'ben_Beng-sat_Olck': {'average_sentence1_length': 50.03, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 163783}, 'ben_Beng-snd_Deva': {'average_sentence1_length': 50.03, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 157027}, 'ben_Beng-tam_Taml': {'average_sentence1_length': 50.03, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 169270}, 'ben_Beng-tel_Telu': {'average_sentence1_length': 50.03, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 152094}, 'ben_Beng-urd_Arab': {'average_sentence1_length': 50.03, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 155709}, 'brx_Deva-asm_Beng': {'average_sentence1_length': 54.06, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 162044}, 'brx_Deva-ben_Beng': {'average_sentence1_length': 54.06, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 156448}, 'brx_Deva-doi_Deva': {'average_sentence1_length': 54.06, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 167492}, 'brx_Deva-eng_Latn': {'average_sentence1_length': 54.06, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 161176}, 'brx_Deva-gom_Deva': {'average_sentence1_length': 54.06, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 156742}, 'brx_Deva-guj_Gujr': {'average_sentence1_length': 54.06, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 158729}, 'brx_Deva-hin_Deva': {'average_sentence1_length': 54.06, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 160424}, 'brx_Deva-kan_Knda': {'average_sentence1_length': 54.06, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 165637}, 'brx_Deva-kas_Arab': {'average_sentence1_length': 54.06, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 165141}, 'brx_Deva-mai_Deva': {'average_sentence1_length': 54.06, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 162868}, 'brx_Deva-mal_Mlym': {'average_sentence1_length': 54.06, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 173298}, 'brx_Deva-mar_Deva': {'average_sentence1_length': 54.06, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 163207}, 'brx_Deva-mni_Mtei': {'average_sentence1_length': 54.06, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 157776}, 'brx_Deva-npi_Deva': {'average_sentence1_length': 54.06, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 161366}, 'brx_Deva-ory_Orya': {'average_sentence1_length': 54.06, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 164683}, 'brx_Deva-pan_Guru': {'average_sentence1_length': 54.06, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 160661}, 'brx_Deva-san_Deva': {'average_sentence1_length': 54.06, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 158553}, 'brx_Deva-sat_Olck': {'average_sentence1_length': 54.06, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 169839}, 'brx_Deva-snd_Deva': {'average_sentence1_length': 54.06, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 163083}, 'brx_Deva-tam_Taml': {'average_sentence1_length': 54.06, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 175326}, 'brx_Deva-tel_Telu': {'average_sentence1_length': 54.06, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 158150}, 'brx_Deva-urd_Arab': {'average_sentence1_length': 54.06, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 161765}, 'doi_Deva-asm_Beng': {'average_sentence1_length': 57.38, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 167032}, 'doi_Deva-ben_Beng': {'average_sentence1_length': 57.38, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 161436}, 'doi_Deva-brx_Deva': {'average_sentence1_length': 57.38, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 167492}, 'doi_Deva-eng_Latn': {'average_sentence1_length': 57.38, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 166164}, 'doi_Deva-gom_Deva': {'average_sentence1_length': 57.38, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 161730}, 'doi_Deva-guj_Gujr': {'average_sentence1_length': 57.38, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 163717}, 'doi_Deva-hin_Deva': {'average_sentence1_length': 57.38, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 165412}, 'doi_Deva-kan_Knda': {'average_sentence1_length': 57.38, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 170625}, 'doi_Deva-kas_Arab': {'average_sentence1_length': 57.38, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 170129}, 'doi_Deva-mai_Deva': {'average_sentence1_length': 57.38, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 167856}, 'doi_Deva-mal_Mlym': {'average_sentence1_length': 57.38, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 178286}, 'doi_Deva-mar_Deva': {'average_sentence1_length': 57.38, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 168195}, 'doi_Deva-mni_Mtei': {'average_sentence1_length': 57.38, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 162764}, 'doi_Deva-npi_Deva': {'average_sentence1_length': 57.38, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 166354}, 'doi_Deva-ory_Orya': {'average_sentence1_length': 57.38, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 169671}, 'doi_Deva-pan_Guru': {'average_sentence1_length': 57.38, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 165649}, 'doi_Deva-san_Deva': {'average_sentence1_length': 57.38, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 163541}, 'doi_Deva-sat_Olck': {'average_sentence1_length': 57.38, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 174827}, 'doi_Deva-snd_Deva': {'average_sentence1_length': 57.38, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 168071}, 'doi_Deva-tam_Taml': {'average_sentence1_length': 57.38, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 180314}, 'doi_Deva-tel_Telu': {'average_sentence1_length': 57.38, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 163138}, 'doi_Deva-urd_Arab': {'average_sentence1_length': 57.38, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 166753}, 'eng_Latn-asm_Beng': {'average_sentence1_length': 53.18, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 160716}, 'eng_Latn-ben_Beng': {'average_sentence1_length': 53.18, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 155120}, 'eng_Latn-brx_Deva': {'average_sentence1_length': 53.18, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 161176}, 'eng_Latn-doi_Deva': {'average_sentence1_length': 53.18, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 166164}, 'eng_Latn-gom_Deva': {'average_sentence1_length': 53.18, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 155414}, 'eng_Latn-guj_Gujr': {'average_sentence1_length': 53.18, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 157401}, 'eng_Latn-hin_Deva': {'average_sentence1_length': 53.18, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 159096}, 'eng_Latn-kan_Knda': {'average_sentence1_length': 53.18, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 164309}, 'eng_Latn-kas_Arab': {'average_sentence1_length': 53.18, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 163813}, 'eng_Latn-mai_Deva': {'average_sentence1_length': 53.18, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 161540}, 'eng_Latn-mal_Mlym': {'average_sentence1_length': 53.18, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 171970}, 'eng_Latn-mar_Deva': {'average_sentence1_length': 53.18, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 161879}, 'eng_Latn-mni_Mtei': {'average_sentence1_length': 53.18, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 156448}, 'eng_Latn-npi_Deva': {'average_sentence1_length': 53.18, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 160038}, 'eng_Latn-ory_Orya': {'average_sentence1_length': 53.18, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 163355}, 'eng_Latn-pan_Guru': {'average_sentence1_length': 53.18, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 159333}, 'eng_Latn-san_Deva': {'average_sentence1_length': 53.18, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 157225}, 'eng_Latn-sat_Olck': {'average_sentence1_length': 53.18, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 168511}, 'eng_Latn-snd_Deva': {'average_sentence1_length': 53.18, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 161755}, 'eng_Latn-tam_Taml': {'average_sentence1_length': 53.18, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 173998}, 'eng_Latn-tel_Telu': {'average_sentence1_length': 53.18, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 156822}, 'eng_Latn-urd_Arab': {'average_sentence1_length': 53.18, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 160437}, 'gom_Deva-asm_Beng': {'average_sentence1_length': 50.23, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 156282}, 'gom_Deva-ben_Beng': {'average_sentence1_length': 50.23, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 150686}, 'gom_Deva-brx_Deva': {'average_sentence1_length': 50.23, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 156742}, 'gom_Deva-doi_Deva': {'average_sentence1_length': 50.23, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 161730}, 'gom_Deva-eng_Latn': {'average_sentence1_length': 50.23, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 155414}, 'gom_Deva-guj_Gujr': {'average_sentence1_length': 50.23, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 152967}, 'gom_Deva-hin_Deva': {'average_sentence1_length': 50.23, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 154662}, 'gom_Deva-kan_Knda': {'average_sentence1_length': 50.23, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 159875}, 'gom_Deva-kas_Arab': {'average_sentence1_length': 50.23, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 159379}, 'gom_Deva-mai_Deva': {'average_sentence1_length': 50.23, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 157106}, 'gom_Deva-mal_Mlym': {'average_sentence1_length': 50.23, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 167536}, 'gom_Deva-mar_Deva': {'average_sentence1_length': 50.23, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 157445}, 'gom_Deva-mni_Mtei': {'average_sentence1_length': 50.23, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 152014}, 'gom_Deva-npi_Deva': {'average_sentence1_length': 50.23, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 155604}, 'gom_Deva-ory_Orya': {'average_sentence1_length': 50.23, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 158921}, 'gom_Deva-pan_Guru': {'average_sentence1_length': 50.23, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 154899}, 'gom_Deva-san_Deva': {'average_sentence1_length': 50.23, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 152791}, 'gom_Deva-sat_Olck': {'average_sentence1_length': 50.23, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 164077}, 'gom_Deva-snd_Deva': {'average_sentence1_length': 50.23, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 157321}, 'gom_Deva-tam_Taml': {'average_sentence1_length': 50.23, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 169564}, 'gom_Deva-tel_Telu': {'average_sentence1_length': 50.23, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 152388}, 'gom_Deva-urd_Arab': {'average_sentence1_length': 50.23, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 156003}, 'guj_Gujr-asm_Beng': {'average_sentence1_length': 51.55, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 158269}, 'guj_Gujr-ben_Beng': {'average_sentence1_length': 51.55, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 152673}, 'guj_Gujr-brx_Deva': {'average_sentence1_length': 51.55, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 158729}, 'guj_Gujr-doi_Deva': {'average_sentence1_length': 51.55, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 163717}, 'guj_Gujr-eng_Latn': {'average_sentence1_length': 51.55, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 157401}, 'guj_Gujr-gom_Deva': {'average_sentence1_length': 51.55, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 152967}, 'guj_Gujr-hin_Deva': {'average_sentence1_length': 51.55, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 156649}, 'guj_Gujr-kan_Knda': {'average_sentence1_length': 51.55, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 161862}, 'guj_Gujr-kas_Arab': {'average_sentence1_length': 51.55, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 161366}, 'guj_Gujr-mai_Deva': {'average_sentence1_length': 51.55, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 159093}, 'guj_Gujr-mal_Mlym': {'average_sentence1_length': 51.55, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 169523}, 'guj_Gujr-mar_Deva': {'average_sentence1_length': 51.55, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 159432}, 'guj_Gujr-mni_Mtei': {'average_sentence1_length': 51.55, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 154001}, 'guj_Gujr-npi_Deva': {'average_sentence1_length': 51.55, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 157591}, 'guj_Gujr-ory_Orya': {'average_sentence1_length': 51.55, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 160908}, 'guj_Gujr-pan_Guru': {'average_sentence1_length': 51.55, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 156886}, 'guj_Gujr-san_Deva': {'average_sentence1_length': 51.55, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 154778}, 'guj_Gujr-sat_Olck': {'average_sentence1_length': 51.55, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 166064}, 'guj_Gujr-snd_Deva': {'average_sentence1_length': 51.55, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 159308}, 'guj_Gujr-tam_Taml': {'average_sentence1_length': 51.55, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 171551}, 'guj_Gujr-tel_Telu': {'average_sentence1_length': 51.55, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 154375}, 'guj_Gujr-urd_Arab': {'average_sentence1_length': 51.55, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 157990}, 'hin_Deva-asm_Beng': {'average_sentence1_length': 52.68, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 159964}, 'hin_Deva-ben_Beng': {'average_sentence1_length': 52.68, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 154368}, 'hin_Deva-brx_Deva': {'average_sentence1_length': 52.68, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 160424}, 'hin_Deva-doi_Deva': {'average_sentence1_length': 52.68, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 165412}, 'hin_Deva-eng_Latn': {'average_sentence1_length': 52.68, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 159096}, 'hin_Deva-gom_Deva': {'average_sentence1_length': 52.68, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 154662}, 'hin_Deva-guj_Gujr': {'average_sentence1_length': 52.68, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 156649}, 'hin_Deva-kan_Knda': {'average_sentence1_length': 52.68, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 163557}, 'hin_Deva-kas_Arab': {'average_sentence1_length': 52.68, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 163061}, 'hin_Deva-mai_Deva': {'average_sentence1_length': 52.68, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 160788}, 'hin_Deva-mal_Mlym': {'average_sentence1_length': 52.68, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 171218}, 'hin_Deva-mar_Deva': {'average_sentence1_length': 52.68, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 161127}, 'hin_Deva-mni_Mtei': {'average_sentence1_length': 52.68, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 155696}, 'hin_Deva-npi_Deva': {'average_sentence1_length': 52.68, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 159286}, 'hin_Deva-ory_Orya': {'average_sentence1_length': 52.68, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 162603}, 'hin_Deva-pan_Guru': {'average_sentence1_length': 52.68, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 158581}, 'hin_Deva-san_Deva': {'average_sentence1_length': 52.68, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 156473}, 'hin_Deva-sat_Olck': {'average_sentence1_length': 52.68, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 167759}, 'hin_Deva-snd_Deva': {'average_sentence1_length': 52.68, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 161003}, 'hin_Deva-tam_Taml': {'average_sentence1_length': 52.68, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 173246}, 'hin_Deva-tel_Telu': {'average_sentence1_length': 52.68, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 156070}, 'hin_Deva-urd_Arab': {'average_sentence1_length': 52.68, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 159685}, 'kan_Knda-asm_Beng': {'average_sentence1_length': 56.14, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 165177}, 'kan_Knda-ben_Beng': {'average_sentence1_length': 56.14, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 159581}, 'kan_Knda-brx_Deva': {'average_sentence1_length': 56.14, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 165637}, 'kan_Knda-doi_Deva': {'average_sentence1_length': 56.14, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 170625}, 'kan_Knda-eng_Latn': {'average_sentence1_length': 56.14, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 164309}, 'kan_Knda-gom_Deva': {'average_sentence1_length': 56.14, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 159875}, 'kan_Knda-guj_Gujr': {'average_sentence1_length': 56.14, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 161862}, 'kan_Knda-hin_Deva': {'average_sentence1_length': 56.14, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 163557}, 'kan_Knda-kas_Arab': {'average_sentence1_length': 56.14, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 168274}, 'kan_Knda-mai_Deva': {'average_sentence1_length': 56.14, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 166001}, 'kan_Knda-mal_Mlym': {'average_sentence1_length': 56.14, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 176431}, 'kan_Knda-mar_Deva': {'average_sentence1_length': 56.14, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 166340}, 'kan_Knda-mni_Mtei': {'average_sentence1_length': 56.14, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 160909}, 'kan_Knda-npi_Deva': {'average_sentence1_length': 56.14, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 164499}, 'kan_Knda-ory_Orya': {'average_sentence1_length': 56.14, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 167816}, 'kan_Knda-pan_Guru': {'average_sentence1_length': 56.14, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 163794}, 'kan_Knda-san_Deva': {'average_sentence1_length': 56.14, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 161686}, 'kan_Knda-sat_Olck': {'average_sentence1_length': 56.14, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 172972}, 'kan_Knda-snd_Deva': {'average_sentence1_length': 56.14, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 166216}, 'kan_Knda-tam_Taml': {'average_sentence1_length': 56.14, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 178459}, 'kan_Knda-tel_Telu': {'average_sentence1_length': 56.14, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 161283}, 'kan_Knda-urd_Arab': {'average_sentence1_length': 56.14, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 164898}, 'kas_Arab-asm_Beng': {'average_sentence1_length': 55.81, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 164681}, 'kas_Arab-ben_Beng': {'average_sentence1_length': 55.81, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 159085}, 'kas_Arab-brx_Deva': {'average_sentence1_length': 55.81, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 165141}, 'kas_Arab-doi_Deva': {'average_sentence1_length': 55.81, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 170129}, 'kas_Arab-eng_Latn': {'average_sentence1_length': 55.81, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 163813}, 'kas_Arab-gom_Deva': {'average_sentence1_length': 55.81, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 159379}, 'kas_Arab-guj_Gujr': {'average_sentence1_length': 55.81, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 161366}, 'kas_Arab-hin_Deva': {'average_sentence1_length': 55.81, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 163061}, 'kas_Arab-kan_Knda': {'average_sentence1_length': 55.81, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 168274}, 'kas_Arab-mai_Deva': {'average_sentence1_length': 55.81, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 165505}, 'kas_Arab-mal_Mlym': {'average_sentence1_length': 55.81, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 175935}, 'kas_Arab-mar_Deva': {'average_sentence1_length': 55.81, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 165844}, 'kas_Arab-mni_Mtei': {'average_sentence1_length': 55.81, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 160413}, 'kas_Arab-npi_Deva': {'average_sentence1_length': 55.81, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 164003}, 'kas_Arab-ory_Orya': {'average_sentence1_length': 55.81, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 167320}, 'kas_Arab-pan_Guru': {'average_sentence1_length': 55.81, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 163298}, 'kas_Arab-san_Deva': {'average_sentence1_length': 55.81, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 161190}, 'kas_Arab-sat_Olck': {'average_sentence1_length': 55.81, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 172476}, 'kas_Arab-snd_Deva': {'average_sentence1_length': 55.81, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 165720}, 'kas_Arab-tam_Taml': {'average_sentence1_length': 55.81, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 177963}, 'kas_Arab-tel_Telu': {'average_sentence1_length': 55.81, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 160787}, 'kas_Arab-urd_Arab': {'average_sentence1_length': 55.81, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 164402}, 'mai_Deva-asm_Beng': {'average_sentence1_length': 54.3, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 162408}, 'mai_Deva-ben_Beng': {'average_sentence1_length': 54.3, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 156812}, 'mai_Deva-brx_Deva': {'average_sentence1_length': 54.3, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 162868}, 'mai_Deva-doi_Deva': {'average_sentence1_length': 54.3, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 167856}, 'mai_Deva-eng_Latn': {'average_sentence1_length': 54.3, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 161540}, 'mai_Deva-gom_Deva': {'average_sentence1_length': 54.3, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 157106}, 'mai_Deva-guj_Gujr': {'average_sentence1_length': 54.3, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 159093}, 'mai_Deva-hin_Deva': {'average_sentence1_length': 54.3, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 160788}, 'mai_Deva-kan_Knda': {'average_sentence1_length': 54.3, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 166001}, 'mai_Deva-kas_Arab': {'average_sentence1_length': 54.3, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 165505}, 'mai_Deva-mal_Mlym': {'average_sentence1_length': 54.3, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 173662}, 'mai_Deva-mar_Deva': {'average_sentence1_length': 54.3, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 163571}, 'mai_Deva-mni_Mtei': {'average_sentence1_length': 54.3, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 158140}, 'mai_Deva-npi_Deva': {'average_sentence1_length': 54.3, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 161730}, 'mai_Deva-ory_Orya': {'average_sentence1_length': 54.3, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 165047}, 'mai_Deva-pan_Guru': {'average_sentence1_length': 54.3, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 161025}, 'mai_Deva-san_Deva': {'average_sentence1_length': 54.3, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 158917}, 'mai_Deva-sat_Olck': {'average_sentence1_length': 54.3, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 170203}, 'mai_Deva-snd_Deva': {'average_sentence1_length': 54.3, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 163447}, 'mai_Deva-tam_Taml': {'average_sentence1_length': 54.3, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 175690}, 'mai_Deva-tel_Telu': {'average_sentence1_length': 54.3, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 158514}, 'mai_Deva-urd_Arab': {'average_sentence1_length': 54.3, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 162129}, 'mal_Mlym-asm_Beng': {'average_sentence1_length': 61.24, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 172838}, 'mal_Mlym-ben_Beng': {'average_sentence1_length': 61.24, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 167242}, 'mal_Mlym-brx_Deva': {'average_sentence1_length': 61.24, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 173298}, 'mal_Mlym-doi_Deva': {'average_sentence1_length': 61.24, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 178286}, 'mal_Mlym-eng_Latn': {'average_sentence1_length': 61.24, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 171970}, 'mal_Mlym-gom_Deva': {'average_sentence1_length': 61.24, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 167536}, 'mal_Mlym-guj_Gujr': {'average_sentence1_length': 61.24, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 169523}, 'mal_Mlym-hin_Deva': {'average_sentence1_length': 61.24, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 171218}, 'mal_Mlym-kan_Knda': {'average_sentence1_length': 61.24, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 176431}, 'mal_Mlym-kas_Arab': {'average_sentence1_length': 61.24, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 175935}, 'mal_Mlym-mai_Deva': {'average_sentence1_length': 61.24, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 173662}, 'mal_Mlym-mar_Deva': {'average_sentence1_length': 61.24, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 174001}, 'mal_Mlym-mni_Mtei': {'average_sentence1_length': 61.24, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 168570}, 'mal_Mlym-npi_Deva': {'average_sentence1_length': 61.24, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 172160}, 'mal_Mlym-ory_Orya': {'average_sentence1_length': 61.24, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 175477}, 'mal_Mlym-pan_Guru': {'average_sentence1_length': 61.24, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 171455}, 'mal_Mlym-san_Deva': {'average_sentence1_length': 61.24, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 169347}, 'mal_Mlym-sat_Olck': {'average_sentence1_length': 61.24, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 180633}, 'mal_Mlym-snd_Deva': {'average_sentence1_length': 61.24, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 173877}, 'mal_Mlym-tam_Taml': {'average_sentence1_length': 61.24, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 186120}, 'mal_Mlym-tel_Telu': {'average_sentence1_length': 61.24, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 168944}, 'mal_Mlym-urd_Arab': {'average_sentence1_length': 61.24, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 172559}, 'mar_Deva-asm_Beng': {'average_sentence1_length': 54.53, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 162747}, 'mar_Deva-ben_Beng': {'average_sentence1_length': 54.53, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 157151}, 'mar_Deva-brx_Deva': {'average_sentence1_length': 54.53, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 163207}, 'mar_Deva-doi_Deva': {'average_sentence1_length': 54.53, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 168195}, 'mar_Deva-eng_Latn': {'average_sentence1_length': 54.53, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 161879}, 'mar_Deva-gom_Deva': {'average_sentence1_length': 54.53, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 157445}, 'mar_Deva-guj_Gujr': {'average_sentence1_length': 54.53, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 159432}, 'mar_Deva-hin_Deva': {'average_sentence1_length': 54.53, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 161127}, 'mar_Deva-kan_Knda': {'average_sentence1_length': 54.53, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 166340}, 'mar_Deva-kas_Arab': {'average_sentence1_length': 54.53, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 165844}, 'mar_Deva-mai_Deva': {'average_sentence1_length': 54.53, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 163571}, 'mar_Deva-mal_Mlym': {'average_sentence1_length': 54.53, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 174001}, 'mar_Deva-mni_Mtei': {'average_sentence1_length': 54.53, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 158479}, 'mar_Deva-npi_Deva': {'average_sentence1_length': 54.53, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 162069}, 'mar_Deva-ory_Orya': {'average_sentence1_length': 54.53, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 165386}, 'mar_Deva-pan_Guru': {'average_sentence1_length': 54.53, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 161364}, 'mar_Deva-san_Deva': {'average_sentence1_length': 54.53, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 159256}, 'mar_Deva-sat_Olck': {'average_sentence1_length': 54.53, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 170542}, 'mar_Deva-snd_Deva': {'average_sentence1_length': 54.53, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 163786}, 'mar_Deva-tam_Taml': {'average_sentence1_length': 54.53, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 176029}, 'mar_Deva-tel_Telu': {'average_sentence1_length': 54.53, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 158853}, 'mar_Deva-urd_Arab': {'average_sentence1_length': 54.53, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 162468}, 'mni_Mtei-asm_Beng': {'average_sentence1_length': 50.91, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 157316}, 'mni_Mtei-ben_Beng': {'average_sentence1_length': 50.91, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 151720}, 'mni_Mtei-brx_Deva': {'average_sentence1_length': 50.91, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 157776}, 'mni_Mtei-doi_Deva': {'average_sentence1_length': 50.91, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 162764}, 'mni_Mtei-eng_Latn': {'average_sentence1_length': 50.91, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 156448}, 'mni_Mtei-gom_Deva': {'average_sentence1_length': 50.91, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 152014}, 'mni_Mtei-guj_Gujr': {'average_sentence1_length': 50.91, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 154001}, 'mni_Mtei-hin_Deva': {'average_sentence1_length': 50.91, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 155696}, 'mni_Mtei-kan_Knda': {'average_sentence1_length': 50.91, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 160909}, 'mni_Mtei-kas_Arab': {'average_sentence1_length': 50.91, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 160413}, 'mni_Mtei-mai_Deva': {'average_sentence1_length': 50.91, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 158140}, 'mni_Mtei-mal_Mlym': {'average_sentence1_length': 50.91, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 168570}, 'mni_Mtei-mar_Deva': {'average_sentence1_length': 50.91, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 158479}, 'mni_Mtei-npi_Deva': {'average_sentence1_length': 50.91, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 156638}, 'mni_Mtei-ory_Orya': {'average_sentence1_length': 50.91, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 159955}, 'mni_Mtei-pan_Guru': {'average_sentence1_length': 50.91, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 155933}, 'mni_Mtei-san_Deva': {'average_sentence1_length': 50.91, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 153825}, 'mni_Mtei-sat_Olck': {'average_sentence1_length': 50.91, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 165111}, 'mni_Mtei-snd_Deva': {'average_sentence1_length': 50.91, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 158355}, 'mni_Mtei-tam_Taml': {'average_sentence1_length': 50.91, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 170598}, 'mni_Mtei-tel_Telu': {'average_sentence1_length': 50.91, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 153422}, 'mni_Mtei-urd_Arab': {'average_sentence1_length': 50.91, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 157037}, 'npi_Deva-asm_Beng': {'average_sentence1_length': 53.3, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 160906}, 'npi_Deva-ben_Beng': {'average_sentence1_length': 53.3, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 155310}, 'npi_Deva-brx_Deva': {'average_sentence1_length': 53.3, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 161366}, 'npi_Deva-doi_Deva': {'average_sentence1_length': 53.3, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 166354}, 'npi_Deva-eng_Latn': {'average_sentence1_length': 53.3, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 160038}, 'npi_Deva-gom_Deva': {'average_sentence1_length': 53.3, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 155604}, 'npi_Deva-guj_Gujr': {'average_sentence1_length': 53.3, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 157591}, 'npi_Deva-hin_Deva': {'average_sentence1_length': 53.3, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 159286}, 'npi_Deva-kan_Knda': {'average_sentence1_length': 53.3, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 164499}, 'npi_Deva-kas_Arab': {'average_sentence1_length': 53.3, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 164003}, 'npi_Deva-mai_Deva': {'average_sentence1_length': 53.3, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 161730}, 'npi_Deva-mal_Mlym': {'average_sentence1_length': 53.3, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 172160}, 'npi_Deva-mar_Deva': {'average_sentence1_length': 53.3, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 162069}, 'npi_Deva-mni_Mtei': {'average_sentence1_length': 53.3, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 156638}, 'npi_Deva-ory_Orya': {'average_sentence1_length': 53.3, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 163545}, 'npi_Deva-pan_Guru': {'average_sentence1_length': 53.3, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 159523}, 'npi_Deva-san_Deva': {'average_sentence1_length': 53.3, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 157415}, 'npi_Deva-sat_Olck': {'average_sentence1_length': 53.3, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 168701}, 'npi_Deva-snd_Deva': {'average_sentence1_length': 53.3, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 161945}, 'npi_Deva-tam_Taml': {'average_sentence1_length': 53.3, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 174188}, 'npi_Deva-tel_Telu': {'average_sentence1_length': 53.3, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 157012}, 'npi_Deva-urd_Arab': {'average_sentence1_length': 53.3, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 160627}, 'ory_Orya-asm_Beng': {'average_sentence1_length': 55.51, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 164223}, 'ory_Orya-ben_Beng': {'average_sentence1_length': 55.51, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 158627}, 'ory_Orya-brx_Deva': {'average_sentence1_length': 55.51, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 164683}, 'ory_Orya-doi_Deva': {'average_sentence1_length': 55.51, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 169671}, 'ory_Orya-eng_Latn': {'average_sentence1_length': 55.51, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 163355}, 'ory_Orya-gom_Deva': {'average_sentence1_length': 55.51, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 158921}, 'ory_Orya-guj_Gujr': {'average_sentence1_length': 55.51, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 160908}, 'ory_Orya-hin_Deva': {'average_sentence1_length': 55.51, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 162603}, 'ory_Orya-kan_Knda': {'average_sentence1_length': 55.51, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 167816}, 'ory_Orya-kas_Arab': {'average_sentence1_length': 55.51, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 167320}, 'ory_Orya-mai_Deva': {'average_sentence1_length': 55.51, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 165047}, 'ory_Orya-mal_Mlym': {'average_sentence1_length': 55.51, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 175477}, 'ory_Orya-mar_Deva': {'average_sentence1_length': 55.51, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 165386}, 'ory_Orya-mni_Mtei': {'average_sentence1_length': 55.51, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 159955}, 'ory_Orya-npi_Deva': {'average_sentence1_length': 55.51, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 163545}, 'ory_Orya-pan_Guru': {'average_sentence1_length': 55.51, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 162840}, 'ory_Orya-san_Deva': {'average_sentence1_length': 55.51, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 160732}, 'ory_Orya-sat_Olck': {'average_sentence1_length': 55.51, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 172018}, 'ory_Orya-snd_Deva': {'average_sentence1_length': 55.51, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 165262}, 'ory_Orya-tam_Taml': {'average_sentence1_length': 55.51, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 177505}, 'ory_Orya-tel_Telu': {'average_sentence1_length': 55.51, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 160329}, 'ory_Orya-urd_Arab': {'average_sentence1_length': 55.51, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 163944}, 'pan_Guru-asm_Beng': {'average_sentence1_length': 52.83, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 160201}, 'pan_Guru-ben_Beng': {'average_sentence1_length': 52.83, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 154605}, 'pan_Guru-brx_Deva': {'average_sentence1_length': 52.83, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 160661}, 'pan_Guru-doi_Deva': {'average_sentence1_length': 52.83, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 165649}, 'pan_Guru-eng_Latn': {'average_sentence1_length': 52.83, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 159333}, 'pan_Guru-gom_Deva': {'average_sentence1_length': 52.83, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 154899}, 'pan_Guru-guj_Gujr': {'average_sentence1_length': 52.83, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 156886}, 'pan_Guru-hin_Deva': {'average_sentence1_length': 52.83, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 158581}, 'pan_Guru-kan_Knda': {'average_sentence1_length': 52.83, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 163794}, 'pan_Guru-kas_Arab': {'average_sentence1_length': 52.83, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 163298}, 'pan_Guru-mai_Deva': {'average_sentence1_length': 52.83, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 161025}, 'pan_Guru-mal_Mlym': {'average_sentence1_length': 52.83, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 171455}, 'pan_Guru-mar_Deva': {'average_sentence1_length': 52.83, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 161364}, 'pan_Guru-mni_Mtei': {'average_sentence1_length': 52.83, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 155933}, 'pan_Guru-npi_Deva': {'average_sentence1_length': 52.83, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 159523}, 'pan_Guru-ory_Orya': {'average_sentence1_length': 52.83, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 162840}, 'pan_Guru-san_Deva': {'average_sentence1_length': 52.83, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 156710}, 'pan_Guru-sat_Olck': {'average_sentence1_length': 52.83, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 167996}, 'pan_Guru-snd_Deva': {'average_sentence1_length': 52.83, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 161240}, 'pan_Guru-tam_Taml': {'average_sentence1_length': 52.83, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 173483}, 'pan_Guru-tel_Telu': {'average_sentence1_length': 52.83, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 156307}, 'pan_Guru-urd_Arab': {'average_sentence1_length': 52.83, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 159922}, 'san_Deva-asm_Beng': {'average_sentence1_length': 51.43, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 158093}, 'san_Deva-ben_Beng': {'average_sentence1_length': 51.43, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 152497}, 'san_Deva-brx_Deva': {'average_sentence1_length': 51.43, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 158553}, 'san_Deva-doi_Deva': {'average_sentence1_length': 51.43, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 163541}, 'san_Deva-eng_Latn': {'average_sentence1_length': 51.43, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 157225}, 'san_Deva-gom_Deva': {'average_sentence1_length': 51.43, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 152791}, 'san_Deva-guj_Gujr': {'average_sentence1_length': 51.43, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 154778}, 'san_Deva-hin_Deva': {'average_sentence1_length': 51.43, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 156473}, 'san_Deva-kan_Knda': {'average_sentence1_length': 51.43, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 161686}, 'san_Deva-kas_Arab': {'average_sentence1_length': 51.43, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 161190}, 'san_Deva-mai_Deva': {'average_sentence1_length': 51.43, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 158917}, 'san_Deva-mal_Mlym': {'average_sentence1_length': 51.43, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 169347}, 'san_Deva-mar_Deva': {'average_sentence1_length': 51.43, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 159256}, 'san_Deva-mni_Mtei': {'average_sentence1_length': 51.43, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 153825}, 'san_Deva-npi_Deva': {'average_sentence1_length': 51.43, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 157415}, 'san_Deva-ory_Orya': {'average_sentence1_length': 51.43, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 160732}, 'san_Deva-pan_Guru': {'average_sentence1_length': 51.43, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 156710}, 'san_Deva-sat_Olck': {'average_sentence1_length': 51.43, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 165888}, 'san_Deva-snd_Deva': {'average_sentence1_length': 51.43, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 159132}, 'san_Deva-tam_Taml': {'average_sentence1_length': 51.43, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 171375}, 'san_Deva-tel_Telu': {'average_sentence1_length': 51.43, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 154199}, 'san_Deva-urd_Arab': {'average_sentence1_length': 51.43, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 157814}, 'sat_Olck-asm_Beng': {'average_sentence1_length': 58.94, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 169379}, 'sat_Olck-ben_Beng': {'average_sentence1_length': 58.94, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 163783}, 'sat_Olck-brx_Deva': {'average_sentence1_length': 58.94, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 169839}, 'sat_Olck-doi_Deva': {'average_sentence1_length': 58.94, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 174827}, 'sat_Olck-eng_Latn': {'average_sentence1_length': 58.94, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 168511}, 'sat_Olck-gom_Deva': {'average_sentence1_length': 58.94, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 164077}, 'sat_Olck-guj_Gujr': {'average_sentence1_length': 58.94, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 166064}, 'sat_Olck-hin_Deva': {'average_sentence1_length': 58.94, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 167759}, 'sat_Olck-kan_Knda': {'average_sentence1_length': 58.94, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 172972}, 'sat_Olck-kas_Arab': {'average_sentence1_length': 58.94, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 172476}, 'sat_Olck-mai_Deva': {'average_sentence1_length': 58.94, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 170203}, 'sat_Olck-mal_Mlym': {'average_sentence1_length': 58.94, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 180633}, 'sat_Olck-mar_Deva': {'average_sentence1_length': 58.94, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 170542}, 'sat_Olck-mni_Mtei': {'average_sentence1_length': 58.94, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 165111}, 'sat_Olck-npi_Deva': {'average_sentence1_length': 58.94, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 168701}, 'sat_Olck-ory_Orya': {'average_sentence1_length': 58.94, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 172018}, 'sat_Olck-pan_Guru': {'average_sentence1_length': 58.94, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 167996}, 'sat_Olck-san_Deva': {'average_sentence1_length': 58.94, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 165888}, 'sat_Olck-snd_Deva': {'average_sentence1_length': 58.94, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 170418}, 'sat_Olck-tam_Taml': {'average_sentence1_length': 58.94, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 182661}, 'sat_Olck-tel_Telu': {'average_sentence1_length': 58.94, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 165485}, 'sat_Olck-urd_Arab': {'average_sentence1_length': 58.94, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 169100}, 'snd_Deva-asm_Beng': {'average_sentence1_length': 54.45, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 162623}, 'snd_Deva-ben_Beng': {'average_sentence1_length': 54.45, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 157027}, 'snd_Deva-brx_Deva': {'average_sentence1_length': 54.45, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 163083}, 'snd_Deva-doi_Deva': {'average_sentence1_length': 54.45, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 168071}, 'snd_Deva-eng_Latn': {'average_sentence1_length': 54.45, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 161755}, 'snd_Deva-gom_Deva': {'average_sentence1_length': 54.45, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 157321}, 'snd_Deva-guj_Gujr': {'average_sentence1_length': 54.45, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 159308}, 'snd_Deva-hin_Deva': {'average_sentence1_length': 54.45, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 161003}, 'snd_Deva-kan_Knda': {'average_sentence1_length': 54.45, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 166216}, 'snd_Deva-kas_Arab': {'average_sentence1_length': 54.45, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 165720}, 'snd_Deva-mai_Deva': {'average_sentence1_length': 54.45, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 163447}, 'snd_Deva-mal_Mlym': {'average_sentence1_length': 54.45, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 173877}, 'snd_Deva-mar_Deva': {'average_sentence1_length': 54.45, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 163786}, 'snd_Deva-mni_Mtei': {'average_sentence1_length': 54.45, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 158355}, 'snd_Deva-npi_Deva': {'average_sentence1_length': 54.45, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 161945}, 'snd_Deva-ory_Orya': {'average_sentence1_length': 54.45, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 165262}, 'snd_Deva-pan_Guru': {'average_sentence1_length': 54.45, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 161240}, 'snd_Deva-san_Deva': {'average_sentence1_length': 54.45, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 159132}, 'snd_Deva-sat_Olck': {'average_sentence1_length': 54.45, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 170418}, 'snd_Deva-tam_Taml': {'average_sentence1_length': 54.45, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 175905}, 'snd_Deva-tel_Telu': {'average_sentence1_length': 54.45, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 158729}, 'snd_Deva-urd_Arab': {'average_sentence1_length': 54.45, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 162344}, 'tam_Taml-asm_Beng': {'average_sentence1_length': 62.59, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 174866}, 'tam_Taml-ben_Beng': {'average_sentence1_length': 62.59, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 169270}, 'tam_Taml-brx_Deva': {'average_sentence1_length': 62.59, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 175326}, 'tam_Taml-doi_Deva': {'average_sentence1_length': 62.59, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 180314}, 'tam_Taml-eng_Latn': {'average_sentence1_length': 62.59, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 173998}, 'tam_Taml-gom_Deva': {'average_sentence1_length': 62.59, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 169564}, 'tam_Taml-guj_Gujr': {'average_sentence1_length': 62.59, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 171551}, 'tam_Taml-hin_Deva': {'average_sentence1_length': 62.59, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 173246}, 'tam_Taml-kan_Knda': {'average_sentence1_length': 62.59, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 178459}, 'tam_Taml-kas_Arab': {'average_sentence1_length': 62.59, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 177963}, 'tam_Taml-mai_Deva': {'average_sentence1_length': 62.59, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 175690}, 'tam_Taml-mal_Mlym': {'average_sentence1_length': 62.59, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 186120}, 'tam_Taml-mar_Deva': {'average_sentence1_length': 62.59, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 176029}, 'tam_Taml-mni_Mtei': {'average_sentence1_length': 62.59, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 170598}, 'tam_Taml-npi_Deva': {'average_sentence1_length': 62.59, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 174188}, 'tam_Taml-ory_Orya': {'average_sentence1_length': 62.59, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 177505}, 'tam_Taml-pan_Guru': {'average_sentence1_length': 62.59, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 173483}, 'tam_Taml-san_Deva': {'average_sentence1_length': 62.59, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 171375}, 'tam_Taml-sat_Olck': {'average_sentence1_length': 62.59, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 182661}, 'tam_Taml-snd_Deva': {'average_sentence1_length': 62.59, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 175905}, 'tam_Taml-tel_Telu': {'average_sentence1_length': 62.59, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 170972}, 'tam_Taml-urd_Arab': {'average_sentence1_length': 62.59, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 174587}, 'tel_Telu-asm_Beng': {'average_sentence1_length': 51.16, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 157690}, 'tel_Telu-ben_Beng': {'average_sentence1_length': 51.16, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 152094}, 'tel_Telu-brx_Deva': {'average_sentence1_length': 51.16, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 158150}, 'tel_Telu-doi_Deva': {'average_sentence1_length': 51.16, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 163138}, 'tel_Telu-eng_Latn': {'average_sentence1_length': 51.16, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 156822}, 'tel_Telu-gom_Deva': {'average_sentence1_length': 51.16, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 152388}, 'tel_Telu-guj_Gujr': {'average_sentence1_length': 51.16, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 154375}, 'tel_Telu-hin_Deva': {'average_sentence1_length': 51.16, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 156070}, 'tel_Telu-kan_Knda': {'average_sentence1_length': 51.16, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 161283}, 'tel_Telu-kas_Arab': {'average_sentence1_length': 51.16, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 160787}, 'tel_Telu-mai_Deva': {'average_sentence1_length': 51.16, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 158514}, 'tel_Telu-mal_Mlym': {'average_sentence1_length': 51.16, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 168944}, 'tel_Telu-mar_Deva': {'average_sentence1_length': 51.16, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 158853}, 'tel_Telu-mni_Mtei': {'average_sentence1_length': 51.16, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 153422}, 'tel_Telu-npi_Deva': {'average_sentence1_length': 51.16, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 157012}, 'tel_Telu-ory_Orya': {'average_sentence1_length': 51.16, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 160329}, 'tel_Telu-pan_Guru': {'average_sentence1_length': 51.16, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 156307}, 'tel_Telu-san_Deva': {'average_sentence1_length': 51.16, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 154199}, 'tel_Telu-sat_Olck': {'average_sentence1_length': 51.16, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 165485}, 'tel_Telu-snd_Deva': {'average_sentence1_length': 51.16, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 158729}, 'tel_Telu-tam_Taml': {'average_sentence1_length': 51.16, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 170972}, 'tel_Telu-urd_Arab': {'average_sentence1_length': 51.16, 'average_sentence2_length': 53.57, 'num_samples': 1503, 'number_of_characters': 157411}, 'urd_Arab-asm_Beng': {'average_sentence1_length': 53.57, 'average_sentence2_length': 53.75, 'num_samples': 1503, 'number_of_characters': 161305}, 'urd_Arab-ben_Beng': {'average_sentence1_length': 53.57, 'average_sentence2_length': 50.03, 'num_samples': 1503, 'number_of_characters': 155709}, 'urd_Arab-brx_Deva': {'average_sentence1_length': 53.57, 'average_sentence2_length': 54.06, 'num_samples': 1503, 'number_of_characters': 161765}, 'urd_Arab-doi_Deva': {'average_sentence1_length': 53.57, 'average_sentence2_length': 57.38, 'num_samples': 1503, 'number_of_characters': 166753}, 'urd_Arab-eng_Latn': {'average_sentence1_length': 53.57, 'average_sentence2_length': 53.18, 'num_samples': 1503, 'number_of_characters': 160437}, 'urd_Arab-gom_Deva': {'average_sentence1_length': 53.57, 'average_sentence2_length': 50.23, 'num_samples': 1503, 'number_of_characters': 156003}, 'urd_Arab-guj_Gujr': {'average_sentence1_length': 53.57, 'average_sentence2_length': 51.55, 'num_samples': 1503, 'number_of_characters': 157990}, 'urd_Arab-hin_Deva': {'average_sentence1_length': 53.57, 'average_sentence2_length': 52.68, 'num_samples': 1503, 'number_of_characters': 159685}, 'urd_Arab-kan_Knda': {'average_sentence1_length': 53.57, 'average_sentence2_length': 56.14, 'num_samples': 1503, 'number_of_characters': 164898}, 'urd_Arab-kas_Arab': {'average_sentence1_length': 53.57, 'average_sentence2_length': 55.81, 'num_samples': 1503, 'number_of_characters': 164402}, 'urd_Arab-mai_Deva': {'average_sentence1_length': 53.57, 'average_sentence2_length': 54.3, 'num_samples': 1503, 'number_of_characters': 162129}, 'urd_Arab-mal_Mlym': {'average_sentence1_length': 53.57, 'average_sentence2_length': 61.24, 'num_samples': 1503, 'number_of_characters': 172559}, 'urd_Arab-mar_Deva': {'average_sentence1_length': 53.57, 'average_sentence2_length': 54.53, 'num_samples': 1503, 'number_of_characters': 162468}, 'urd_Arab-mni_Mtei': {'average_sentence1_length': 53.57, 'average_sentence2_length': 50.91, 'num_samples': 1503, 'number_of_characters': 157037}, 'urd_Arab-npi_Deva': {'average_sentence1_length': 53.57, 'average_sentence2_length': 53.3, 'num_samples': 1503, 'number_of_characters': 160627}, 'urd_Arab-ory_Orya': {'average_sentence1_length': 53.57, 'average_sentence2_length': 55.51, 'num_samples': 1503, 'number_of_characters': 163944}, 'urd_Arab-pan_Guru': {'average_sentence1_length': 53.57, 'average_sentence2_length': 52.83, 'num_samples': 1503, 'number_of_characters': 159922}, 'urd_Arab-san_Deva': {'average_sentence1_length': 53.57, 'average_sentence2_length': 51.43, 'num_samples': 1503, 'number_of_characters': 157814}, 'urd_Arab-sat_Olck': {'average_sentence1_length': 53.57, 'average_sentence2_length': 58.94, 'num_samples': 1503, 'number_of_characters': 169100}, 'urd_Arab-snd_Deva': {'average_sentence1_length': 53.57, 'average_sentence2_length': 54.45, 'num_samples': 1503, 'number_of_characters': 162344}, 'urd_Arab-tam_Taml': {'average_sentence1_length': 53.57, 'average_sentence2_length': 62.59, 'num_samples': 1503, 'number_of_characters': 174587}, 'urd_Arab-tel_Telu': {'average_sentence1_length': 53.57, 'average_sentence2_length': 51.16, 'num_samples': 1503, 'number_of_characters': 157411}}}} | +| [IN22GenBitextMining](https://huggingface.co/datasets/ai4bharat/IN22-Gen) (Jay Gala, 2023) | ['asm', 'ben', 'brx', 'doi', 'eng', 'gom', 'guj', 'hin', 'kan', 'kas', 'mai', 'mal', 'mar', 'mni', 'npi', 'ory', 'pan', 'san', 'sat', 'snd', 'tam', 'tel', 'urd'] | BitextMining | s2s | [Web, Legal, Government, News, Religious, Non-fiction, Written] | None | None | +| [IWSLT2017BitextMining](https://aclanthology.org/2017.iwslt-1.1/) | ['ara', 'cmn', 'deu', 'eng', 'fra', 'ita', 'jpn', 'kor', 'nld', 'ron'] | BitextMining | s2s | [Non-fiction, Fiction, Written] | None | None | +| [ImdbClassification](http://www.aclweb.org/anthology/P11-1015) | ['eng'] | Classification | p2p | [Reviews, Written] | None | None | +| [InappropriatenessClassification](https://aclanthology.org/2021.bsnlp-1.4) | ['rus'] | Classification | s2s | [Web, Social, Written] | None | None | +| [IndicCrosslingualSTS](https://huggingface.co/datasets/jaygala24/indic_sts) (Ramesh et al., 2022) | ['asm', 'ben', 'eng', 'guj', 'hin', 'kan', 'mal', 'mar', 'ory', 'pan', 'tam', 'tel', 'urd'] | STS | s2s | [News, Non-fiction, Web, Spoken, Government, Written, Spoken] | None | None | +| [IndicGenBenchFloresBitextMining](https://github.com/google-research-datasets/indic-gen-bench/) (Harman Singh, 2024) | ['asm', 'awa', 'ben', 'bgc', 'bho', 'bod', 'boy', 'eng', 'gbm', 'gom', 'guj', 'hin', 'hne', 'kan', 'mai', 'mal', 'mar', 'mni', 'mup', 'mwr', 'nep', 'ory', 'pan', 'pus', 'raj', 'san', 'sat', 'tam', 'tel', 'urd'] | BitextMining | s2s | [Web, News, Written] | None | None | +| [IndicLangClassification](https://arxiv.org/abs/2305.15814) | ['asm', 'ben', 'brx', 'doi', 'gom', 'guj', 'hin', 'kan', 'kas', 'mai', 'mal', 'mar', 'mni', 'npi', 'ory', 'pan', 'san', 'sat', 'snd', 'tam', 'tel', 'urd'] | Classification | s2s | [Web, Non-fiction, Written] | None | None | +| [IndicNLPNewsClassification](https://github.com/AI4Bharat/indicnlp_corpus#indicnlp-news-article-classification-dataset) (Anoop Kunchukuttan, 2020) | ['guj', 'kan', 'mal', 'mar', 'ori', 'pan', 'tam', 'tel'] | Classification | s2s | [News, Written] | None | None | +| [IndicQARetrieval](https://arxiv.org/abs/2212.05409) (Sumanth Doddapaneni, 2022) | ['asm', 'ben', 'guj', 'hin', 'kan', 'mal', 'mar', 'ory', 'pan', 'tam', 'tel'] | Retrieval | s2p | [Web, Written] | None | None | +| [IndicReviewsClusteringP2P](https://arxiv.org/abs/2212.05409) (Sumanth Doddapaneni, 2022) | ['asm', 'ben', 'brx', 'guj', 'hin', 'kan', 'mal', 'mar', 'ory', 'pan', 'tam', 'tel', 'urd'] | Clustering | p2p | [Reviews, Written] | None | None | +| [IndicSentimentClassification](https://arxiv.org/abs/2212.05409) (Sumanth Doddapaneni, 2022) | ['asm', 'ben', 'brx', 'guj', 'hin', 'kan', 'mal', 'mar', 'ory', 'pan', 'tam', 'tel', 'urd'] | Classification | s2s | [Reviews, Written] | None | None | +| [IndonesianIdClickbaitClassification](http://www.sciencedirect.com/science/article/pii/S2352340920311252) | ['ind'] | Classification | s2s | [News, Written] | None | None | +| [IndonesianMongabayConservationClassification](https://aclanthology.org/2023.sealp-1.4/) | ['ind'] | Classification | s2s | [Web, Written] | None | None | +| [InsurancePolicyInterpretationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [InternationalCitizenshipQuestionsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [IsiZuluNewsClassification](https://huggingface.co/datasets/dsfsi/za-isizulu-siswati-news) (Madodonga et al., 2023) | ['zul'] | Classification | s2s | [News, Written] | None | None | +| [ItaCaseholdClassification](https://doi.org/10.1145/3594536.3595177) (Licari et al., 2023) | ['ita'] | Classification | s2s | [Legal, Government, Written] | None | None | +| [Itacola](https://aclanthology.org/2021.findings-emnlp.250/) | ['ita'] | Classification | s2s | [Non-fiction, Spoken, Written] | None | None | +| [JCrewBlockerLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | | [JDReview](https://aclanthology.org/2023.nodalida-1.20/) (Xiao et al., 2023) | ['cmn'] | Classification | s2s | | None | None | -| [JSICK](https://github.com/sbintuitions/JMTEB) (Yanaka et al., 2022) | ['jpn'] | STS | s2s | [Web, Written] | {'test': 1986} | {'test': 21.47} | -| [JSTS](https://aclanthology.org/2022.lrec-1.317.pdf#page=2.00) | ['jpn'] | STS | s2s | [Web, Written] | {'valudtion': 1457} | {'valudtion': 46.34} | -| [JaGovFaqsRetrieval](https://github.com/sbintuitions/JMTEB) | ['jpn'] | Retrieval | s2s | [Web, Written] | {'test': 2048} | {'test': {'average_document_length': 210.02601561814512, 'average_query_length': 59.48193359375, 'num_documents': 22794, 'num_queries': 2048, 'average_relevant_docs_per_query': 1.0}} | -| [JaQuADRetrieval](https://arxiv.org/abs/2202.01764) (ByungHoon So, 2022) | ['jpn'] | Retrieval | p2p | [Encyclopaedic, Non-fiction, Written] | {'validation': 2048} | {'validation': {'average_document_length': 155.80922362309224, 'average_query_length': 30.826171875, 'num_documents': 3014, 'num_queries': 2048, 'average_relevant_docs_per_query': 2.0}} | -| [JaqketRetrieval](https://github.com/kumapo/JAQKET-dataset) | ['jpn'] | Retrieval | s2p | [Encyclopaedic, Non-fiction, Written] | | | -| [JavaneseIMDBClassification](https://github.com/w11wo/nlp-datasets#javanese-imdb) (Wongso et al., 2021) | ['jav'] | Classification | s2s | [Reviews, Written] | {'test': 25000} | {'test': 481.83} | -| [KLUE-NLI](https://arxiv.org/abs/2105.09680) (Sungjoon Park, 2021) | ['kor'] | PairClassification | s2s | [News, Encyclopaedic, Written] | {'validation': 2000} | {'validation': 35.01} | -| [KLUE-STS](https://arxiv.org/abs/2105.09680) (Sungjoon Park, 2021) | ['kor'] | STS | s2s | [Reviews, News, Spoken, Written, Spoken] | {'validation': 519} | {'validation': 33.178227360308284} | -| [KLUE-TC](https://arxiv.org/abs/2105.09680) (Sungjoon Park, 2021) | ['kor'] | Classification | s2s | [News, Written] | {'validation': 2048} | {'validation': 27.079609091907326} | -| [KannadaNewsClassification](https://github.com/goru001/nlp-for-kannada) (Anoop Kunchukuttan, 2020) | ['kan'] | Classification | s2s | [News, Written] | {'train': 6460} | {'train': 65.88} | -| [KinopoiskClassification](https://www.dialog-21.ru/media/1226/blinovpd.pdf) (Blinov et al., 2013) | ['rus'] | Classification | p2p | [Reviews, Written] | {'test': 1500} | {'test': 1897.3} | -| Ko-StrategyQA (Geva et al., 2021) | ['kor'] | Retrieval | s2p | | None | {'dev': {'average_document_length': 319.25953950924225, 'average_query_length': 22.75337837837838, 'num_documents': 9251, 'num_queries': 592, 'average_relevant_docs_per_query': 1.9341216216216217}} | -| [KorFin](https://huggingface.co/datasets/amphora/korfin-asc) (Son et al., 2023) | ['kor'] | Classification | s2s | [News, Written] | {'test': 2048} | {'test': 75.28} | -| [KorHateClassification](https://paperswithcode.com/dataset/korean-hatespeech-dataset) (Jihyung Moon, 2020) | ['kor'] | Classification | s2s | [Social, Written] | {'train': 2048, 'test': 471} | {'train': 38.57, 'test': 38.86} | -| [KorHateSpeechMLClassification](https://paperswithcode.com/dataset/korean-multi-label-hate-speech-dataset) | ['kor'] | MultilabelClassification | s2s | [Social, Written] | {'train': 8192, 'test': 2048} | {'train': 33.67, 'test': 34.67} | -| [KorSTS](https://arxiv.org/abs/2004.03289) (Ham et al., 2020) | ['kor'] | STS | s2s | [News, Web] | {'test': 1379} | {'test': 29.279433139534884} | -| [KorSarcasmClassification](https://github.com/SpellOnYou/korean-sarcasm) (Kim et al., 2019) | ['kor'] | Classification | s2s | [Social, Written] | {'train': 2048, 'test': 301} | {'train': 48.45, 'test': 46.77} | -| [KurdishSentimentClassification](https://link.springer.com/article/10.1007/s10579-023-09716-6) (Badawi et al., 2024) | ['kur'] | Classification | s2s | [Web, Written] | {'train': 6000, 'test': 1987} | {'train': 59.38, 'test': 56.11} | +| [JSICK](https://github.com/sbintuitions/JMTEB) (Yanaka et al., 2022) | ['jpn'] | STS | s2s | [Web, Written] | None | None | +| [JSTS](https://aclanthology.org/2022.lrec-1.317.pdf#page=2.00) | ['jpn'] | STS | s2s | [Web, Written] | None | None | +| [JaGovFaqsRetrieval](https://github.com/sbintuitions/JMTEB) | ['jpn'] | Retrieval | s2s | [Web, Written] | None | None | +| [JaQuADRetrieval](https://arxiv.org/abs/2202.01764) (ByungHoon So, 2022) | ['jpn'] | Retrieval | p2p | [Encyclopaedic, Non-fiction, Written] | None | None | +| [JaqketRetrieval](https://github.com/kumapo/JAQKET-dataset) | ['jpn'] | Retrieval | s2p | [Encyclopaedic, Non-fiction, Written] | {'test': 115226} | {'test': {'number_of_characters': 3799.7, 'num_samples': 115226, 'num_queries': 997, 'num_documents': 114229, 'average_document_length': 0.03, 'average_query_length': 0.05, 'average_relevant_docs_per_query': 1.0}} | +| [JavaneseIMDBClassification](https://github.com/w11wo/nlp-datasets#javanese-imdb) (Wongso et al., 2021) | ['jav'] | Classification | s2s | [Reviews, Written] | None | None | +| [KLUE-NLI](https://arxiv.org/abs/2105.09680) (Sungjoon Park, 2021) | ['kor'] | PairClassification | s2s | [News, Encyclopaedic, Written] | None | None | +| [KLUE-STS](https://arxiv.org/abs/2105.09680) (Sungjoon Park, 2021) | ['kor'] | STS | s2s | [Reviews, News, Spoken, Written, Spoken] | None | None | +| [KLUE-TC](https://arxiv.org/abs/2105.09680) (Sungjoon Park, 2021) | ['kor'] | Classification | s2s | [News, Written] | None | None | +| [KannadaNewsClassification](https://github.com/goru001/nlp-for-kannada) (Anoop Kunchukuttan, 2020) | ['kan'] | Classification | s2s | [News, Written] | None | None | +| [KinopoiskClassification](https://www.dialog-21.ru/media/1226/blinovpd.pdf) (Blinov et al., 2013) | ['rus'] | Classification | p2p | [Reviews, Written] | None | None | +| Ko-StrategyQA (Geva et al., 2021) | ['kor'] | Retrieval | s2p | | None | None | +| [KorFin](https://huggingface.co/datasets/amphora/korfin-asc) (Son et al., 2023) | ['kor'] | Classification | s2s | [News, Written] | None | None | +| [KorHateClassification](https://paperswithcode.com/dataset/korean-hatespeech-dataset) (Jihyung Moon, 2020) | ['kor'] | Classification | s2s | [Social, Written] | None | None | +| [KorHateSpeechMLClassification](https://paperswithcode.com/dataset/korean-multi-label-hate-speech-dataset) | ['kor'] | MultilabelClassification | s2s | [Social, Written] | None | None | +| [KorSTS](https://arxiv.org/abs/2004.03289) (Ham et al., 2020) | ['kor'] | STS | s2s | [News, Web] | None | None | +| [KorSarcasmClassification](https://github.com/SpellOnYou/korean-sarcasm) (Kim et al., 2019) | ['kor'] | Classification | s2s | [Social, Written] | None | None | +| [KurdishSentimentClassification](https://link.springer.com/article/10.1007/s10579-023-09716-6) (Badawi et al., 2024) | ['kur'] | Classification | s2s | [Web, Written] | None | None | | [LCQMC](https://aclanthology.org/2021.emnlp-main.357) (Shitao Xiao, 2024) | ['cmn'] | STS | s2s | | None | None | -| [LEMBNarrativeQARetrieval](https://huggingface.co/datasets/dwzhu/LongEmbed) | ['eng'] | Retrieval | s2p | [Fiction, Non-fiction, Written] | {'test': 10804} | {'test': {'average_document_length': 326753.5323943662, 'average_query_length': 47.89453536223562, 'num_documents': 355, 'num_queries': 10449, 'average_relevant_docs_per_query': 1.0}} | -| [LEMBNeedleRetrieval](https://huggingface.co/datasets/dwzhu/LongEmbed) (Zhu et al., 2024) | ['eng'] | Retrieval | s2p | [Academic, Blog, Written] | {'test_256': 150, 'test_512': 150, 'test_1024': 150, 'test_2048': 150, 'test_4096': 150, 'test_8192': 150, 'test_16384': 150, 'test_32768': 150} | {'test_256': {'average_document_length': 1013.22, 'average_query_length': 60.48, 'num_documents': 100, 'num_queries': 50, 'average_relevant_docs_per_query': 1.0}, 'test_512': {'average_document_length': 2009.96, 'average_query_length': 57.3, 'num_documents': 100, 'num_queries': 50, 'average_relevant_docs_per_query': 1.0}, 'test_1024': {'average_document_length': 4069.9, 'average_query_length': 58.28, 'num_documents': 100, 'num_queries': 50, 'average_relevant_docs_per_query': 1.0}, 'test_2048': {'average_document_length': 8453.82, 'average_query_length': 59.92, 'num_documents': 100, 'num_queries': 50, 'average_relevant_docs_per_query': 1.0}, 'test_4096': {'average_document_length': 17395.8, 'average_query_length': 55.86, 'num_documents': 100, 'num_queries': 50, 'average_relevant_docs_per_query': 1.0}, 'test_8192': {'average_document_length': 35203.82, 'average_query_length': 59.6, 'num_documents': 100, 'num_queries': 50, 'average_relevant_docs_per_query': 1.0}, 'test_16384': {'average_document_length': 72054.8, 'average_query_length': 59.12, 'num_documents': 100, 'num_queries': 50, 'average_relevant_docs_per_query': 1.0}, 'test_32768': {'average_document_length': 141769.8, 'average_query_length': 58.34, 'num_documents': 100, 'num_queries': 50, 'average_relevant_docs_per_query': 1.0}} | -| [LEMBPasskeyRetrieval](https://huggingface.co/datasets/dwzhu/LongEmbed) (Zhu et al., 2024) | ['eng'] | Retrieval | s2p | [Fiction, Written] | {'test_256': 150, 'test_512': 150, 'test_1024': 150, 'test_2048': 150, 'test_4096': 150, 'test_8192': 150, 'test_16384': 150, 'test_32768': 150} | {'test_256': {'average_document_length': 876.24, 'average_query_length': 38.1, 'num_documents': 100, 'num_queries': 50, 'average_relevant_docs_per_query': 1.0}, 'test_512': {'average_document_length': 1785.2, 'average_query_length': 37.76, 'num_documents': 100, 'num_queries': 50, 'average_relevant_docs_per_query': 1.0}, 'test_1024': {'average_document_length': 3607.18, 'average_query_length': 37.68, 'num_documents': 100, 'num_queries': 50, 'average_relevant_docs_per_query': 1.0}, 'test_2048': {'average_document_length': 7242.2, 'average_query_length': 37.8, 'num_documents': 100, 'num_queries': 50, 'average_relevant_docs_per_query': 1.0}, 'test_4096': {'average_document_length': 14518.16, 'average_query_length': 37.64, 'num_documents': 100, 'num_queries': 50, 'average_relevant_docs_per_query': 1.0}, 'test_8192': {'average_document_length': 29071.16, 'average_query_length': 37.54, 'num_documents': 100, 'num_queries': 50, 'average_relevant_docs_per_query': 1.0}, 'test_16384': {'average_document_length': 58175.16, 'average_query_length': 38.12, 'num_documents': 100, 'num_queries': 50, 'average_relevant_docs_per_query': 1.0}, 'test_32768': {'average_document_length': 116380.16, 'average_query_length': 37.74, 'num_documents': 100, 'num_queries': 50, 'average_relevant_docs_per_query': 1.0}} | -| [LEMBQMSumRetrieval](https://huggingface.co/datasets/dwzhu/LongEmbed) | ['eng'] | Retrieval | s2p | [Spoken, Written] | {'test': 1724} | {'test': {'average_document_length': 53335.817258883246, 'average_query_length': 433.50294695481335, 'num_documents': 197, 'num_queries': 1527, 'average_relevant_docs_per_query': 1.0}} | -| [LEMBSummScreenFDRetrieval](https://huggingface.co/datasets/dwzhu/LongEmbed) | ['eng'] | Retrieval | s2p | [Spoken, Written] | {'validation': 672} | {'validation': {'average_document_length': 30854.32738095238, 'average_query_length': 591.4910714285714, 'num_documents': 336, 'num_queries': 336, 'average_relevant_docs_per_query': 1.0}} | -| [LEMBWikimQARetrieval](https://huggingface.co/datasets/dwzhu/LongEmbed) (Ho et al., 2020) | ['eng'] | Retrieval | s2p | [Encyclopaedic, Written] | {'test': 500} | {'test': {'average_document_length': 37445.60333333333, 'average_query_length': 67.57, 'num_documents': 300, 'num_queries': 300, 'average_relevant_docs_per_query': 1.0}} | -| [LanguageClassification](https://huggingface.co/datasets/papluca/language-identification) (Conneau et al., 2018) | ['ara', 'bul', 'cmn', 'deu', 'ell', 'eng', 'fra', 'hin', 'ita', 'jpn', 'nld', 'pol', 'por', 'rus', 'spa', 'swa', 'tha', 'tur', 'urd', 'vie'] | Classification | s2s | [Reviews, Web, Non-fiction, Fiction, Government, Written] | {'test': 2048} | {'test': {'num_samples': 2048, 'average_text_length': 109.546875, 'unique_labels': 20, 'labels': {'17': {'count': 102}, '0': {'count': 102}, '11': {'count': 102}, '4': {'count': 103}, '3': {'count': 102}, '1': {'count': 102}, '10': {'count': 102}, '2': {'count': 103}, '16': {'count': 103}, '9': {'count': 103}, '5': {'count': 102}, '7': {'count': 102}, '13': {'count': 102}, '14': {'count': 103}, '12': {'count': 102}, '15': {'count': 103}, '19': {'count': 102}, '18': {'count': 102}, '6': {'count': 103}, '8': {'count': 103}}}, 'train': {'num_samples': 70000, 'average_text_length': 110.86141428571429, 'unique_labels': 20, 'labels': {'12': {'count': 3500}, '1': {'count': 3500}, '19': {'count': 3500}, '15': {'count': 3500}, '13': {'count': 3500}, '11': {'count': 3500}, '17': {'count': 3500}, '14': {'count': 3500}, '16': {'count': 3500}, '5': {'count': 3500}, '0': {'count': 3500}, '8': {'count': 3500}, '7': {'count': 3500}, '2': {'count': 3500}, '3': {'count': 3500}, '10': {'count': 3500}, '6': {'count': 3500}, '18': {'count': 3500}, '4': {'count': 3500}, '9': {'count': 3500}}}} | -| [LccSentimentClassification](https://github.com/fnielsen/lcc-sentiment) | ['dan'] | Classification | s2s | [News, Web, Written] | {'test': 150} | {'test': 118.7} | -| [LeCaRDv2](https://github.com/THUIR/LeCaRDv2) (Haitao Li, 2023) | ['zho'] | Retrieval | p2p | [Legal, Written] | None | {'test': {'average_document_length': 7232.823978919631, 'average_query_length': 4259.440251572327, 'num_documents': 3795, 'num_queries': 159, 'average_relevant_docs_per_query': 24.50314465408805}} | -| [LearnedHandsBenefitsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 66} | {'test': 1308.44} | -| [LearnedHandsBusinessLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 174} | {'test': 1144.51} | -| [LearnedHandsConsumerLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 614} | {'test': 1277.45} | -| [LearnedHandsCourtsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 192} | {'test': 1171.02} | -| [LearnedHandsCrimeLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 688} | {'test': 1212.9} | -| [LearnedHandsDivorceLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 150} | {'test': 1242.43} | -| [LearnedHandsDomesticViolenceLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 174} | {'test': 1360.83} | -| [LearnedHandsEducationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 56} | {'test': 1397.44} | -| [LearnedHandsEmploymentLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 710} | {'test': 1262.74} | -| [LearnedHandsEstatesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 178} | {'test': 1200.7} | -| [LearnedHandsFamilyLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 2048} | {'test': 1338.27} | -| [LearnedHandsHealthLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 226} | {'test': 1472.59} | -| [LearnedHandsHousingLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 2048} | {'test': 1322.54} | -| [LearnedHandsImmigrationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 134} | {'test': 1216.31} | -| [LearnedHandsTortsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 432} | {'test': 1406.97} | -| [LearnedHandsTrafficLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 556} | {'test': 1182.91} | -| [LegalBenchConsumerContractsQA](https://huggingface.co/datasets/nguha/legalbench/viewer/consumer_contracts_qa) (Koreeda et al., 2021) | ['eng'] | Retrieval | s2p | [Legal, Written] | None | {'test': {'average_document_length': 2745.8246753246754, 'average_query_length': 92.4090909090909, 'num_documents': 154, 'num_queries': 396, 'average_relevant_docs_per_query': 1.0}} | -| [LegalBenchCorporateLobbying](https://huggingface.co/datasets/nguha/legalbench/viewer/corporate_lobbying) (Neel Guha, 2023) | ['eng'] | Retrieval | s2p | [Legal, Written] | None | {'test': {'average_document_length': 1157.2225705329154, 'average_query_length': 177.87941176470588, 'num_documents': 319, 'num_queries': 340, 'average_relevant_docs_per_query': 1.0}} | -| [LegalBenchPC](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | PairClassification | s2s | [Legal, Written] | {'test': 2048} | {'test': 287.18} | -| [LegalQuAD](https://github.com/Christoph911/AIKE2021_Appendix) (Hoppe et al., 2021) | ['deu'] | Retrieval | s2p | [Legal, Written] | None | {'test': {'average_document_length': 19481.955, 'average_query_length': 71.965, 'num_documents': 200, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}} | -| [LegalReasoningCausalityLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 55} | {'test': 1563.76} | -| [LegalSummarization](https://github.com/lauramanor/legal_summarization) | ['eng'] | Retrieval | s2p | [Legal, Written] | None | {'test': {'average_document_length': 606.1643835616438, 'average_query_length': 103.19014084507042, 'num_documents': 438, 'num_queries': 284, 'average_relevant_docs_per_query': 1.545774647887324}} | -| [LinceMTBitextMining](https://ritual.uh.edu/lince/) (Aguilar et al., 2020) | ['eng', 'hin'] | BitextMining | s2s | [Social, Written] | {'train': 8060} | {'train': 58.67} | -| [LitSearchRetrieval](https://github.com/princeton-nlp/LitSearch) (Ajith et al., 2024) | ['eng'] | Retrieval | s2p | [Academic, Non-fiction, Written] | {'test': 597} | {'test': {'average_document_length': 841.2769, 'average_query_length': 141.2, 'num_documents': 64183, 'num_queries': 597, 'average_relevant_docs_per_query': 1.070351}} | -| [LivedoorNewsClustering.v2](https://github.com/sbintuitions/JMTEB) | ['jpn'] | Clustering | s2s | [News, Written] | {'test': 1106} | {'test': 1082.61} | -| [MAUDLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 2048} | {'test': 1802.93} | -| [MIRACLReranking](https://project-miracl.github.io/) (Zhang et al., 2023) | ['ara', 'ben', 'deu', 'eng', 'fas', 'fin', 'fra', 'hin', 'ind', 'jpn', 'kor', 'rus', 'spa', 'swa', 'tel', 'tha', 'yor', 'zho'] | Reranking | s2s | [Encyclopaedic, Written] | {'dev': 44608} | {'dev': 506.3} | -| [MIRACLRetrieval](http://miracl.ai/) (Zhang et al., 2023) | ['ara', 'ben', 'deu', 'eng', 'fas', 'fin', 'fra', 'hin', 'ind', 'jpn', 'kor', 'rus', 'spa', 'swa', 'tel', 'tha', 'yor', 'zho'] | Retrieval | s2p | [Encyclopaedic, Written] | None | {'dev': {'ar': {'average_document_length': 318.6539598547405, 'average_query_length': 29.480662983425415, 'num_documents': 2061414, 'num_queries': 2896, 'average_relevant_docs_per_query': 1.953729281767956}, 'bn': {'average_document_length': 383.2428136511194, 'average_query_length': 46.98053527980535, 'num_documents': 297265, 'num_queries': 411, 'average_relevant_docs_per_query': 2.099756690997567}, 'de': {'average_document_length': 414.28004442393404, 'average_query_length': 46.0, 'num_documents': 15866222, 'num_queries': 305, 'average_relevant_docs_per_query': 2.6590163934426227}, 'en': {'average_document_length': 401.0042914921588, 'average_query_length': 40.247809762202756, 'num_documents': 32893221, 'num_queries': 799, 'average_relevant_docs_per_query': 2.911138923654568}, 'es': {'average_document_length': 403.71153493754986, 'average_query_length': 47.373456790123456, 'num_documents': 10373953, 'num_queries': 648, 'average_relevant_docs_per_query': 4.609567901234568}, 'fa': {'average_document_length': 262.6478385010321, 'average_query_length': 41.1503164556962, 'num_documents': 2207172, 'num_queries': 632, 'average_relevant_docs_per_query': 2.079113924050633}, 'fi': {'average_document_length': 359.87767671935734, 'average_query_length': 38.63493312352478, 'num_documents': 1883509, 'num_queries': 1271, 'average_relevant_docs_per_query': 1.925255704169945}, 'fr': {'average_document_length': 343.6283550271699, 'average_query_length': 43.883381924198254, 'num_documents': 14636953, 'num_queries': 343, 'average_relevant_docs_per_query': 2.131195335276968}, 'hi': {'average_document_length': 370.96196845914386, 'average_query_length': 53.34, 'num_documents': 506264, 'num_queries': 350, 'average_relevant_docs_per_query': 2.1485714285714286}, 'id': {'average_document_length': 350.2785651811673, 'average_query_length': 37.958333333333336, 'num_documents': 1446315, 'num_queries': 960, 'average_relevant_docs_per_query': 3.216666666666667}, 'ja': {'average_document_length': 145.8538220556965, 'average_query_length': 17.71395348837209, 'num_documents': 6953614, 'num_queries': 860, 'average_relevant_docs_per_query': 2.0813953488372094}, 'ko': {'average_document_length': 173.97649170809927, 'average_query_length': 21.624413145539908, 'num_documents': 1486752, 'num_queries': 213, 'average_relevant_docs_per_query': 2.568075117370892}, 'ru': {'average_document_length': 332.2475377512674, 'average_query_length': 44.13258785942492, 'num_documents': 9543918, 'num_queries': 1252, 'average_relevant_docs_per_query': 2.8434504792332267}, 'sw': {'average_document_length': 228.71348655286377, 'average_query_length': 38.97095435684647, 'num_documents': 131924, 'num_queries': 482, 'average_relevant_docs_per_query': 1.887966804979253}, 'te': {'average_document_length': 396.2108674545774, 'average_query_length': 38.11231884057971, 'num_documents': 518079, 'num_queries': 828, 'average_relevant_docs_per_query': 1.0314009661835748}, 'th': {'average_document_length': 356.8283496198581, 'average_query_length': 42.87585266030014, 'num_documents': 542166, 'num_queries': 733, 'average_relevant_docs_per_query': 1.8321964529331514}, 'yo': {'average_document_length': 159.35250698366738, 'average_query_length': 37.6890756302521, 'num_documents': 49043, 'num_queries': 119, 'average_relevant_docs_per_query': 1.2100840336134453}, 'zh': {'average_document_length': 119.9458931721347, 'average_query_length': 10.867684478371501, 'num_documents': 4934368, 'num_queries': 393, 'average_relevant_docs_per_query': 2.5292620865139948}}} | -| [MIRACLRetrievalHardNegatives](http://miracl.ai/) (Zhang et al., 2023) | ['ara', 'ben', 'deu', 'eng', 'fas', 'fin', 'fra', 'hin', 'ind', 'jpn', 'kor', 'rus', 'spa', 'swa', 'tel', 'tha', 'yor', 'zho'] | Retrieval | s2p | [Encyclopaedic, Written] | None | {'dev': {'average_document_length': 417.6655323669399, 'average_query_length': 37.46957385337667, 'num_documents': 2449382, 'num_queries': 11076, 'average_relevant_docs_per_query': 2.3643011917659806, 'hf_subset_descriptive_stats': {'ar': {'average_document_length': 438.1872433017704, 'average_query_length': 29.584, 'num_documents': 192103, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.982}, 'bn': {'average_document_length': 383.2428136511194, 'average_query_length': 46.98053527980535, 'num_documents': 297265, 'num_queries': 411, 'average_relevant_docs_per_query': 2.099756690997567}, 'de': {'average_document_length': 513.7796484139344, 'average_query_length': 46.0, 'num_documents': 71277, 'num_queries': 305, 'average_relevant_docs_per_query': 2.6590163934426227}, 'en': {'average_document_length': 529.2486406963214, 'average_query_length': 40.247809762202756, 'num_documents': 178768, 'num_queries': 799, 'average_relevant_docs_per_query': 2.911138923654568}, 'es': {'average_document_length': 535.8023645655877, 'average_query_length': 47.373456790123456, 'num_documents': 146750, 'num_queries': 648, 'average_relevant_docs_per_query': 4.609567901234568}, 'fa': {'average_document_length': 411.2648282882721, 'average_query_length': 41.1503164556962, 'num_documents': 133596, 'num_queries': 632, 'average_relevant_docs_per_query': 2.079113924050633}, 'fi': {'average_document_length': 462.9445310289844, 'average_query_length': 38.646, 'num_documents': 194415, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.918}, 'fr': {'average_document_length': 460.40909271865917, 'average_query_length': 43.883381924198254, 'num_documents': 75357, 'num_queries': 343, 'average_relevant_docs_per_query': 2.131195335276968}, 'hi': {'average_document_length': 498.6759426632417, 'average_query_length': 53.34, 'num_documents': 63066, 'num_queries': 350, 'average_relevant_docs_per_query': 2.1485714285714286}, 'id': {'average_document_length': 494.1689807519638, 'average_query_length': 37.958333333333336, 'num_documents': 168173, 'num_queries': 960, 'average_relevant_docs_per_query': 3.216666666666667}, 'ja': {'average_document_length': 206.13654293407583, 'average_query_length': 17.71395348837209, 'num_documents': 185319, 'num_queries': 860, 'average_relevant_docs_per_query': 2.0813953488372094}, 'ko': {'average_document_length': 257.82646155267594, 'average_query_length': 21.624413145539908, 'num_documents': 43293, 'num_queries': 213, 'average_relevant_docs_per_query': 2.568075117370892}, 'ru': {'average_document_length': 476.0820349224605, 'average_query_length': 44.055, 'num_documents': 219114, 'num_queries': 1000, 'average_relevant_docs_per_query': 2.833}, 'sw': {'average_document_length': 228.71348655286377, 'average_query_length': 38.97095435684647, 'num_documents': 131924, 'num_queries': 482, 'average_relevant_docs_per_query': 1.887966804979253}, 'te': {'average_document_length': 601.7099283059209, 'average_query_length': 38.11231884057971, 'num_documents': 101961, 'num_queries': 828, 'average_relevant_docs_per_query': 1.0314009661835748}, 'th': {'average_document_length': 478.8818849711528, 'average_query_length': 42.87585266030014, 'num_documents': 116649, 'num_queries': 733, 'average_relevant_docs_per_query': 1.8321964529331514}, 'yo': {'average_document_length': 159.35250698366738, 'average_query_length': 37.6890756302521, 'num_documents': 49043, 'num_queries': 119, 'average_relevant_docs_per_query': 1.2100840336134453}, 'zh': {'average_document_length': 147.36211243527777, 'average_query_length': 10.867684478371501, 'num_documents': 81309, 'num_queries': 393, 'average_relevant_docs_per_query': 2.5292620865139948}}}} | -| [MLQARetrieval](https://huggingface.co/datasets/mlqa) | ['ara', 'deu', 'eng', 'hin', 'spa', 'vie', 'zho'] | Retrieval | s2p | [Encyclopaedic, Written] | {'test': 158083, 'validation': 15747} | {'validation': {'ara-ara': {'average_document_length': 693.8883826879271, 'average_query_length': 42.321083172147, 'num_documents': 439, 'num_queries': 517, 'average_relevant_docs_per_query': 1.0}, 'ara-deu': {'average_document_length': 759.3882352941176, 'average_query_length': 55.14492753623188, 'num_documents': 170, 'num_queries': 207, 'average_relevant_docs_per_query': 1.0}, 'ara-eng': {'average_document_length': 693.8883826879271, 'average_query_length': 50.029013539651835, 'num_documents': 439, 'num_queries': 517, 'average_relevant_docs_per_query': 1.0}, 'ara-spa': {'average_document_length': 654.3071428571428, 'average_query_length': 53.68944099378882, 'num_documents': 140, 'num_queries': 161, 'average_relevant_docs_per_query': 1.0}, 'ara-hin': {'average_document_length': 626.5935483870968, 'average_query_length': 51.956989247311824, 'num_documents': 155, 'num_queries': 186, 'average_relevant_docs_per_query': 1.0}, 'ara-vie': {'average_document_length': 804.6216216216217, 'average_query_length': 49.57055214723926, 'num_documents': 148, 'num_queries': 163, 'average_relevant_docs_per_query': 1.0}, 'ara-zho': {'average_document_length': 787.3161290322581, 'average_query_length': 15.617021276595745, 'num_documents': 155, 'num_queries': 188, 'average_relevant_docs_per_query': 1.0}, 'deu-ara': {'average_document_length': 702.1675977653631, 'average_query_length': 43.06280193236715, 'num_documents': 179, 'num_queries': 207, 'average_relevant_docs_per_query': 1.0}, 'deu-deu': {'average_document_length': 721.405701754386, 'average_query_length': 52.572265625, 'num_documents': 456, 'num_queries': 512, 'average_relevant_docs_per_query': 1.0}, 'deu-eng': {'average_document_length': 721.405701754386, 'average_query_length': 48.33984375, 'num_documents': 456, 'num_queries': 512, 'average_relevant_docs_per_query': 1.0}, 'deu-spa': {'average_document_length': 677.2762430939226, 'average_query_length': 50.60204081632653, 'num_documents': 181, 'num_queries': 196, 'average_relevant_docs_per_query': 1.0}, 'deu-hin': {'average_document_length': 685.917808219178, 'average_query_length': 47.01840490797546, 'num_documents': 146, 'num_queries': 163, 'average_relevant_docs_per_query': 1.0}, 'deu-vie': {'average_document_length': 921.6196319018405, 'average_query_length': 46.81868131868132, 'num_documents': 163, 'num_queries': 182, 'average_relevant_docs_per_query': 1.0}, 'deu-zho': {'average_document_length': 736.6347305389221, 'average_query_length': 14.936842105263159, 'num_documents': 167, 'num_queries': 190, 'average_relevant_docs_per_query': 1.0}, 'eng-ara': {'average_document_length': 979.3447488584475, 'average_query_length': 42.321083172147, 'num_documents': 438, 'num_queries': 517, 'average_relevant_docs_per_query': 1.0}, 'eng-deu': {'average_document_length': 947.3109619686801, 'average_query_length': 52.572265625, 'num_documents': 447, 'num_queries': 512, 'average_relevant_docs_per_query': 1.0}, 'eng-eng': {'average_document_length': 940.2842535787321, 'average_query_length': 49.01480836236934, 'num_documents': 978, 'num_queries': 1148, 'average_relevant_docs_per_query': 1.0}, 'eng-spa': {'average_document_length': 904.3166287015945, 'average_query_length': 52.146, 'num_documents': 439, 'num_queries': 500, 'average_relevant_docs_per_query': 1.0}, 'eng-hin': {'average_document_length': 926.9621749408983, 'average_query_length': 49.3905325443787, 'num_documents': 423, 'num_queries': 507, 'average_relevant_docs_per_query': 1.0}, 'eng-vie': {'average_document_length': 1011.8296460176991, 'average_query_length': 48.082191780821915, 'num_documents': 452, 'num_queries': 511, 'average_relevant_docs_per_query': 1.0}, 'eng-zho': {'average_document_length': 1001.5046511627907, 'average_query_length': 15.39484126984127, 'num_documents': 430, 'num_queries': 504, 'average_relevant_docs_per_query': 1.0}, 'spa-ara': {'average_document_length': 674.3586206896551, 'average_query_length': 41.36024844720497, 'num_documents': 145, 'num_queries': 161, 'average_relevant_docs_per_query': 1.0}, 'spa-deu': {'average_document_length': 544.0489130434783, 'average_query_length': 51.86734693877551, 'num_documents': 184, 'num_queries': 196, 'average_relevant_docs_per_query': 1.0}, 'spa-eng': {'average_document_length': 641.8215859030837, 'average_query_length': 49.156, 'num_documents': 454, 'num_queries': 500, 'average_relevant_docs_per_query': 1.0}, 'spa-spa': {'average_document_length': 641.8215859030837, 'average_query_length': 52.146, 'num_documents': 454, 'num_queries': 500, 'average_relevant_docs_per_query': 1.0}, 'spa-hin': {'average_document_length': 703.3212121212122, 'average_query_length': 48.080213903743314, 'num_documents': 165, 'num_queries': 187, 'average_relevant_docs_per_query': 1.0}, 'spa-vie': {'average_document_length': 737.8579545454545, 'average_query_length': 48.82539682539682, 'num_documents': 176, 'num_queries': 189, 'average_relevant_docs_per_query': 1.0}, 'spa-zho': {'average_document_length': 605.52, 'average_query_length': 15.590062111801242, 'num_documents': 150, 'num_queries': 161, 'average_relevant_docs_per_query': 1.0}, 'hin-ara': {'average_document_length': 670.0394736842105, 'average_query_length': 43.623655913978496, 'num_documents': 152, 'num_queries': 186, 'average_relevant_docs_per_query': 1.0}, 'hin-deu': {'average_document_length': 596.9718309859155, 'average_query_length': 51.41717791411043, 'num_documents': 142, 'num_queries': 163, 'average_relevant_docs_per_query': 1.0}, 'hin-eng': {'average_document_length': 691.5482352941176, 'average_query_length': 49.75936883629191, 'num_documents': 425, 'num_queries': 507, 'average_relevant_docs_per_query': 1.0}, 'hin-spa': {'average_document_length': 718.4904458598726, 'average_query_length': 52.75935828877005, 'num_documents': 157, 'num_queries': 187, 'average_relevant_docs_per_query': 1.0}, 'hin-hin': {'average_document_length': 691.5482352941176, 'average_query_length': 49.3905325443787, 'num_documents': 425, 'num_queries': 507, 'average_relevant_docs_per_query': 1.0}, 'hin-vie': {'average_document_length': 778.484076433121, 'average_query_length': 48.35028248587571, 'num_documents': 157, 'num_queries': 177, 'average_relevant_docs_per_query': 1.0}, 'hin-zho': {'average_document_length': 685.0679012345679, 'average_query_length': 15.97883597883598, 'num_documents': 162, 'num_queries': 189, 'average_relevant_docs_per_query': 1.0}, 'vie-ara': {'average_document_length': 886.6052631578947, 'average_query_length': 41.214723926380366, 'num_documents': 152, 'num_queries': 163, 'average_relevant_docs_per_query': 1.0}, 'vie-deu': {'average_document_length': 981.4534161490683, 'average_query_length': 51.27472527472528, 'num_documents': 161, 'num_queries': 182, 'average_relevant_docs_per_query': 1.0}, 'vie-eng': {'average_document_length': 892.7250554323725, 'average_query_length': 48.09001956947162, 'num_documents': 451, 'num_queries': 511, 'average_relevant_docs_per_query': 1.0}, 'vie-spa': {'average_document_length': 936.6746987951807, 'average_query_length': 51.851851851851855, 'num_documents': 166, 'num_queries': 189, 'average_relevant_docs_per_query': 1.0}, 'vie-hin': {'average_document_length': 869.0509554140127, 'average_query_length': 46.44632768361582, 'num_documents': 157, 'num_queries': 177, 'average_relevant_docs_per_query': 1.0}, 'vie-vie': {'average_document_length': 892.7250554323725, 'average_query_length': 48.082191780821915, 'num_documents': 451, 'num_queries': 511, 'average_relevant_docs_per_query': 1.0}, 'vie-zho': {'average_document_length': 960.7349397590361, 'average_query_length': 15.048913043478262, 'num_documents': 166, 'num_queries': 184, 'average_relevant_docs_per_query': 1.0}, 'zho-ara': {'average_document_length': 238.75155279503105, 'average_query_length': 44.34574468085106, 'num_documents': 161, 'num_queries': 188, 'average_relevant_docs_per_query': 1.0}, 'zho-deu': {'average_document_length': 257.109756097561, 'average_query_length': 53.84736842105263, 'num_documents': 164, 'num_queries': 190, 'average_relevant_docs_per_query': 1.0}, 'zho-eng': {'average_document_length': 246.65237020316027, 'average_query_length': 50.15079365079365, 'num_documents': 443, 'num_queries': 504, 'average_relevant_docs_per_query': 1.0}, 'zho-spa': {'average_document_length': 249.6081081081081, 'average_query_length': 52.857142857142854, 'num_documents': 148, 'num_queries': 161, 'average_relevant_docs_per_query': 1.0}, 'zho-hin': {'average_document_length': 238.5521472392638, 'average_query_length': 52.05291005291005, 'num_documents': 163, 'num_queries': 189, 'average_relevant_docs_per_query': 1.0}, 'zho-vie': {'average_document_length': 268.32142857142856, 'average_query_length': 49.33695652173913, 'num_documents': 168, 'num_queries': 184, 'average_relevant_docs_per_query': 1.0}, 'zho-zho': {'average_document_length': 246.65237020316027, 'average_query_length': 15.39484126984127, 'num_documents': 443, 'num_queries': 504, 'average_relevant_docs_per_query': 1.0}}, 'test': {'ara-ara': {'average_document_length': 698.5714593198451, 'average_query_length': 41.26176636039752, 'num_documents': 4646, 'num_queries': 5333, 'average_relevant_docs_per_query': 1.000375023438965}, 'ara-deu': {'average_document_length': 592.5728542914171, 'average_query_length': 51.27730582524272, 'num_documents': 1503, 'num_queries': 1648, 'average_relevant_docs_per_query': 1.0006067961165048}, 'ara-eng': {'average_document_length': 698.5714593198451, 'average_query_length': 48.556451612903224, 'num_documents': 4646, 'num_queries': 5332, 'average_relevant_docs_per_query': 1.000562640660165}, 'ara-spa': {'average_document_length': 713.4833239118146, 'average_query_length': 51.406471183013146, 'num_documents': 1769, 'num_queries': 1978, 'average_relevant_docs_per_query': 1.0}, 'ara-hin': {'average_document_length': 702.1388888888889, 'average_query_length': 48.71818678317859, 'num_documents': 1512, 'num_queries': 1831, 'average_relevant_docs_per_query': 1.0}, 'ara-vie': {'average_document_length': 745.4528096017458, 'average_query_length': 48.815828041035665, 'num_documents': 1833, 'num_queries': 2047, 'average_relevant_docs_per_query': 1.0}, 'ara-zho': {'average_document_length': 774.4593639575971, 'average_query_length': 14.985355648535565, 'num_documents': 1698, 'num_queries': 1912, 'average_relevant_docs_per_query': 1.0}, 'deu-ara': {'average_document_length': 719.6800267201069, 'average_query_length': 39.54578532443905, 'num_documents': 1497, 'num_queries': 1649, 'average_relevant_docs_per_query': 1.0}, 'deu-deu': {'average_document_length': 725.5304712558599, 'average_query_length': 51.610680257035234, 'num_documents': 4053, 'num_queries': 4513, 'average_relevant_docs_per_query': 1.0008863283846665}, 'deu-eng': {'average_document_length': 725.5304712558599, 'average_query_length': 47.07777531575449, 'num_documents': 4053, 'num_queries': 4513, 'average_relevant_docs_per_query': 1.0008863283846665}, 'deu-spa': {'average_document_length': 740.5414052697616, 'average_query_length': 50.098591549295776, 'num_documents': 1594, 'num_queries': 1775, 'average_relevant_docs_per_query': 1.0005633802816902}, 'deu-hin': {'average_document_length': 674.3714063714064, 'average_query_length': 45.146153846153844, 'num_documents': 1287, 'num_queries': 1430, 'average_relevant_docs_per_query': 1.0}, 'deu-vie': {'average_document_length': 760.1198945981555, 'average_query_length': 46.64358208955224, 'num_documents': 1518, 'num_queries': 1675, 'average_relevant_docs_per_query': 1.0}, 'deu-zho': {'average_document_length': 771.3367697594501, 'average_query_length': 14.942592592592593, 'num_documents': 1455, 'num_queries': 1620, 'average_relevant_docs_per_query': 1.0006172839506173}, 'eng-ara': {'average_document_length': 1008.3584455058619, 'average_query_length': 41.26176636039752, 'num_documents': 4606, 'num_queries': 5333, 'average_relevant_docs_per_query': 1.000375023438965}, 'eng-deu': {'average_document_length': 910.3226686507936, 'average_query_length': 51.610680257035234, 'num_documents': 4032, 'num_queries': 4513, 'average_relevant_docs_per_query': 1.0008863283846665}, 'eng-eng': {'average_document_length': 983.0993344090359, 'average_query_length': 47.960714902434816, 'num_documents': 9916, 'num_queries': 11582, 'average_relevant_docs_per_query': 1.000690726990157}, 'eng-spa': {'average_document_length': 967.4622376109068, 'average_query_length': 50.923252713768804, 'num_documents': 4621, 'num_queries': 5251, 'average_relevant_docs_per_query': 1.000380879832413}, 'eng-hin': {'average_document_length': 986.0465631929046, 'average_query_length': 47.328315703824245, 'num_documents': 4059, 'num_queries': 4916, 'average_relevant_docs_per_query': 1.000406834825061}, 'eng-vie': {'average_document_length': 1048.6062197940744, 'average_query_length': 48.094085532302095, 'num_documents': 4759, 'num_queries': 5495, 'average_relevant_docs_per_query': 1.0}, 'eng-zho': {'average_document_length': 1063.8536257833482, 'average_query_length': 15.019080996884735, 'num_documents': 4468, 'num_queries': 5136, 'average_relevant_docs_per_query': 1.0001947040498442}, 'spa-ara': {'average_document_length': 645.5182320441988, 'average_query_length': 40.78412537917088, 'num_documents': 1810, 'num_queries': 1978, 'average_relevant_docs_per_query': 1.0}, 'spa-deu': {'average_document_length': 586.6057810578105, 'average_query_length': 51.870913190529876, 'num_documents': 1626, 'num_queries': 1774, 'average_relevant_docs_per_query': 1.0011273957158964}, 'spa-eng': {'average_document_length': 630.6735979836169, 'average_query_length': 47.827907862173994, 'num_documents': 4761, 'num_queries': 5253, 'average_relevant_docs_per_query': 1.0}, 'spa-spa': {'average_document_length': 630.6735979836169, 'average_query_length': 50.923252713768804, 'num_documents': 4761, 'num_queries': 5251, 'average_relevant_docs_per_query': 1.000380879832413}, 'spa-hin': {'average_document_length': 613.3478260869565, 'average_query_length': 46.36680208937899, 'num_documents': 1518, 'num_queries': 1723, 'average_relevant_docs_per_query': 1.0}, 'spa-vie': {'average_document_length': 659.6179295624333, 'average_query_length': 48.1595639246779, 'num_documents': 1874, 'num_queries': 2018, 'average_relevant_docs_per_query': 1.0}, 'spa-zho': {'average_document_length': 668.6646171045277, 'average_query_length': 15.115562403697997, 'num_documents': 1789, 'num_queries': 1947, 'average_relevant_docs_per_query': 1.0}, 'hin-ara': {'average_document_length': 765.0352862849534, 'average_query_length': 42.04642271982523, 'num_documents': 1502, 'num_queries': 1831, 'average_relevant_docs_per_query': 1.0}, 'hin-deu': {'average_document_length': 719.676862745098, 'average_query_length': 51.002799160251925, 'num_documents': 1275, 'num_queries': 1429, 'average_relevant_docs_per_query': 1.000699790062981}, 'hin-eng': {'average_document_length': 760.9956086850451, 'average_query_length': 47.91232709519935, 'num_documents': 4099, 'num_queries': 4916, 'average_relevant_docs_per_query': 1.000406834825061}, 'hin-spa': {'average_document_length': 753.5010281014394, 'average_query_length': 50.46689895470383, 'num_documents': 1459, 'num_queries': 1722, 'average_relevant_docs_per_query': 1.0005807200929153}, 'hin-hin': {'average_document_length': 760.9956086850451, 'average_query_length': 47.328315703824245, 'num_documents': 4099, 'num_queries': 4916, 'average_relevant_docs_per_query': 1.000406834825061}, 'hin-vie': {'average_document_length': 789.9253822629969, 'average_query_length': 48.21160760143811, 'num_documents': 1635, 'num_queries': 1947, 'average_relevant_docs_per_query': 1.0}, 'hin-zho': {'average_document_length': 834.2057448229793, 'average_query_length': 15.101301641199774, 'num_documents': 1497, 'num_queries': 1767, 'average_relevant_docs_per_query': 1.0}, 'vie-ara': {'average_document_length': 992.2129527991218, 'average_query_length': 41.82462139716659, 'num_documents': 1822, 'num_queries': 2047, 'average_relevant_docs_per_query': 1.0}, 'vie-deu': {'average_document_length': 861.0610079575597, 'average_query_length': 51.58721624850657, 'num_documents': 1508, 'num_queries': 1674, 'average_relevant_docs_per_query': 1.0005973715651135}, 'vie-eng': {'average_document_length': 913.8633993743483, 'average_query_length': 48.11086837793555, 'num_documents': 4795, 'num_queries': 5493, 'average_relevant_docs_per_query': 1.0003640997633352}, 'vie-spa': {'average_document_length': 940.0322580645161, 'average_query_length': 51.13386217154189, 'num_documents': 1829, 'num_queries': 2017, 'average_relevant_docs_per_query': 1.0004957858205255}, 'vie-hin': {'average_document_length': 838.1713414634146, 'average_query_length': 47.484334874165384, 'num_documents': 1640, 'num_queries': 1947, 'average_relevant_docs_per_query': 1.0}, 'vie-vie': {'average_document_length': 913.8633993743483, 'average_query_length': 48.094085532302095, 'num_documents': 4795, 'num_queries': 5495, 'average_relevant_docs_per_query': 1.0}, 'vie-zho': {'average_document_length': 999.064534883721, 'average_query_length': 15.045805455481215, 'num_documents': 1720, 'num_queries': 1943, 'average_relevant_docs_per_query': 1.0}, 'zho-ara': {'average_document_length': 253.71303841676368, 'average_query_length': 42.04866562009419, 'num_documents': 1718, 'num_queries': 1911, 'average_relevant_docs_per_query': 1.000523286237572}, 'zho-deu': {'average_document_length': 241.84631147540983, 'average_query_length': 52.25107958050586, 'num_documents': 1464, 'num_queries': 1621, 'average_relevant_docs_per_query': 1.0}, 'zho-eng': {'average_document_length': 247.55609326880776, 'average_query_length': 48.64167478091529, 'num_documents': 4546, 'num_queries': 5135, 'average_relevant_docs_per_query': 1.0003894839337877}, 'zho-spa': {'average_document_length': 254.44552196235026, 'average_query_length': 51.90446841294299, 'num_documents': 1753, 'num_queries': 1947, 'average_relevant_docs_per_query': 1.0}, 'zho-hin': {'average_document_length': 229.60590163934427, 'average_query_length': 49.06625141562854, 'num_documents': 1525, 'num_queries': 1766, 'average_relevant_docs_per_query': 1.0005662514156286}, 'zho-vie': {'average_document_length': 266.1140401146132, 'average_query_length': 49.27328872876994, 'num_documents': 1745, 'num_queries': 1943, 'average_relevant_docs_per_query': 1.0}, 'zho-zho': {'average_document_length': 247.55609326880776, 'average_query_length': 15.019080996884735, 'num_documents': 4546, 'num_queries': 5136, 'average_relevant_docs_per_query': 1.0001947040498442}}} | -| [MLQuestions](https://github.com/McGill-NLP/MLQuestions) | ['eng'] | Retrieval | s2p | [Encyclopaedic, Academic, Written] | {'dev': 1500, 'test': 1500} | {'dev': {'average_document_length': 258.8772727272727, 'average_query_length': 45.05533333333333, 'num_documents': 11000, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}, 'test': {'average_document_length': 258.8772727272727, 'average_query_length': 45.75333333333333, 'num_documents': 11000, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}} | -| [MLSUMClusteringP2P.v2](https://huggingface.co/datasets/mteb/mlsum) (Scialom et al., 2020) | ['deu', 'fra', 'rus', 'spa'] | Clustering | p2p | [News, Written] | {'validation': 2048, 'test': 2048} | {'validation': 4613, 'test': 4810} | -| [MLSUMClusteringS2S.v2](https://huggingface.co/datasets/mteb/mlsum) (Scialom et al., 2020) | ['deu', 'fra', 'rus', 'spa'] | Clustering | s2s | [News, Written] | {'validation': 750, 'test': 756} | {'validation': 4613, 'test': 4810} | +| [LEMBNarrativeQARetrieval](https://huggingface.co/datasets/dwzhu/LongEmbed) | ['eng'] | Retrieval | s2p | [Fiction, Non-fiction, Written] | None | None | +| [LEMBNeedleRetrieval](https://huggingface.co/datasets/dwzhu/LongEmbed) (Zhu et al., 2024) | ['eng'] | Retrieval | s2p | [Academic, Blog, Written] | None | None | +| [LEMBPasskeyRetrieval](https://huggingface.co/datasets/dwzhu/LongEmbed) (Zhu et al., 2024) | ['eng'] | Retrieval | s2p | [Fiction, Written] | None | None | +| [LEMBQMSumRetrieval](https://huggingface.co/datasets/dwzhu/LongEmbed) | ['eng'] | Retrieval | s2p | [Spoken, Written] | None | None | +| [LEMBSummScreenFDRetrieval](https://huggingface.co/datasets/dwzhu/LongEmbed) | ['eng'] | Retrieval | s2p | [Spoken, Written] | None | None | +| [LEMBWikimQARetrieval](https://huggingface.co/datasets/dwzhu/LongEmbed) (Ho et al., 2020) | ['eng'] | Retrieval | s2p | [Encyclopaedic, Written] | None | None | +| [LanguageClassification](https://huggingface.co/datasets/papluca/language-identification) (Conneau et al., 2018) | ['ara', 'bul', 'cmn', 'deu', 'ell', 'eng', 'fra', 'hin', 'ita', 'jpn', 'nld', 'pol', 'por', 'rus', 'spa', 'swa', 'tha', 'tur', 'urd', 'vie'] | Classification | s2s | [Reviews, Web, Non-fiction, Fiction, Government, Written] | {'test': 2048} | {'test': {'num_samples': 2048, 'number_of_characters': 224352, 'average_text_length': 109.55, 'unique_labels': 20, 'labels': {'17': {'count': 102}, '0': {'count': 102}, '11': {'count': 102}, '4': {'count': 103}, '3': {'count': 102}, '1': {'count': 102}, '10': {'count': 102}, '2': {'count': 103}, '16': {'count': 103}, '9': {'count': 103}, '5': {'count': 102}, '7': {'count': 102}, '13': {'count': 102}, '14': {'count': 103}, '12': {'count': 102}, '15': {'count': 103}, '19': {'count': 102}, '18': {'count': 102}, '6': {'count': 103}, '8': {'count': 103}}}} | +| [LccSentimentClassification](https://github.com/fnielsen/lcc-sentiment) | ['dan'] | Classification | s2s | [News, Web, Written] | None | None | +| [LeCaRDv2](https://github.com/THUIR/LeCaRDv2) (Haitao Li, 2023) | ['zho'] | Retrieval | p2p | [Legal, Written] | None | None | +| [LearnedHandsBenefitsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LearnedHandsBusinessLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LearnedHandsConsumerLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LearnedHandsCourtsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LearnedHandsCrimeLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LearnedHandsDivorceLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LearnedHandsDomesticViolenceLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LearnedHandsEducationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LearnedHandsEmploymentLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LearnedHandsEstatesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LearnedHandsFamilyLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LearnedHandsHealthLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LearnedHandsHousingLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LearnedHandsImmigrationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LearnedHandsTortsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LearnedHandsTrafficLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LegalBenchConsumerContractsQA](https://huggingface.co/datasets/nguha/legalbench/viewer/consumer_contracts_qa) (Koreeda et al., 2021) | ['eng'] | Retrieval | s2p | [Legal, Written] | None | None | +| [LegalBenchCorporateLobbying](https://huggingface.co/datasets/nguha/legalbench/viewer/corporate_lobbying) (Neel Guha, 2023) | ['eng'] | Retrieval | s2p | [Legal, Written] | None | None | +| [LegalBenchPC](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | PairClassification | s2s | [Legal, Written] | None | None | +| [LegalQuAD](https://github.com/Christoph911/AIKE2021_Appendix) (Hoppe et al., 2021) | ['deu'] | Retrieval | s2p | [Legal, Written] | None | None | +| [LegalReasoningCausalityLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [LegalSummarization](https://github.com/lauramanor/legal_summarization) | ['eng'] | Retrieval | s2p | [Legal, Written] | None | None | +| [LinceMTBitextMining](https://ritual.uh.edu/lince/) (Aguilar et al., 2020) | ['eng', 'hin'] | BitextMining | s2s | [Social, Written] | None | None | +| [LitSearchRetrieval](https://github.com/princeton-nlp/LitSearch) (Ajith et al., 2024) | ['eng'] | Retrieval | s2p | [Academic, Non-fiction, Written] | None | None | +| [LivedoorNewsClustering.v2](https://github.com/sbintuitions/JMTEB) | ['jpn'] | Clustering | s2s | [News, Written] | None | None | +| [MAUDLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [MIRACLReranking](https://project-miracl.github.io/) (Zhang et al., 2023) | ['ara', 'ben', 'deu', 'eng', 'fas', 'fin', 'fra', 'hin', 'ind', 'jpn', 'kor', 'rus', 'spa', 'swa', 'tel', 'tha', 'yor', 'zho'] | Reranking | s2s | [Encyclopaedic, Written] | None | None | +| [MIRACLRetrieval](http://miracl.ai/) (Zhang et al., 2023) | ['ara', 'ben', 'deu', 'eng', 'fas', 'fin', 'fra', 'hin', 'ind', 'jpn', 'kor', 'rus', 'spa', 'swa', 'tel', 'tha', 'yor', 'zho'] | Retrieval | s2p | [Encyclopaedic, Written] | None | None | +| [MIRACLRetrievalHardNegatives](http://miracl.ai/) (Zhang et al., 2023) | ['ara', 'ben', 'deu', 'eng', 'fas', 'fin', 'fra', 'hin', 'ind', 'jpn', 'kor', 'rus', 'spa', 'swa', 'tel', 'tha', 'yor', 'zho'] | Retrieval | s2p | [Encyclopaedic, Written] | None | None | +| [MLQARetrieval](https://huggingface.co/datasets/mlqa) | ['ara', 'deu', 'eng', 'hin', 'spa', 'vie', 'zho'] | Retrieval | s2p | [Encyclopaedic, Written] | None | None | +| [MLQuestions](https://github.com/McGill-NLP/MLQuestions) | ['eng'] | Retrieval | s2p | [Encyclopaedic, Academic, Written] | None | None | +| [MLSUMClusteringP2P.v2](https://huggingface.co/datasets/mteb/mlsum) (Scialom et al., 2020) | ['deu', 'fra', 'rus', 'spa'] | Clustering | p2p | [News, Written] | None | None | +| [MLSUMClusteringS2S.v2](https://huggingface.co/datasets/mteb/mlsum) (Scialom et al., 2020) | ['deu', 'fra', 'rus', 'spa'] | Clustering | s2s | [News, Written] | None | None | | [MMarcoReranking](https://github.com/unicamp-dl/mMARCO) (Luiz Henrique Bonifacio, 2021) | ['cmn'] | Reranking | s2s | | None | None | -| [MMarcoRetrieval](https://arxiv.org/abs/2309.07597) (Shitao Xiao, 2024) | ['cmn'] | Retrieval | s2p | | None | {'dev': {'average_document_length': 114.41787048392986, 'average_query_length': 10.51131805157593, 'num_documents': 106813, 'num_queries': 6980, 'average_relevant_docs_per_query': 1.0654727793696275}} | -| [MSMARCO](https://microsoft.github.io/msmarco/) (Tri Nguyen and Mir Rosenberg and Xia Song and Jianfeng Gao and Saurabh Tiwary and Rangan Majumder and Li Deng, 2016) | ['eng'] | Retrieval | s2p | | None | {'train': {'average_document_length': 335.79716603691344, 'average_query_length': 33.21898281898998, 'num_documents': 8841823, 'num_queries': 502939, 'average_relevant_docs_per_query': 1.0592755781516248}, 'dev': {'average_document_length': 335.79716603691344, 'average_query_length': 33.2621776504298, 'num_documents': 8841823, 'num_queries': 6980, 'average_relevant_docs_per_query': 1.0654727793696275}, 'test': {'average_document_length': 335.79716603691344, 'average_query_length': 32.74418604651163, 'num_documents': 8841823, 'num_queries': 43, 'average_relevant_docs_per_query': 95.3953488372093}} | -| [MSMARCO-PL](https://microsoft.github.io/msmarco/) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | [Web, Written] | None | {'test': {'average_document_length': 349.3574939240471, 'average_query_length': 33.02325581395349, 'num_documents': 8841823, 'num_queries': 43, 'average_relevant_docs_per_query': 95.3953488372093}} | -| [MSMARCO-PLHardNegatives](https://microsoft.github.io/msmarco/) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | [Web, Written] | {'test': 43} | {'test': {'average_document_length': 382.3476426537285, 'average_query_length': 33.02325581395349, 'num_documents': 9481, 'num_queries': 43, 'average_relevant_docs_per_query': 95.3953488372093}} | -| [MSMARCOHardNegatives](https://microsoft.github.io/msmarco/) (Tri Nguyen and Mir Rosenberg and Xia Song and Jianfeng Gao and Saurabh Tiwary and Rangan Majumder and Li Deng, 2016) | ['eng'] | Retrieval | s2p | | {'test': 43} | {'test': {'average_document_length': 355.2909668633681, 'average_query_length': 32.74418604651163, 'num_documents': 8812, 'num_queries': 43, 'average_relevant_docs_per_query': 95.3953488372093}} | +| [MMarcoRetrieval](https://arxiv.org/abs/2309.07597) (Shitao Xiao, 2024) | ['cmn'] | Retrieval | s2p | | None | None | +| [MSMARCO](https://microsoft.github.io/msmarco/) (Tri Nguyen and Mir Rosenberg and Xia Song and Jianfeng Gao and Saurabh Tiwary and Rangan Majumder and Li Deng, 2016) | ['eng'] | Retrieval | s2p | | None | None | +| [MSMARCO-PL](https://microsoft.github.io/msmarco/) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | [Web, Written] | None | None | +| [MSMARCO-PLHardNegatives](https://microsoft.github.io/msmarco/) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | [Web, Written] | None | None | +| [MSMARCOHardNegatives](https://microsoft.github.io/msmarco/) (Tri Nguyen and Mir Rosenberg and Xia Song and Jianfeng Gao and Saurabh Tiwary and Rangan Majumder and Li Deng, 2016) | ['eng'] | Retrieval | s2p | | None | None | | [MSMARCOv2](https://microsoft.github.io/msmarco/TREC-Deep-Learning.html) (Tri Nguyen and Mir Rosenberg and Xia Song and Jianfeng Gao and Saurabh Tiwary and Rangan Majumder and Li Deng, 2016) | ['eng'] | Retrieval | s2p | | None | None | -| [MTOPDomainClassification](https://arxiv.org/pdf/2008.09335.pdf) | ['deu', 'eng', 'fra', 'hin', 'spa', 'tha'] | Classification | s2s | [Spoken, Spoken] | {'validation': 2235, 'test': 4386} | {'validation': {'num_samples': 10837, 'average_text_length': 39.85374181046415, 'unique_labels': 11, 'labels': {'1': {'count': 1688}, '10': {'count': 754}, '7': {'count': 849}, '3': {'count': 681}, '6': {'count': 985}, '2': {'count': 647}, '9': {'count': 872}, '0': {'count': 833}, '5': {'count': 1182}, '4': {'count': 982}, '8': {'count': 1364}}, 'hf_subset_descriptive_stats': {}, 'en': {'num_samples': 2235, 'average_text_length': 36.53825503355705, 'unique_labels': 11, 'labels': {'1': {'count': 329}, '10': {'count': 185}, '7': {'count': 183}, '3': {'count': 134}, '6': {'count': 186}, '2': {'count': 123}, '9': {'count': 196}, '0': {'count': 176}, '5': {'count': 228}, '4': {'count': 207}, '8': {'count': 288}}}, 'de': {'num_samples': 1815, 'average_text_length': 42.824793388429754, 'unique_labels': 11, 'labels': {'0': {'count': 99}, '1': {'count': 303}, '2': {'count': 104}, '3': {'count': 122}, '6': {'count': 165}, '4': {'count': 157}, '7': {'count': 141}, '5': {'count': 203}, '8': {'count': 220}, '10': {'count': 133}, '9': {'count': 168}}}, 'es': {'num_samples': 1527, 'average_text_length': 44.34839554682384, 'unique_labels': 11, 'labels': {'1': {'count': 197}, '6': {'count': 166}, '4': {'count': 138}, '10': {'count': 103}, '3': {'count': 104}, '5': {'count': 190}, '2': {'count': 115}, '8': {'count': 212}, '7': {'count': 82}, '9': {'count': 76}, '0': {'count': 144}}}, 'fr': {'num_samples': 1577, 'average_text_length': 43.12492073557387, 'unique_labels': 11, 'labels': {'0': {'count': 125}, '1': {'count': 278}, '2': {'count': 92}, '3': {'count': 89}, '4': {'count': 137}, '7': {'count': 145}, '6': {'count': 138}, '5': {'count': 168}, '8': {'count': 203}, '9': {'count': 124}, '10': {'count': 78}}}, 'hi': {'num_samples': 2012, 'average_text_length': 39.139662027833005, 'unique_labels': 11, 'labels': {'0': {'count': 161}, '1': {'count': 304}, '3': {'count': 126}, '4': {'count': 193}, '2': {'count': 109}, '10': {'count': 154}, '5': {'count': 208}, '6': {'count': 167}, '7': {'count': 172}, '8': {'count': 235}, '9': {'count': 183}}}, 'th': {'num_samples': 1671, 'average_text_length': 34.726511071214844, 'unique_labels': 11, 'labels': {'0': {'count': 128}, '1': {'count': 277}, '2': {'count': 104}, '3': {'count': 106}, '4': {'count': 150}, '5': {'count': 185}, '6': {'count': 163}, '7': {'count': 126}, '8': {'count': 206}, '9': {'count': 125}, '10': {'count': 101}}}}, 'test': {'num_samples': 19680, 'average_text_length': 39.71443089430894, 'unique_labels': 11, 'labels': {'2': {'count': 977}, '5': {'count': 2372}, '6': {'count': 2014}, '8': {'count': 2572}, '9': {'count': 1317}, '1': {'count': 3065}, '10': {'count': 1330}, '3': {'count': 1351}, '0': {'count': 1459}, '7': {'count': 1535}, '4': {'count': 1688}}, 'hf_subset_descriptive_stats': {}, 'en': {'num_samples': 4386, 'average_text_length': 36.79343365253078, 'unique_labels': 11, 'labels': {'2': {'count': 197}, '5': {'count': 487}, '6': {'count': 418}, '8': {'count': 613}, '9': {'count': 346}, '1': {'count': 613}, '10': {'count': 358}, '3': {'count': 290}, '0': {'count': 341}, '7': {'count': 354}, '4': {'count': 369}}}, 'de': {'num_samples': 3549, 'average_text_length': 42.67258382642998, 'unique_labels': 11, 'labels': {'0': {'count': 193}, '10': {'count': 264}, '1': {'count': 553}, '2': {'count': 163}, '3': {'count': 256}, '5': {'count': 439}, '4': {'count': 306}, '6': {'count': 353}, '7': {'count': 279}, '8': {'count': 452}, '9': {'count': 291}}}, 'es': {'num_samples': 2998, 'average_text_length': 43.552034689793196, 'unique_labels': 11, 'labels': {'1': {'count': 401}, '6': {'count': 352}, '4': {'count': 246}, '10': {'count': 206}, '3': {'count': 231}, '5': {'count': 404}, '2': {'count': 177}, '8': {'count': 435}, '7': {'count': 156}, '9': {'count': 126}, '0': {'count': 264}}}, 'fr': {'num_samples': 3193, 'average_text_length': 43.854995302223614, 'unique_labels': 11, 'labels': {'0': {'count': 253}, '1': {'count': 551}, '2': {'count': 159}, '3': {'count': 190}, '4': {'count': 280}, '6': {'count': 330}, '5': {'count': 356}, '7': {'count': 272}, '8': {'count': 462}, '10': {'count': 159}, '9': {'count': 181}}}, 'hi': {'num_samples': 2789, 'average_text_length': 37.395123700250984, 'unique_labels': 11, 'labels': {'0': {'count': 208}, '1': {'count': 470}, '5': {'count': 335}, '3': {'count': 195}, '4': {'count': 242}, '2': {'count': 132}, '6': {'count': 267}, '7': {'count': 262}, '8': {'count': 265}, '10': {'count': 186}, '9': {'count': 227}}}, 'th': {'num_samples': 2765, 'average_text_length': 33.94792043399638, 'unique_labels': 11, 'labels': {'0': {'count': 200}, '1': {'count': 477}, '2': {'count': 149}, '3': {'count': 189}, '4': {'count': 245}, '6': {'count': 294}, '5': {'count': 351}, '7': {'count': 212}, '8': {'count': 345}, '9': {'count': 146}, '10': {'count': 157}}}}, 'train': {'num_samples': 73928, 'average_text_length': 39.73095444215994, 'unique_labels': 11, 'labels': {'0': {'count': 5262}, '5': {'count': 8334}, '6': {'count': 6961}, '9': {'count': 5313}, '1': {'count': 11107}, '8': {'count': 9698}, '10': {'count': 5084}, '2': {'count': 4770}, '4': {'count': 6644}, '3': {'count': 5191}, '7': {'count': 5564}}, 'hf_subset_descriptive_stats': {}, 'en': {'num_samples': 15667, 'average_text_length': 36.57222186761984, 'unique_labels': 11, 'labels': {'0': {'count': 1165}, '5': {'count': 1657}, '6': {'count': 1402}, '9': {'count': 1303}, '1': {'count': 2187}, '8': {'count': 2157}, '10': {'count': 1219}, '2': {'count': 929}, '4': {'count': 1353}, '3': {'count': 1064}, '7': {'count': 1231}}}, 'de': {'num_samples': 13424, 'average_text_length': 43.226013110846246, 'unique_labels': 11, 'labels': {'0': {'count': 761}, '10': {'count': 996}, '4': {'count': 1185}, '1': {'count': 2016}, '7': {'count': 1029}, '5': {'count': 1484}, '2': {'count': 814}, '3': {'count': 980}, '6': {'count': 1265}, '8': {'count': 1767}, '9': {'count': 1127}}}, 'es': {'num_samples': 10934, 'average_text_length': 43.60691421254801, 'unique_labels': 11, 'labels': {'1': {'count': 1459}, '6': {'count': 1188}, '4': {'count': 928}, '10': {'count': 743}, '3': {'count': 830}, '5': {'count': 1396}, '2': {'count': 823}, '8': {'count': 1555}, '7': {'count': 525}, '9': {'count': 560}, '0': {'count': 927}}}, 'fr': {'num_samples': 11814, 'average_text_length': 43.594802776367025, 'unique_labels': 11, 'labels': {'0': {'count': 861}, '10': {'count': 668}, '1': {'count': 1968}, '7': {'count': 975}, '5': {'count': 1261}, '2': {'count': 799}, '3': {'count': 734}, '4': {'count': 1082}, '6': {'count': 1113}, '8': {'count': 1656}, '9': {'count': 697}}}, 'hi': {'num_samples': 11330, 'average_text_length': 37.592144748455425, 'unique_labels': 11, 'labels': {'0': {'count': 794}, '1': {'count': 1741}, '7': {'count': 974}, '2': {'count': 670}, '3': {'count': 831}, '5': {'count': 1272}, '6': {'count': 940}, '4': {'count': 1073}, '10': {'count': 786}, '8': {'count': 1281}, '9': {'count': 968}}}, 'th': {'num_samples': 10759, 'average_text_length': 34.04043126684636, 'unique_labels': 11, 'labels': {'0': {'count': 754}, '10': {'count': 672}, '1': {'count': 1736}, '7': {'count': 830}, '2': {'count': 735}, '3': {'count': 752}, '5': {'count': 1264}, '6': {'count': 1053}, '4': {'count': 1023}, '8': {'count': 1282}, '9': {'count': 658}}}}} | -| [MTOPIntentClassification](https://arxiv.org/pdf/2008.09335.pdf) | ['deu', 'eng', 'fra', 'hin', 'spa', 'tha'] | Classification | s2s | [Spoken, Spoken] | {'validation': 2235, 'test': 4386} | {'validation': 36.5, 'test': 36.8} | -| [MacedonianTweetSentimentClassification](https://aclanthology.org/R15-1034/) | ['mkd'] | Classification | s2s | [Social, Written] | {'test': 1139} | {'test': 67.6} | -| [MalayalamNewsClassification](https://github.com/goru001/nlp-for-malyalam) (Anoop Kunchukuttan, 2020) | ['mal'] | Classification | s2s | [News, Written] | {'train': 5036, 'test': 1260} | {'train': 79.48, 'test': 80.44} | -| [MalteseNewsClassification](https://huggingface.co/datasets/MLRS/maltese_news_categories) | ['mlt'] | MultilabelClassification | s2s | [Constructed, Written] | {'train': 10784, 'test': 2297} | {'train': 1595.63, 'test': 1752.1} | -| [MarathiNewsClassification](https://github.com/goru001/nlp-for-marathi) (Anoop Kunchukuttan, 2020) | ['mar'] | Classification | s2s | [News, Written] | {'test': 2048} | {'test': 52.37} | -| [MasakhaNEWSClassification](https://arxiv.org/abs/2304.09972) (David Ifeoluwa Adelani, 2023) | ['amh', 'eng', 'fra', 'hau', 'ibo', 'lin', 'lug', 'orm', 'pcm', 'run', 'sna', 'som', 'swa', 'tir', 'xho', 'yor'] | Classification | s2s | [News, Written] | {'test': 422} | {'test': 5116.6} | +| [MTOPDomainClassification](https://arxiv.org/pdf/2008.09335.pdf) | ['deu', 'eng', 'fra', 'hin', 'spa', 'tha'] | Classification | s2s | [Spoken, Spoken] | None | None | +| [MTOPIntentClassification](https://arxiv.org/pdf/2008.09335.pdf) | ['deu', 'eng', 'fra', 'hin', 'spa', 'tha'] | Classification | s2s | [Spoken, Spoken] | None | None | +| [MacedonianTweetSentimentClassification](https://aclanthology.org/R15-1034/) | ['mkd'] | Classification | s2s | [Social, Written] | None | None | +| [MalayalamNewsClassification](https://github.com/goru001/nlp-for-malyalam) (Anoop Kunchukuttan, 2020) | ['mal'] | Classification | s2s | [News, Written] | None | None | +| [MalteseNewsClassification](https://huggingface.co/datasets/MLRS/maltese_news_categories) | ['mlt'] | MultilabelClassification | s2s | [Constructed, Written] | None | None | +| [MarathiNewsClassification](https://github.com/goru001/nlp-for-marathi) (Anoop Kunchukuttan, 2020) | ['mar'] | Classification | s2s | [News, Written] | None | None | +| [MasakhaNEWSClassification](https://arxiv.org/abs/2304.09972) (David Ifeoluwa Adelani, 2023) | ['amh', 'eng', 'fra', 'hau', 'ibo', 'lin', 'lug', 'orm', 'pcm', 'run', 'sna', 'som', 'swa', 'tir', 'xho', 'yor'] | Classification | s2s | [News, Written] | None | None | | [MasakhaNEWSClusteringP2P](https://huggingface.co/datasets/masakhane/masakhanews) (David Ifeoluwa Adelani, 2023) | ['amh', 'eng', 'fra', 'hau', 'ibo', 'lin', 'lug', 'orm', 'pcm', 'run', 'sna', 'som', 'swa', 'tir', 'xho', 'yor'] | Clustering | p2p | [News, Written, Non-fiction] | None | None | | [MasakhaNEWSClusteringS2S](https://huggingface.co/datasets/masakhane/masakhanews) (David Ifeoluwa Adelani, 2023) | ['amh', 'eng', 'fra', 'hau', 'ibo', 'lin', 'lug', 'orm', 'pcm', 'run', 'sna', 'som', 'swa', 'tir', 'xho', 'yor'] | Clustering | s2s | | None | None | -| [MassiveIntentClassification](https://arxiv.org/abs/2204.08582) (Jack FitzGerald, 2022) | ['afr', 'amh', 'ara', 'aze', 'ben', 'cmo', 'cym', 'dan', 'deu', 'ell', 'eng', 'fas', 'fin', 'fra', 'heb', 'hin', 'hun', 'hye', 'ind', 'isl', 'ita', 'jav', 'jpn', 'kan', 'kat', 'khm', 'kor', 'lav', 'mal', 'mon', 'msa', 'mya', 'nld', 'nob', 'pol', 'por', 'ron', 'rus', 'slv', 'spa', 'sqi', 'swa', 'swe', 'tam', 'tel', 'tgl', 'tha', 'tur', 'urd', 'vie'] | Classification | s2s | [Spoken] | {'validation': 2033, 'test': 2974} | {'validation': 34.8, 'test': 34.6} | -| [MassiveScenarioClassification](https://arxiv.org/abs/2204.08582) (Jack FitzGerald, 2022) | ['afr', 'amh', 'ara', 'aze', 'ben', 'cmo', 'cym', 'dan', 'deu', 'ell', 'eng', 'fas', 'fin', 'fra', 'heb', 'hin', 'hun', 'hye', 'ind', 'isl', 'ita', 'jav', 'jpn', 'kan', 'kat', 'khm', 'kor', 'lav', 'mal', 'mon', 'msa', 'mya', 'nld', 'nob', 'pol', 'por', 'ron', 'rus', 'slv', 'spa', 'sqi', 'swa', 'swe', 'tam', 'tel', 'tgl', 'tha', 'tur', 'urd', 'vie'] | Classification | s2s | [Spoken] | {'validation': 2033, 'test': 2974} | {'validation': 34.8, 'test': 34.6} | -| [MedicalQARetrieval](https://bmcbioinformatics.biomedcentral.com/articles/10.1186/s12859-019-3119-4) (Asma et al., 2019) | ['eng'] | Retrieval | s2s | [Medical, Written] | {'test': 2048} | {'test': {'average_document_length': 1153.482421875, 'average_query_length': 52.4794921875, 'num_documents': 2048, 'num_queries': 2048, 'average_relevant_docs_per_query': 1.0}} | -| [MedicalRetrieval](https://arxiv.org/abs/2203.03367) | ['cmn'] | Retrieval | s2p | | None | {'dev': {'average_document_length': 122.04231725066585, 'average_query_length': 17.938, 'num_documents': 100999, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}} | -| [MedrxivClusteringP2P.v2](https://api.medrxiv.org/) | ['eng'] | Clustering | p2p | [Academic, Medical, Written] | {'test': 1500} | {'test': 1984.7} | -| [MedrxivClusteringS2S.v2](https://api.medrxiv.org/) | ['eng'] | Clustering | s2s | [Academic, Medical, Written] | {'test': 1500} | {'test': 114.9} | -| [MewsC16JaClustering](https://github.com/sbintuitions/JMTEB) | ['jpn'] | Clustering | s2s | [News, Written] | {'test': 992} | {'test': 95} | -| [MindSmallReranking](https://msnews.github.io/assets/doc/ACL2020_MIND.pdf) | ['eng'] | Reranking | s2s | [News, Written] | {'test': 107968} | {'test': 70.9} | -| MintakaRetrieval | ['ara', 'deu', 'fra', 'hin', 'ita', 'jpn', 'por', 'spa'] | Retrieval | s2p | [Encyclopaedic, Written] | None | {'test': {'ar': {'average_document_length': 12.736418511066399, 'average_query_length': 55.275533363595095, 'num_documents': 1491, 'num_queries': 2203, 'average_relevant_docs_per_query': 1.0}, 'de': {'average_document_length': 14.40060422960725, 'average_query_length': 65.41322662173546, 'num_documents': 1655, 'num_queries': 2374, 'average_relevant_docs_per_query': 1.0}, 'es': {'average_document_length': 14.291789722386296, 'average_query_length': 64.88325082508251, 'num_documents': 1693, 'num_queries': 2424, 'average_relevant_docs_per_query': 1.0}, 'fr': {'average_document_length': 14.407234539089849, 'average_query_length': 68.88452088452088, 'num_documents': 1714, 'num_queries': 2442, 'average_relevant_docs_per_query': 1.0}, 'hi': {'average_document_length': 12.71038961038961, 'average_query_length': 58.404637247569184, 'num_documents': 770, 'num_queries': 1337, 'average_relevant_docs_per_query': 1.0}, 'it': {'average_document_length': 14.365985576923077, 'average_query_length': 64.39707724425887, 'num_documents': 1664, 'num_queries': 2395, 'average_relevant_docs_per_query': 1.0004175365344468}, 'ja': {'average_document_length': 9.167713567839195, 'average_query_length': 29.961937716262977, 'num_documents': 1592, 'num_queries': 2312, 'average_relevant_docs_per_query': 1.0}, 'pt': {'average_document_length': 14.244471744471744, 'average_query_length': 60.42225998300765, 'num_documents': 1628, 'num_queries': 2354, 'average_relevant_docs_per_query': 1.0004248088360237}}} | -| [Moroco](https://huggingface.co/datasets/moroco) (Andrei M. Butnaru, 2019) | ['ron'] | Classification | s2s | [News, Written] | {'test': 2048} | {'test': 1710.94} | -| [MovieReviewSentimentClassification](https://github.com/TheophileBlard/french-sentiment-analysis-with-bert) (Théophile Blard, 2020) | ['fra'] | Classification | s2s | [Reviews, Written] | {'validation': 1024, 'test': 1024} | {'validation': 550.3, 'test': 558.1} | -| [MrTidyRetrieval](https://huggingface.co/datasets/castorini/mr-tydi) (Xinyu Zhang, 2021) | ['ara', 'ben', 'eng', 'fin', 'ind', 'jpn', 'kor', 'rus', 'swa', 'tel', 'tha'] | Retrieval | s2p | [Encyclopaedic, Written] | | | -| [MultiEURLEXMultilabelClassification](https://huggingface.co/datasets/coastalcph/multi_eurlex) (Chalkidis et al., 2021) | ['bul', 'ces', 'dan', 'deu', 'ell', 'eng', 'est', 'fin', 'fra', 'hrv', 'hun', 'ita', 'lav', 'lit', 'mlt', 'nld', 'pol', 'por', 'ron', 'slk', 'slv', 'spa', 'swe'] | MultilabelClassification | p2p | [Legal, Government, Written] | {'test': 5000} | {'test': {'average_text_length': 12014.408930434782, 'average_label_per_text': 3.5938, 'num_samples': 115000, 'unique_labels': 21, 'labels': {'18': {'count': 50784}, '15': {'count': 30981}, '5': {'count': 24978}, '6': {'count': 45080}, '3': {'count': 63687}, '17': {'count': 37743}, '1': {'count': 15019}, '20': {'count': 14030}, '0': {'count': 17802}, '2': {'count': 22402}, '19': {'count': 10212}, '9': {'count': 3772}, '4': {'count': 9062}, '10': {'count': 7705}, '11': {'count': 12213}, '7': {'count': 14306}, '12': {'count': 11799}, '8': {'count': 13800}, '13': {'count': 2346}, '14': {'count': 4255}, '16': {'count': 1311}}, 'hf_subset_descriptive_stats': {'en': {'average_text_length': 11720.2926, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'de': {'average_text_length': 12865.4162, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'fr': {'average_text_length': 13081.1098, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'it': {'average_text_length': 12763.4786, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'es': {'average_text_length': 13080.29, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'pl': {'average_text_length': 12282.5926, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'ro': {'average_text_length': 12836.9322, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'nl': {'average_text_length': 12857.9742, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'el': {'average_text_length': 12998.143, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'hu': {'average_text_length': 12424.641, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'pt': {'average_text_length': 12482.4616, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'cs': {'average_text_length': 10783.4676, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'sv': {'average_text_length': 11612.4774, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'bg': {'average_text_length': 12235.4268, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'da': {'average_text_length': 11773.958, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'fi': {'average_text_length': 12087.6862, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'sk': {'average_text_length': 11130.814, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'lt': {'average_text_length': 11245.3566, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'hr': {'average_text_length': 11022.142, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'sl': {'average_text_length': 10620.0594, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'et': {'average_text_length': 10898.4312, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'lv': {'average_text_length': 10938.5102, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'mt': {'average_text_length': 12589.7442, 'average_label_per_text': 3.5938, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}}}} | -| [MultiHateClassification](https://aclanthology.org/2022.woah-1.15/) | ['ara', 'cmn', 'deu', 'eng', 'fra', 'hin', 'ita', 'nld', 'pol', 'por', 'spa'] | Classification | s2s | [Constructed, Written] | {'test': 10000} | {'test': 45.9} | -| [MultiLongDocRetrieval](https://arxiv.org/abs/2402.03216) (Jianlv Chen, 2024) | ['ara', 'cmn', 'deu', 'eng', 'fra', 'hin', 'ita', 'jpn', 'kor', 'por', 'rus', 'spa', 'tha'] | Retrieval | s2p | [Encyclopaedic, Written, Web, Non-fiction, Fiction] | None | {'dev': {'ar': {'average_document_length': 29234.48153016958, 'average_query_length': 69.27, 'num_documents': 7607, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'de': {'average_document_length': 33771.2111, 'average_query_length': 153.63, 'num_documents': 10000, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'en': {'average_document_length': 13332.76764, 'average_query_length': 81.22, 'num_documents': 200000, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'es': {'average_document_length': 36567.1736990891, 'average_query_length': 123.11, 'num_documents': 9551, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'fr': {'average_document_length': 36009.4934, 'average_query_length': 142.165, 'num_documents': 10000, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'hi': {'average_document_length': 18688.50788229112, 'average_query_length': 77.995, 'num_documents': 3806, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'it': {'average_document_length': 36633.9969, 'average_query_length': 99.615, 'num_documents': 10000, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'ja': {'average_document_length': 14480.7508, 'average_query_length': 61.625, 'num_documents': 10000, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'ko': {'average_document_length': 13813.441224093263, 'average_query_length': 58.845, 'num_documents': 6176, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'pt': {'average_document_length': 32127.576952351956, 'average_query_length': 122.275, 'num_documents': 6569, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'ru': {'average_document_length': 35934.8756, 'average_query_length': 87.875, 'num_documents': 10000, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'th': {'average_document_length': 25993.2696, 'average_query_length': 107.81, 'num_documents': 10000, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'zh': {'average_document_length': 6039.059725, 'average_query_length': 26.79, 'num_documents': 200000, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}}, 'test': {'ar': {'average_document_length': 29234.48153016958, 'average_query_length': 75.77, 'num_documents': 7607, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'de': {'average_document_length': 33771.2111, 'average_query_length': 123.65, 'num_documents': 10000, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'en': {'average_document_length': 13332.76764, 'average_query_length': 81.33, 'num_documents': 200000, 'num_queries': 800, 'average_relevant_docs_per_query': 1.0}, 'es': {'average_document_length': 36567.1736990891, 'average_query_length': 131.985, 'num_documents': 9551, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'fr': {'average_document_length': 36009.4934, 'average_query_length': 149.795, 'num_documents': 10000, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'hi': {'average_document_length': 18688.50788229112, 'average_query_length': 103.76, 'num_documents': 3806, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'it': {'average_document_length': 36633.9969, 'average_query_length': 114.595, 'num_documents': 10000, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'ja': {'average_document_length': 14480.7508, 'average_query_length': 55.73, 'num_documents': 10000, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'ko': {'average_document_length': 13813.441224093263, 'average_query_length': 58.72, 'num_documents': 6176, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'pt': {'average_document_length': 32127.576952351956, 'average_query_length': 113.455, 'num_documents': 6569, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'ru': {'average_document_length': 35934.8756, 'average_query_length': 94.87, 'num_documents': 10000, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'th': {'average_document_length': 25993.2696, 'average_query_length': 97.99, 'num_documents': 10000, 'num_queries': 200, 'average_relevant_docs_per_query': 1.0}, 'zh': {'average_document_length': 6039.059725, 'average_query_length': 24.70875, 'num_documents': 200000, 'num_queries': 800, 'average_relevant_docs_per_query': 1.0}}} | +| [MassiveIntentClassification](https://arxiv.org/abs/2204.08582) (Jack FitzGerald, 2022) | ['afr', 'amh', 'ara', 'aze', 'ben', 'cmo', 'cym', 'dan', 'deu', 'ell', 'eng', 'fas', 'fin', 'fra', 'heb', 'hin', 'hun', 'hye', 'ind', 'isl', 'ita', 'jav', 'jpn', 'kan', 'kat', 'khm', 'kor', 'lav', 'mal', 'mon', 'msa', 'mya', 'nld', 'nob', 'pol', 'por', 'ron', 'rus', 'slv', 'spa', 'sqi', 'swa', 'swe', 'tam', 'tel', 'tgl', 'tha', 'tur', 'urd', 'vie'] | Classification | s2s | [Spoken] | None | None | +| [MassiveScenarioClassification](https://arxiv.org/abs/2204.08582) (Jack FitzGerald, 2022) | ['afr', 'amh', 'ara', 'aze', 'ben', 'cmo', 'cym', 'dan', 'deu', 'ell', 'eng', 'fas', 'fin', 'fra', 'heb', 'hin', 'hun', 'hye', 'ind', 'isl', 'ita', 'jav', 'jpn', 'kan', 'kat', 'khm', 'kor', 'lav', 'mal', 'mon', 'msa', 'mya', 'nld', 'nob', 'pol', 'por', 'ron', 'rus', 'slv', 'spa', 'sqi', 'swa', 'swe', 'tam', 'tel', 'tgl', 'tha', 'tur', 'urd', 'vie'] | Classification | s2s | [Spoken] | None | None | +| [MedicalQARetrieval](https://bmcbioinformatics.biomedcentral.com/articles/10.1186/s12859-019-3119-4) (Asma et al., 2019) | ['eng'] | Retrieval | s2s | [Medical, Written] | None | None | +| [MedicalRetrieval](https://arxiv.org/abs/2203.03367) | ['cmn'] | Retrieval | s2p | | None | None | +| [MedrxivClusteringP2P.v2](https://api.medrxiv.org/) | ['eng'] | Clustering | p2p | [Academic, Medical, Written] | None | None | +| [MedrxivClusteringS2S.v2](https://api.medrxiv.org/) | ['eng'] | Clustering | s2s | [Academic, Medical, Written] | None | None | +| [MewsC16JaClustering](https://github.com/sbintuitions/JMTEB) | ['jpn'] | Clustering | s2s | [News, Written] | None | None | +| [MindSmallReranking](https://msnews.github.io/assets/doc/ACL2020_MIND.pdf) | ['eng'] | Reranking | s2s | [News, Written] | None | None | +| MintakaRetrieval | ['ara', 'deu', 'fra', 'hin', 'ita', 'jpn', 'por', 'spa'] | Retrieval | s2p | [Encyclopaedic, Written] | None | None | +| [Moroco](https://huggingface.co/datasets/moroco) (Andrei M. Butnaru, 2019) | ['ron'] | Classification | s2s | [News, Written] | None | None | +| [MovieReviewSentimentClassification](https://github.com/TheophileBlard/french-sentiment-analysis-with-bert) (Théophile Blard, 2020) | ['fra'] | Classification | s2s | [Reviews, Written] | None | None | +| [MrTidyRetrieval](https://huggingface.co/datasets/castorini/mr-tydi) (Xinyu Zhang, 2021) | ['ara', 'ben', 'eng', 'fin', 'ind', 'jpn', 'kor', 'rus', 'swa', 'tel', 'tha'] | Retrieval | s2p | [Encyclopaedic, Written] | None | None | +| [MultiEURLEXMultilabelClassification](https://huggingface.co/datasets/coastalcph/multi_eurlex) (Chalkidis et al., 2021) | ['bul', 'ces', 'dan', 'deu', 'ell', 'eng', 'est', 'fin', 'fra', 'hrv', 'hun', 'ita', 'lav', 'lit', 'mlt', 'nld', 'pol', 'por', 'ron', 'slk', 'slv', 'spa', 'swe'] | MultilabelClassification | p2p | [Legal, Government, Written] | {'test': 115000} | {'test': {'average_text_length': 12014.41, 'number_of_characters': 1381657027, 'average_label_per_text': 3.59, 'num_samples': 115000, 'unique_labels': 21, 'labels': {'18': {'count': 50784}, '15': {'count': 30981}, '5': {'count': 24978}, '6': {'count': 45080}, '3': {'count': 63687}, '17': {'count': 37743}, '1': {'count': 15019}, '20': {'count': 14030}, '0': {'count': 17802}, '2': {'count': 22402}, '19': {'count': 10212}, '9': {'count': 3772}, '4': {'count': 9062}, '10': {'count': 7705}, '11': {'count': 12213}, '7': {'count': 14306}, '12': {'count': 11799}, '8': {'count': 13800}, '13': {'count': 2346}, '14': {'count': 4255}, '16': {'count': 1311}}, 'hf_subset_descriptive_stats': {'en': {'average_text_length': 11720.29, 'number_of_characters': 58601463, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'de': {'average_text_length': 12865.42, 'number_of_characters': 64327081, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'fr': {'average_text_length': 13081.11, 'number_of_characters': 65405549, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'it': {'average_text_length': 12763.48, 'number_of_characters': 63817393, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'es': {'average_text_length': 13080.29, 'number_of_characters': 65401450, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'pl': {'average_text_length': 12282.59, 'number_of_characters': 61412963, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'ro': {'average_text_length': 12836.93, 'number_of_characters': 64184661, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'nl': {'average_text_length': 12857.97, 'number_of_characters': 64289871, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'el': {'average_text_length': 12998.14, 'number_of_characters': 64990715, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'hu': {'average_text_length': 12424.64, 'number_of_characters': 62123205, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'pt': {'average_text_length': 12482.46, 'number_of_characters': 62412308, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'cs': {'average_text_length': 10783.47, 'number_of_characters': 53917338, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'sv': {'average_text_length': 11612.48, 'number_of_characters': 58062387, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'bg': {'average_text_length': 12235.43, 'number_of_characters': 61177134, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'da': {'average_text_length': 11773.96, 'number_of_characters': 58869790, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'fi': {'average_text_length': 12087.69, 'number_of_characters': 60438431, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'sk': {'average_text_length': 11130.81, 'number_of_characters': 55654070, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'lt': {'average_text_length': 11245.36, 'number_of_characters': 56226783, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'hr': {'average_text_length': 11022.14, 'number_of_characters': 55110710, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'sl': {'average_text_length': 10620.06, 'number_of_characters': 53100297, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'et': {'average_text_length': 10898.43, 'number_of_characters': 54492156, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'lv': {'average_text_length': 10938.51, 'number_of_characters': 54692551, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}, 'mt': {'average_text_length': 12589.74, 'number_of_characters': 62948721, 'average_label_per_text': 3.59, 'num_samples': 5000, 'unique_labels': 21, 'labels': {'18': {'count': 2208}, '15': {'count': 1347}, '5': {'count': 1086}, '6': {'count': 1960}, '3': {'count': 2769}, '17': {'count': 1641}, '1': {'count': 653}, '20': {'count': 610}, '0': {'count': 774}, '2': {'count': 974}, '19': {'count': 444}, '9': {'count': 164}, '4': {'count': 394}, '10': {'count': 335}, '11': {'count': 531}, '7': {'count': 622}, '12': {'count': 513}, '8': {'count': 600}, '13': {'count': 102}, '14': {'count': 185}, '16': {'count': 57}}}}}} | +| [MultiHateClassification](https://aclanthology.org/2022.woah-1.15/) | ['ara', 'cmn', 'deu', 'eng', 'fra', 'hin', 'ita', 'nld', 'pol', 'por', 'spa'] | Classification | s2s | [Constructed, Written] | None | None | +| [MultiLongDocRetrieval](https://arxiv.org/abs/2402.03216) (Jianlv Chen, 2024) | ['ara', 'cmn', 'deu', 'eng', 'fra', 'hin', 'ita', 'jpn', 'kor', 'por', 'rus', 'spa', 'tha'] | Retrieval | s2p | [Encyclopaedic, Written, Web, Non-fiction, Fiction] | None | None | | [MultilingualSentiment](https://github.com/tyqiangz/multilingual-sentiment-datasets) | ['cmn'] | Classification | s2s | | None | None | -| [MultilingualSentimentClassification](https://huggingface.co/datasets/mteb/multilingual-sentiment-classification) | ['ara', 'bam', 'bul', 'cmn', 'cym', 'deu', 'dza', 'ell', 'eng', 'eus', 'fas', 'fin', 'heb', 'hrv', 'ind', 'jpn', 'kor', 'mlt', 'nor', 'pol', 'rus', 'slk', 'spa', 'tha', 'tur', 'uig', 'urd', 'vie', 'zho'] | Classification | s2s | [Reviews, Written] | {'test': 7000} | {'test': 56} | -| [MyanmarNews](https://huggingface.co/datasets/myanmar_news) (A. H. Khine, 2017) | ['mya'] | Classification | p2p | [News, Written] | {'train': 2048} | {'train': 174.2} | -| [NFCorpus](https://www.cl.uni-heidelberg.de/statnlpgroup/nfcorpus/) (Boteva et al., 2016) | ['eng'] | Retrieval | s2p | | None | {'test': {'average_document_length': 1589.783925130746, 'average_query_length': 21.764705882352942, 'num_documents': 3633, 'num_queries': 323, 'average_relevant_docs_per_query': 38.18575851393189}} | -| [NFCorpus-PL](https://www.cl.uni-heidelberg.de/statnlpgroup/nfcorpus/) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | | None | {'test': {'average_document_length': 1652.1926782273604, 'average_query_length': 24.390092879256965, 'num_documents': 3633, 'num_queries': 323, 'average_relevant_docs_per_query': 38.18575851393189}} | -| [NLPJournalAbsIntroRetrieval](https://github.com/sbintuitions/JMTEB) | ['jpn'] | Retrieval | s2s | [Academic, Written] | {'test': 404} | {'test': {'average_document_length': 2052.8611111111113, 'average_query_length': 439.2772277227723, 'num_documents': 504, 'num_queries': 404, 'average_relevant_docs_per_query': 1.0}} | -| [NLPJournalTitleAbsRetrieval](https://github.com/sbintuitions/JMTEB) | ['jpn'] | Retrieval | s2s | [Academic, Written] | {'test': 404} | {'test': {'average_document_length': 441.6746031746032, 'average_query_length': 27.60891089108911, 'num_documents': 504, 'num_queries': 404, 'average_relevant_docs_per_query': 1.0}} | -| [NLPJournalTitleIntroRetrieval](https://github.com/sbintuitions/JMTEB) | ['jpn'] | Retrieval | s2s | [Academic, Written] | {'test': 404} | {'test': {'average_document_length': 2052.8611111111113, 'average_query_length': 27.60891089108911, 'num_documents': 504, 'num_queries': 404, 'average_relevant_docs_per_query': 1.0}} | -| [NQ](https://ai.google.com/research/NaturalQuestions/) (Tom Kwiatkowski, 2019) | ['eng'] | Retrieval | s2p | | None | {'test': {'average_document_length': 492.2287851281462, 'average_query_length': 48.17902665121669, 'num_documents': 2681468, 'num_queries': 3452, 'average_relevant_docs_per_query': 1.2169756662804172}} | -| [NQ-PL](https://ai.google.com/research/NaturalQuestions/) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | | None | {'test': {'average_document_length': 502.14302128535564, 'average_query_length': 48.31662804171495, 'num_documents': 2681468, 'num_queries': 3452, 'average_relevant_docs_per_query': 1.2169756662804172}} | -| [NQ-PLHardNegatives](https://ai.google.com/research/NaturalQuestions/) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | | {'test': 1000} | {'test': {'average_document_length': 610.7449138094336, 'average_query_length': 48.381, 'num_documents': 184765, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.213}} | -| [NQHardNegatives](https://ai.google.com/research/NaturalQuestions/) (Tom Kwiatkowski, 2019) | ['eng'] | Retrieval | s2p | | {'test': 1000} | {'test': {'average_document_length': 602.7903551179953, 'average_query_length': 47.878, 'num_documents': 198779, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.213}} | -| [NTREXBitextMining](https://huggingface.co/datasets/davidstap/NTREX) | ['afr', 'amh', 'arb', 'aze', 'bak', 'bel', 'bem', 'ben', 'bod', 'bos', 'bul', 'cat', 'ces', 'ckb', 'cym', 'dan', 'deu', 'div', 'dzo', 'ell', 'eng', 'eus', 'ewe', 'fao', 'fas', 'fij', 'fil', 'fin', 'fra', 'fuc', 'gle', 'glg', 'guj', 'hau', 'heb', 'hin', 'hmn', 'hrv', 'hun', 'hye', 'ibo', 'ind', 'isl', 'ita', 'jpn', 'kan', 'kat', 'kaz', 'khm', 'kin', 'kir', 'kmr', 'kor', 'lao', 'lav', 'lit', 'ltz', 'mal', 'mar', 'mey', 'mkd', 'mlg', 'mlt', 'mon', 'mri', 'msa', 'mya', 'nde', 'nep', 'nld', 'nno', 'nob', 'nso', 'nya', 'orm', 'pan', 'pol', 'por', 'prs', 'pus', 'ron', 'rus', 'shi', 'sin', 'slk', 'slv', 'smo', 'sna', 'snd', 'som', 'spa', 'sqi', 'srp', 'ssw', 'swa', 'swe', 'tah', 'tam', 'tat', 'tel', 'tgk', 'tha', 'tir', 'ton', 'tsn', 'tuk', 'tur', 'uig', 'ukr', 'urd', 'uzb', 'ven', 'vie', 'wol', 'xho', 'yor', 'yue', 'zho', 'zul'] | BitextMining | s2s | [News, Written] | {'test': 3826252} | {'test': 120} | -| [NYSJudicialEthicsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 292} | {'test': 159.45} | -| [NaijaSenti](https://github.com/hausanlp/NaijaSenti) | ['hau', 'ibo', 'pcm', 'yor'] | Classification | s2s | [Social, Written] | {'test': 4800} | {'test': 72.81} | -| [NarrativeQARetrieval](https://metatext.io/datasets/narrativeqa) (Tomáš Kočiský, 2017) | ['eng'] | Retrieval | s2p | | None | {'test': {'average_document_length': 326753.5323943662, 'average_query_length': 47.730889457232166, 'num_documents': 355, 'num_queries': 10557, 'average_relevant_docs_per_query': 1.0}} | -| [NepaliNewsClassification](https://github.com/goru001/nlp-for-nepali) | ['nep'] | Classification | s2s | [News, Written] | {'train': 5975, 'test': 1495} | {'train': 196.61, 'test': 196.017} | -| [NeuCLIR2022Retrieval](https://neuclir.github.io/) (Lawrie et al., 2023) | ['fas', 'rus', 'zho'] | Retrieval | s2p | [News, Written] | {'fas': 2232130, 'zho': 3179323, 'rus': 4627657} | {'test': {'fas': {'average_document_length': 2032.093148525817, 'average_query_length': 85.4298245614035, 'num_documents': 2232016, 'num_queries': 114, 'average_relevant_docs_per_query': 12.912280701754385}, 'rus': {'average_document_length': 1757.9129983233004, 'average_query_length': 85.58771929824562, 'num_documents': 4627543, 'num_queries': 114, 'average_relevant_docs_per_query': 16.57017543859649}, 'zho': {'average_document_length': 743.1426659901881, 'average_query_length': 24.17543859649123, 'num_documents': 3179209, 'num_queries': 114, 'average_relevant_docs_per_query': 18.710526315789473}}} | -| [NeuCLIR2022RetrievalHardNegatives](https://neuclir.github.io/) (Lawrie et al., 2023) | ['fas', 'rus', 'zho'] | Retrieval | s2p | [News, Written] | None | {'test': {'average_document_length': 2066.9453653646488, 'average_query_length': 63.529411764705884, 'num_documents': 27931, 'num_queries': 136, 'average_relevant_docs_per_query': 40.39705882352941, 'hf_subset_descriptive_stats': {'fas': {'average_document_length': 2816.847782031074, 'average_query_length': 83.26666666666667, 'num_documents': 8882, 'num_queries': 45, 'average_relevant_docs_per_query': 32.71111111111111}, 'rus': {'average_document_length': 2446.5574277854193, 'average_query_length': 85.56818181818181, 'num_documents': 8724, 'num_queries': 44, 'average_relevant_docs_per_query': 42.93181818181818}, 'zho': {'average_document_length': 1101.0984987893462, 'average_query_length': 24.0, 'num_documents': 10325, 'num_queries': 47, 'average_relevant_docs_per_query': 45.38297872340426}}}} | -| [NeuCLIR2023Retrieval](https://neuclir.github.io/) (Dawn Lawrie, 2024) | ['fas', 'rus', 'zho'] | Retrieval | s2p | [News, Written] | {'fas': 2232092, 'zho': 3179285, 'rus': 4627619} | {'test': {'fas': {'average_document_length': 2032.093148525817, 'average_query_length': 65.48684210526316, 'num_documents': 2232016, 'num_queries': 76, 'average_relevant_docs_per_query': 66.28947368421052}, 'rus': {'average_document_length': 1757.9129983233004, 'average_query_length': 74.4342105263158, 'num_documents': 4627543, 'num_queries': 76, 'average_relevant_docs_per_query': 62.223684210526315}, 'zho': {'average_document_length': 743.1426659901881, 'average_query_length': 22.210526315789473, 'num_documents': 3179209, 'num_queries': 76, 'average_relevant_docs_per_query': 53.68421052631579}}} | -| [NeuCLIR2023RetrievalHardNegatives](https://neuclir.github.io/) (Dawn Lawrie, 2024) | ['fas', 'rus', 'zho'] | Retrieval | s2p | [News, Written] | None | {'test': {'average_document_length': 2236.175955333482, 'average_query_length': 54.10267857142857, 'num_documents': 49433, 'num_queries': 224, 'average_relevant_docs_per_query': 61.816964285714285, 'hf_subset_descriptive_stats': {'fas': {'average_document_length': 2895.869857421016, 'average_query_length': 65.89189189189189, 'num_documents': 15921, 'num_queries': 74, 'average_relevant_docs_per_query': 68.08108108108108}, 'rus': {'average_document_length': 2724.294762109928, 'average_query_length': 74.41333333333333, 'num_documents': 16247, 'num_queries': 75, 'average_relevant_docs_per_query': 63.053333333333335}, 'zho': {'average_document_length': 1168.4984071821605, 'average_query_length': 22.16, 'num_documents': 17265, 'num_queries': 75, 'average_relevant_docs_per_query': 54.4}}}} | -| [News21InstructionRetrieval](https://arxiv.org/abs/2403.15246) (Orion Weller, 2024) | ['eng'] | InstructionRetrieval | s2p | [News, Written] | {'eng': 61906} | {'eng': 2983.724665391969} | -| [NewsClassification](https://arxiv.org/abs/1509.01626) (Zhang et al., 2015) | ['eng'] | Classification | s2s | [News, Written] | {'test': 7600} | {'test': 235.29} | -| [NoRecClassification](https://aclanthology.org/L18-1661/) | ['nob'] | Classification | s2s | [Written, Reviews] | {'test': 2050} | {'test': 82} | -| [NollySentiBitextMining](https://github.com/IyanuSh/NollySenti) (Shode et al., 2023) | ['eng', 'hau', 'ibo', 'pcm', 'yor'] | BitextMining | s2s | [Social, Reviews, Written] | {'train': 1640} | {'train': 135.91} | -| [NorQuadRetrieval](https://aclanthology.org/2023.nodalida-1.17/) | ['nob'] | Retrieval | p2p | [Encyclopaedic, Non-fiction, Written] | {'test': 2602} | {'test': {'average_document_length': 214.5114503816794, 'average_query_length': 47.896484375, 'num_documents': 1048, 'num_queries': 1024, 'average_relevant_docs_per_query': 2.0}} | -| [NordicLangClassification](https://aclanthology.org/2021.vardial-1.8/) | ['dan', 'fao', 'isl', 'nno', 'nob', 'swe'] | Classification | s2s | [Encyclopaedic] | {'test': 3000} | {'test': 78.2} | -| [NorwegianCourtsBitextMining](https://opus.nlpl.eu/index.php) (Tiedemann et al., 2020) | ['nno', 'nob'] | BitextMining | s2s | [Legal, Written] | {'test': 2050} | {'test': 1884.0} | -| [NorwegianParliamentClassification](https://huggingface.co/datasets/NbAiLab/norwegian_parliament) | ['nob'] | Classification | s2s | [Government, Spoken] | {'test': 1200, 'validation': 1200} | {'test': 1884.0, 'validation': 1911.0} | -| [NusaParagraphEmotionClassification](https://github.com/IndoNLP/nusa-writes) | ['bbc', 'bew', 'bug', 'jav', 'mad', 'mak', 'min', 'mui', 'rej', 'sun'] | Classification | s2s | [Non-fiction, Fiction, Written] | {'train': 15516, 'validation': 2948, 'test': 6250} | {'train': 740.24, 'validation': 740.66, 'test': 740.71} | -| [NusaParagraphTopicClassification](https://github.com/IndoNLP/nusa-writes) | ['bbc', 'bew', 'bug', 'jav', 'mad', 'mak', 'min', 'mui', 'rej', 'sun'] | Classification | s2s | [Non-fiction, Fiction, Written] | {'train': 15516, 'validation': 2948, 'test': 6250} | {'train': 740.24, 'validation': 740.66, 'test': 740.71} | -| [NusaTranslationBitextMining](https://huggingface.co/datasets/indonlp/nusatranslation_mt) (Cahyawijaya et al., 2023) | ['abs', 'bbc', 'bew', 'bhp', 'ind', 'jav', 'mad', 'mak', 'min', 'mui', 'rej', 'sun'] | BitextMining | s2s | [Social, Written] | {'train': 50200} | {'train': {'average_sentence1_length': 145.4552390438247, 'average_sentence2_length': 148.56607569721115, 'num_samples': 50200, 'hf_subset_descriptive_stats': {'ind-abs': {'average_sentence1_length': 148.366, 'average_sentence2_length': 147.314, 'num_samples': 1000}, 'ind-btk': {'average_sentence1_length': 145.36666666666667, 'average_sentence2_length': 146.74045454545455, 'num_samples': 6600}, 'ind-bew': {'average_sentence1_length': 145.4280303030303, 'average_sentence2_length': 148.40530303030303, 'num_samples': 6600}, 'ind-bhp': {'average_sentence1_length': 133.528, 'average_sentence2_length': 128.138, 'num_samples': 1000}, 'ind-jav': {'average_sentence1_length': 145.42772727272728, 'average_sentence2_length': 145.8089393939394, 'num_samples': 6600}, 'ind-mad': {'average_sentence1_length': 145.35545454545453, 'average_sentence2_length': 153.6228787878788, 'num_samples': 6600}, 'ind-mak': {'average_sentence1_length': 145.42772727272728, 'average_sentence2_length': 150.6128787878788, 'num_samples': 6600}, 'ind-min': {'average_sentence1_length': 145.42772727272728, 'average_sentence2_length': 148.0621212121212, 'num_samples': 6600}, 'ind-mui': {'average_sentence1_length': 150.454, 'average_sentence2_length': 150.994, 'num_samples': 1000}, 'ind-rej': {'average_sentence1_length': 151.622, 'average_sentence2_length': 139.583, 'num_samples': 1000}, 'ind-sun': {'average_sentence1_length': 145.42772727272728, 'average_sentence2_length': 150.9880303030303, 'num_samples': 6600}}}} | -| [NusaX-senti](https://arxiv.org/abs/2205.15960) (Winata et al., 2022) | ['ace', 'ban', 'bbc', 'bjn', 'bug', 'eng', 'ind', 'jav', 'mad', 'min', 'nij', 'sun'] | Classification | s2s | [Reviews, Web, Social, Constructed, Written] | {'test': 4800} | {'test': 52.4} | -| [NusaXBitextMining](https://huggingface.co/datasets/indonlp/NusaX-senti/) (Winata et al., 2023) | ['ace', 'ban', 'bbc', 'bjn', 'bug', 'eng', 'ind', 'jav', 'mad', 'min', 'nij', 'sun'] | BitextMining | s2s | [Reviews, Written] | {'train': 5500} | {'train': 157.15} | -| [OPP115DataRetentionLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 88} | {'test': 195.2} | -| [OPP115DataSecurityLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 1334} | {'test': 246.69} | -| [OPP115DoNotTrackLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 110} | {'test': 223.16} | -| [OPP115FirstPartyCollectionUseLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 2086} | {'test': 204.25} | -| [OPP115InternationalAndSpecificAudiencesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 980} | {'test': 327.71} | -| [OPP115PolicyChangeLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 431} | {'test': 200.99} | -| [OPP115ThirdPartySharingCollectionLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 1590} | {'test': 223.64} | -| [OPP115UserAccessEditAndDeletionLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 462} | {'test': 218.59} | -| [OPP115UserChoiceControlLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 1546} | {'test': 210.62} | +| [MultilingualSentimentClassification](https://huggingface.co/datasets/mteb/multilingual-sentiment-classification) | ['ara', 'bam', 'bul', 'cmn', 'cym', 'deu', 'dza', 'ell', 'eng', 'eus', 'fas', 'fin', 'heb', 'hrv', 'ind', 'jpn', 'kor', 'mlt', 'nor', 'pol', 'rus', 'slk', 'spa', 'tha', 'tur', 'uig', 'urd', 'vie', 'zho'] | Classification | s2s | [Reviews, Written] | None | None | +| [MyanmarNews](https://huggingface.co/datasets/myanmar_news) (A. H. Khine, 2017) | ['mya'] | Classification | p2p | [News, Written] | None | None | +| [NFCorpus](https://www.cl.uni-heidelberg.de/statnlpgroup/nfcorpus/) (Boteva et al., 2016) | ['eng'] | Retrieval | s2p | | None | None | +| [NFCorpus-PL](https://www.cl.uni-heidelberg.de/statnlpgroup/nfcorpus/) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | | None | None | +| [NLPJournalAbsIntroRetrieval](https://github.com/sbintuitions/JMTEB) | ['jpn'] | Retrieval | s2s | [Academic, Written] | None | None | +| [NLPJournalTitleAbsRetrieval](https://github.com/sbintuitions/JMTEB) | ['jpn'] | Retrieval | s2s | [Academic, Written] | None | None | +| [NLPJournalTitleIntroRetrieval](https://github.com/sbintuitions/JMTEB) | ['jpn'] | Retrieval | s2s | [Academic, Written] | None | None | +| [NQ](https://ai.google.com/research/NaturalQuestions/) (Tom Kwiatkowski, 2019) | ['eng'] | Retrieval | s2p | | None | None | +| [NQ-PL](https://ai.google.com/research/NaturalQuestions/) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | | None | None | +| [NQ-PLHardNegatives](https://ai.google.com/research/NaturalQuestions/) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | | None | None | +| [NQHardNegatives](https://ai.google.com/research/NaturalQuestions/) (Tom Kwiatkowski, 2019) | ['eng'] | Retrieval | s2p | | None | None | +| [NTREXBitextMining](https://huggingface.co/datasets/davidstap/NTREX) | ['afr', 'amh', 'arb', 'aze', 'bak', 'bel', 'bem', 'ben', 'bod', 'bos', 'bul', 'cat', 'ces', 'ckb', 'cym', 'dan', 'deu', 'div', 'dzo', 'ell', 'eng', 'eus', 'ewe', 'fao', 'fas', 'fij', 'fil', 'fin', 'fra', 'fuc', 'gle', 'glg', 'guj', 'hau', 'heb', 'hin', 'hmn', 'hrv', 'hun', 'hye', 'ibo', 'ind', 'isl', 'ita', 'jpn', 'kan', 'kat', 'kaz', 'khm', 'kin', 'kir', 'kmr', 'kor', 'lao', 'lav', 'lit', 'ltz', 'mal', 'mar', 'mey', 'mkd', 'mlg', 'mlt', 'mon', 'mri', 'msa', 'mya', 'nde', 'nep', 'nld', 'nno', 'nob', 'nso', 'nya', 'orm', 'pan', 'pol', 'por', 'prs', 'pus', 'ron', 'rus', 'shi', 'sin', 'slk', 'slv', 'smo', 'sna', 'snd', 'som', 'spa', 'sqi', 'srp', 'ssw', 'swa', 'swe', 'tah', 'tam', 'tat', 'tel', 'tgk', 'tha', 'tir', 'ton', 'tsn', 'tuk', 'tur', 'uig', 'ukr', 'urd', 'uzb', 'ven', 'vie', 'wol', 'xho', 'yor', 'yue', 'zho', 'zul'] | BitextMining | s2s | [News, Written] | None | None | +| [NYSJudicialEthicsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [NaijaSenti](https://github.com/hausanlp/NaijaSenti) | ['hau', 'ibo', 'pcm', 'yor'] | Classification | s2s | [Social, Written] | None | None | +| [NarrativeQARetrieval](https://metatext.io/datasets/narrativeqa) (Tomáš Kočiský, 2017) | ['eng'] | Retrieval | s2p | | None | None | +| [NepaliNewsClassification](https://github.com/goru001/nlp-for-nepali) | ['nep'] | Classification | s2s | [News, Written] | None | None | +| [NeuCLIR2022Retrieval](https://neuclir.github.io/) (Lawrie et al., 2023) | ['fas', 'rus', 'zho'] | Retrieval | s2p | [News, Written] | None | None | +| [NeuCLIR2022RetrievalHardNegatives](https://neuclir.github.io/) (Lawrie et al., 2023) | ['fas', 'rus', 'zho'] | Retrieval | s2p | [News, Written] | None | None | +| [NeuCLIR2023Retrieval](https://neuclir.github.io/) (Dawn Lawrie, 2024) | ['fas', 'rus', 'zho'] | Retrieval | s2p | [News, Written] | None | None | +| [NeuCLIR2023RetrievalHardNegatives](https://neuclir.github.io/) (Dawn Lawrie, 2024) | ['fas', 'rus', 'zho'] | Retrieval | s2p | [News, Written] | None | None | +| [News21InstructionRetrieval](https://arxiv.org/abs/2403.15246) (Orion Weller, 2024) | ['eng'] | InstructionRetrieval | s2p | [News, Written] | None | None | +| [NewsClassification](https://arxiv.org/abs/1509.01626) (Zhang et al., 2015) | ['eng'] | Classification | s2s | [News, Written] | None | None | +| [NoRecClassification](https://aclanthology.org/L18-1661/) | ['nob'] | Classification | s2s | [Written, Reviews] | None | None | +| [NollySentiBitextMining](https://github.com/IyanuSh/NollySenti) (Shode et al., 2023) | ['eng', 'hau', 'ibo', 'pcm', 'yor'] | BitextMining | s2s | [Social, Reviews, Written] | None | None | +| [NorQuadRetrieval](https://aclanthology.org/2023.nodalida-1.17/) | ['nob'] | Retrieval | p2p | [Encyclopaedic, Non-fiction, Written] | None | None | +| [NordicLangClassification](https://aclanthology.org/2021.vardial-1.8/) | ['dan', 'fao', 'isl', 'nno', 'nob', 'swe'] | Classification | s2s | [Encyclopaedic] | None | None | +| [NorwegianCourtsBitextMining](https://opus.nlpl.eu/index.php) (Tiedemann et al., 2020) | ['nno', 'nob'] | BitextMining | s2s | [Legal, Written] | None | None | +| [NorwegianParliamentClassification](https://huggingface.co/datasets/NbAiLab/norwegian_parliament) | ['nob'] | Classification | s2s | [Government, Spoken] | None | None | +| [NusaParagraphEmotionClassification](https://github.com/IndoNLP/nusa-writes) | ['bbc', 'bew', 'bug', 'jav', 'mad', 'mak', 'min', 'mui', 'rej', 'sun'] | Classification | s2s | [Non-fiction, Fiction, Written] | None | None | +| [NusaParagraphTopicClassification](https://github.com/IndoNLP/nusa-writes) | ['bbc', 'bew', 'bug', 'jav', 'mad', 'mak', 'min', 'mui', 'rej', 'sun'] | Classification | s2s | [Non-fiction, Fiction, Written] | None | None | +| [NusaTranslationBitextMining](https://huggingface.co/datasets/indonlp/nusatranslation_mt) (Cahyawijaya et al., 2023) | ['abs', 'bbc', 'bew', 'bhp', 'ind', 'jav', 'mad', 'mak', 'min', 'mui', 'rej', 'sun'] | BitextMining | s2s | [Social, Written] | {'train': 50200} | {'train': {'average_sentence1_length': 145.46, 'average_sentence2_length': 148.57, 'num_samples': 50200, 'number_of_characters': 14759870, 'hf_subset_descriptive_stats': {'ind-abs': {'average_sentence1_length': 148.37, 'average_sentence2_length': 147.31, 'num_samples': 1000, 'number_of_characters': 295680}, 'ind-btk': {'average_sentence1_length': 145.37, 'average_sentence2_length': 146.74, 'num_samples': 6600, 'number_of_characters': 1927907}, 'ind-bew': {'average_sentence1_length': 145.43, 'average_sentence2_length': 148.41, 'num_samples': 6600, 'number_of_characters': 1939300}, 'ind-bhp': {'average_sentence1_length': 133.53, 'average_sentence2_length': 128.14, 'num_samples': 1000, 'number_of_characters': 261666}, 'ind-jav': {'average_sentence1_length': 145.43, 'average_sentence2_length': 145.81, 'num_samples': 6600, 'number_of_characters': 1922162}, 'ind-mad': {'average_sentence1_length': 145.36, 'average_sentence2_length': 153.62, 'num_samples': 6600, 'number_of_characters': 1973257}, 'ind-mak': {'average_sentence1_length': 145.43, 'average_sentence2_length': 150.61, 'num_samples': 6600, 'number_of_characters': 1953868}, 'ind-min': {'average_sentence1_length': 145.43, 'average_sentence2_length': 148.06, 'num_samples': 6600, 'number_of_characters': 1937033}, 'ind-mui': {'average_sentence1_length': 150.45, 'average_sentence2_length': 150.99, 'num_samples': 1000, 'number_of_characters': 301448}, 'ind-rej': {'average_sentence1_length': 151.62, 'average_sentence2_length': 139.58, 'num_samples': 1000, 'number_of_characters': 291205}, 'ind-sun': {'average_sentence1_length': 145.43, 'average_sentence2_length': 150.99, 'num_samples': 6600, 'number_of_characters': 1956344}}}} | +| [NusaX-senti](https://arxiv.org/abs/2205.15960) (Winata et al., 2022) | ['ace', 'ban', 'bbc', 'bjn', 'bug', 'eng', 'ind', 'jav', 'mad', 'min', 'nij', 'sun'] | Classification | s2s | [Reviews, Web, Social, Constructed, Written] | None | None | +| [NusaXBitextMining](https://huggingface.co/datasets/indonlp/NusaX-senti/) (Winata et al., 2023) | ['ace', 'ban', 'bbc', 'bjn', 'bug', 'eng', 'ind', 'jav', 'mad', 'min', 'nij', 'sun'] | BitextMining | s2s | [Reviews, Written] | None | None | +| [OPP115DataRetentionLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [OPP115DataSecurityLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [OPP115DoNotTrackLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [OPP115FirstPartyCollectionUseLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [OPP115InternationalAndSpecificAudiencesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [OPP115PolicyChangeLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [OPP115ThirdPartySharingCollectionLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [OPP115UserAccessEditAndDeletionLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [OPP115UserChoiceControlLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | | [Ocnli](https://arxiv.org/abs/2010.05444) (Hai Hu, 2020) | ['cmn'] | PairClassification | s2s | | None | None | -| [OdiaNewsClassification](https://github.com/goru001/nlp-for-odia) (Anoop Kunchukuttan, 2020) | ['ory'] | Classification | s2s | [News, Written] | {'test': 2048} | {'test': 49.24} | +| [OdiaNewsClassification](https://github.com/goru001/nlp-for-odia) (Anoop Kunchukuttan, 2020) | ['ory'] | Classification | s2s | [News, Written] | None | None | | [OnlineShopping](https://aclanthology.org/2023.nodalida-1.20/) (Xiao et al., 2023) | ['cmn'] | Classification | s2s | | None | None | -| [OnlineStoreReviewSentimentClassification](https://huggingface.co/datasets/Ruqiya/Arabic_Reviews_of_SHEIN) | ['ara'] | Classification | s2s | [Reviews, Written] | {'train': 2048} | {'train': 137.2} | -| [OpusparcusPC](https://gem-benchmark.com/data_cards/opusparcus) (Mathias Creutz, 2018) | ['deu', 'eng', 'fin', 'fra', 'rus', 'swe'] | PairClassification | s2s | [Spoken, Spoken] | {'validation': 10168, 'test': 10210} | {'validation': 24.4, 'test': 23.8} | -| [OralArgumentQuestionPurposeLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 312} | {'test': 269.71} | -| [OverrulingLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 2048} | {'test': 167.2} | -| [PAC](https://arxiv.org/pdf/2211.13112.pdf) (Łukasz Augustyniak, 2022) | ['pol'] | Classification | p2p | [Legal, Written] | {'test': 3453} | {'test': 185.3} | +| [OnlineStoreReviewSentimentClassification](https://huggingface.co/datasets/Ruqiya/Arabic_Reviews_of_SHEIN) | ['ara'] | Classification | s2s | [Reviews, Written] | None | None | +| [OpusparcusPC](https://gem-benchmark.com/data_cards/opusparcus) (Mathias Creutz, 2018) | ['deu', 'eng', 'fin', 'fra', 'rus', 'swe'] | PairClassification | s2s | [Spoken, Spoken] | None | None | +| [OralArgumentQuestionPurposeLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [OverrulingLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [PAC](https://arxiv.org/pdf/2211.13112.pdf) (Łukasz Augustyniak, 2022) | ['pol'] | Classification | p2p | [Legal, Written] | None | None | | [PAWSX](https://aclanthology.org/2021.emnlp-main.357) (Shitao Xiao, 2024) | ['cmn'] | STS | s2s | | None | None | -| [PIQA](https://arxiv.org/abs/1911.11641) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | {'test': 1838} | {'test': {'average_document_length': 99.89012998705756, 'average_query_length': 36.08052230685528, 'num_documents': 35542, 'num_queries': 1838, 'average_relevant_docs_per_query': 1.0}} | -| [PROALegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 95} | {'test': 251.73} | +| [PIQA](https://arxiv.org/abs/1911.11641) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | None | None | +| [PROALegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | | [PSC](http://www.lrec-conf.org/proceedings/lrec2014/pdf/1211_Paper.pdf) | ['pol'] | PairClassification | s2s | [News, Written] | None | None | -| [PatentClassification](https://aclanthology.org/P19-1212.pdf) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 5000} | {'test': 18620.44} | -| [PawsXPairClassification](https://arxiv.org/abs/1908.11828) (Yinfei Yang, 2019) | ['cmn', 'deu', 'eng', 'fra', 'jpn', 'kor', 'spa'] | PairClassification | s2s | [Web, Encyclopaedic, Written] | {'validation': 14000, 'test': 14000} | {'test': {'num_samples': 14000, 'avg_sentence1_len': 91.17892857142857, 'avg_sentence2_len': 91.10121428571429, 'unique_labels': 2, 'labels': {'1': {'count': 6285}, '0': {'count': 7715}}, 'hf_subset_descriptive_stats': {'de': {'num_samples': 2000, 'avg_sentence1_len': 119.7815, 'avg_sentence2_len': 119.2355, 'unique_labels': 2, 'labels': {'1': {'count': 895}, '0': {'count': 1105}}}, 'en': {'num_samples': 2000, 'avg_sentence1_len': 113.7575, 'avg_sentence2_len': 113.4235, 'unique_labels': 2, 'labels': {'1': {'count': 907}, '0': {'count': 1093}}}, 'es': {'num_samples': 2000, 'avg_sentence1_len': 117.815, 'avg_sentence2_len': 117.798, 'unique_labels': 2, 'labels': {'1': {'count': 907}, '0': {'count': 1093}}}, 'fr': {'num_samples': 2000, 'avg_sentence1_len': 120.028, 'avg_sentence2_len': 119.9885, 'unique_labels': 2, 'labels': {'1': {'count': 903}, '0': {'count': 1097}}}, 'ja': {'num_samples': 2000, 'avg_sentence1_len': 58.678, 'avg_sentence2_len': 58.875, 'unique_labels': 2, 'labels': {'1': {'count': 883}, '0': {'count': 1117}}}, 'ko': {'num_samples': 2000, 'avg_sentence1_len': 64.9605, 'avg_sentence2_len': 65.114, 'unique_labels': 2, 'labels': {'1': {'count': 896}, '0': {'count': 1104}}}, 'zh': {'num_samples': 2000, 'avg_sentence1_len': 43.232, 'avg_sentence2_len': 43.274, 'unique_labels': 2, 'labels': {'1': {'count': 894}, '0': {'count': 1106}}}}}, 'validation': {'num_samples': 14000, 'avg_sentence1_len': 90.12585714285714, 'avg_sentence2_len': 90.2045, 'unique_labels': 2, 'labels': {'1': {'count': 5948}, '0': {'count': 8052}}, 'hf_subset_descriptive_stats': {'de': {'num_samples': 2000, 'avg_sentence1_len': 116.82, 'avg_sentence2_len': 117.0015, 'unique_labels': 2, 'labels': {'1': {'count': 831}, '0': {'count': 1169}}}, 'en': {'num_samples': 2000, 'avg_sentence1_len': 113.1075, 'avg_sentence2_len': 112.858, 'unique_labels': 2, 'labels': {'1': {'count': 863}, '0': {'count': 1137}}}, 'es': {'num_samples': 2000, 'avg_sentence1_len': 116.3285, 'avg_sentence2_len': 116.7275, 'unique_labels': 2, 'labels': {'1': {'count': 847}, '0': {'count': 1153}}}, 'fr': {'num_samples': 2000, 'avg_sentence1_len': 119.5045, 'avg_sentence2_len': 119.7505, 'unique_labels': 2, 'labels': {'1': {'count': 860}, '0': {'count': 1140}}}, 'ja': {'num_samples': 2000, 'avg_sentence1_len': 57.5105, 'avg_sentence2_len': 57.317, 'unique_labels': 2, 'labels': {'1': {'count': 854}, '0': {'count': 1146}}}, 'ko': {'num_samples': 2000, 'avg_sentence1_len': 65.162, 'avg_sentence2_len': 65.5155, 'unique_labels': 2, 'labels': {'1': {'count': 840}, '0': {'count': 1160}}}, 'zh': {'num_samples': 2000, 'avg_sentence1_len': 42.448, 'avg_sentence2_len': 42.2615, 'unique_labels': 2, 'labels': {'1': {'count': 853}, '0': {'count': 1147}}}}}} | -| [PersianFoodSentimentClassification](https://hooshvare.github.io/docs/datasets/sa) (Mehrdad Farahani et al., 2020) | ['fas'] | Classification | s2s | [Reviews, Written] | {'validation': 2048, 'test': 2048} | {'validation': 90.37, 'test': 90.58} | -| [PersonalJurisdictionLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 50} | {'test': 381.14} | -| [PhincBitextMining](https://huggingface.co/datasets/veezbo/phinc) (Srivastava et al., 2020) | ['eng', 'hin'] | BitextMining | s2s | [Social, Written] | {'train': 13738} | {'train': 75.32} | -| [PlscClusteringP2P.v2](https://huggingface.co/datasets/rafalposwiata/plsc) | ['pol'] | Clustering | s2s | [Academic, Written] | {'test': 2048} | {'test': 1023.21} | -| [PlscClusteringS2S.v2](https://huggingface.co/datasets/rafalposwiata/plsc) | ['pol'] | Clustering | s2s | [Academic, Written] | {'test': 2048} | {'test': 84.34} | -| [PoemSentimentClassification](https://arxiv.org/abs/2011.02686) (Emily Sheng, 2020) | ['eng'] | Classification | s2s | [Reviews, Written] | {'validation': 105, 'test': 104} | {'validation': 45.3, 'test': 42.4} | +| [PatentClassification](https://aclanthology.org/P19-1212.pdf) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [PawsXPairClassification](https://arxiv.org/abs/1908.11828) (Yinfei Yang, 2019) | ['cmn', 'deu', 'eng', 'fra', 'jpn', 'kor', 'spa'] | PairClassification | s2s | [Web, Encyclopaedic, Written] | {'test': 14000, 'validation': 14000} | {'test': {'num_samples': 14000, 'number_of_characters': 2551922, 'avg_sentence1_len': 91.18, 'avg_sentence2_len': 91.1, 'unique_labels': 2, 'labels': {'1': {'count': 6285}, '0': {'count': 7715}}, 'hf_subset_descriptive_stats': {'de': {'num_samples': 2000, 'number_of_characters': 478034, 'avg_sentence1_len': 119.78, 'avg_sentence2_len': 119.24, 'unique_labels': 2, 'labels': {'1': {'count': 895}, '0': {'count': 1105}}}, 'en': {'num_samples': 2000, 'number_of_characters': 454362, 'avg_sentence1_len': 113.76, 'avg_sentence2_len': 113.42, 'unique_labels': 2, 'labels': {'1': {'count': 907}, '0': {'count': 1093}}}, 'es': {'num_samples': 2000, 'number_of_characters': 471226, 'avg_sentence1_len': 117.81, 'avg_sentence2_len': 117.8, 'unique_labels': 2, 'labels': {'1': {'count': 907}, '0': {'count': 1093}}}, 'fr': {'num_samples': 2000, 'number_of_characters': 480033, 'avg_sentence1_len': 120.03, 'avg_sentence2_len': 119.99, 'unique_labels': 2, 'labels': {'1': {'count': 903}, '0': {'count': 1097}}}, 'ja': {'num_samples': 2000, 'number_of_characters': 235106, 'avg_sentence1_len': 58.68, 'avg_sentence2_len': 58.88, 'unique_labels': 2, 'labels': {'1': {'count': 883}, '0': {'count': 1117}}}, 'ko': {'num_samples': 2000, 'number_of_characters': 260149, 'avg_sentence1_len': 64.96, 'avg_sentence2_len': 65.11, 'unique_labels': 2, 'labels': {'1': {'count': 896}, '0': {'count': 1104}}}, 'zh': {'num_samples': 2000, 'number_of_characters': 173012, 'avg_sentence1_len': 43.23, 'avg_sentence2_len': 43.27, 'unique_labels': 2, 'labels': {'1': {'count': 894}, '0': {'count': 1106}}}}}, 'validation': {'num_samples': 14000, 'number_of_characters': 2524625, 'avg_sentence1_len': 90.13, 'avg_sentence2_len': 90.2, 'unique_labels': 2, 'labels': {'1': {'count': 5948}, '0': {'count': 8052}}, 'hf_subset_descriptive_stats': {'de': {'num_samples': 2000, 'number_of_characters': 467643, 'avg_sentence1_len': 116.82, 'avg_sentence2_len': 117.0, 'unique_labels': 2, 'labels': {'1': {'count': 831}, '0': {'count': 1169}}}, 'en': {'num_samples': 2000, 'number_of_characters': 451931, 'avg_sentence1_len': 113.11, 'avg_sentence2_len': 112.86, 'unique_labels': 2, 'labels': {'1': {'count': 863}, '0': {'count': 1137}}}, 'es': {'num_samples': 2000, 'number_of_characters': 466112, 'avg_sentence1_len': 116.33, 'avg_sentence2_len': 116.73, 'unique_labels': 2, 'labels': {'1': {'count': 847}, '0': {'count': 1153}}}, 'fr': {'num_samples': 2000, 'number_of_characters': 478510, 'avg_sentence1_len': 119.5, 'avg_sentence2_len': 119.75, 'unique_labels': 2, 'labels': {'1': {'count': 860}, '0': {'count': 1140}}}, 'ja': {'num_samples': 2000, 'number_of_characters': 229655, 'avg_sentence1_len': 57.51, 'avg_sentence2_len': 57.32, 'unique_labels': 2, 'labels': {'1': {'count': 854}, '0': {'count': 1146}}}, 'ko': {'num_samples': 2000, 'number_of_characters': 261355, 'avg_sentence1_len': 65.16, 'avg_sentence2_len': 65.52, 'unique_labels': 2, 'labels': {'1': {'count': 840}, '0': {'count': 1160}}}, 'zh': {'num_samples': 2000, 'number_of_characters': 169419, 'avg_sentence1_len': 42.45, 'avg_sentence2_len': 42.26, 'unique_labels': 2, 'labels': {'1': {'count': 853}, '0': {'count': 1147}}}}}} | +| [PersianFoodSentimentClassification](https://hooshvare.github.io/docs/datasets/sa) (Mehrdad Farahani et al., 2020) | ['fas'] | Classification | s2s | [Reviews, Written] | None | None | +| [PersonalJurisdictionLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [PhincBitextMining](https://huggingface.co/datasets/veezbo/phinc) (Srivastava et al., 2020) | ['eng', 'hin'] | BitextMining | s2s | [Social, Written] | None | None | +| [PlscClusteringP2P.v2](https://huggingface.co/datasets/rafalposwiata/plsc) | ['pol'] | Clustering | s2s | [Academic, Written] | None | None | +| [PlscClusteringS2S.v2](https://huggingface.co/datasets/rafalposwiata/plsc) | ['pol'] | Clustering | s2s | [Academic, Written] | None | None | +| [PoemSentimentClassification](https://arxiv.org/abs/2011.02686) (Emily Sheng, 2020) | ['eng'] | Classification | s2s | [Reviews, Written] | None | None | | [PolEmo2.0-IN](https://aclanthology.org/K19-1092.pdf) | ['pol'] | Classification | s2s | [Written, Social] | None | None | -| [PolEmo2.0-OUT](https://aclanthology.org/K19-1092.pdf) | ['pol'] | Classification | s2s | [Written, Social] | {'test': 722} | {'test': 756.2} | +| [PolEmo2.0-OUT](https://aclanthology.org/K19-1092.pdf) | ['pol'] | Classification | s2s | [Written, Social] | None | None | | [PpcPC](https://arxiv.org/pdf/2207.12759.pdf) (Sławomir Dadas, 2022) | ['pol'] | PairClassification | s2s | [Fiction, Non-fiction, Web, Written, Spoken, Social, News] | None | None | -| [PublicHealthQA](https://huggingface.co/datasets/xhluca/publichealth-qa) | ['ara', 'eng', 'fra', 'kor', 'rus', 'spa', 'vie', 'zho'] | Retrieval | s2p | [Medical, Government, Web, Written] | {'test': 888} | {'test': {'arabic': {'average_document_length': 836.8850574712644, 'average_query_length': 79.84883720930233, 'num_documents': 87, 'num_queries': 87, 'average_relevant_docs_per_query': 1.0}, 'chinese': {'average_document_length': 239.58282208588957, 'average_query_length': 24.828220858895705, 'num_documents': 163, 'num_queries': 163, 'average_relevant_docs_per_query': 1.0}, 'english': {'average_document_length': 799.3430232558139, 'average_query_length': 71.78488372093024, 'num_documents': 172, 'num_queries': 172, 'average_relevant_docs_per_query': 1.0}, 'french': {'average_document_length': 1021.6823529411764, 'average_query_length': 101.88235294117646, 'num_documents': 85, 'num_queries': 85, 'average_relevant_docs_per_query': 1.0}, 'korean': {'average_document_length': 339.0, 'average_query_length': 36.90909090909091, 'num_documents': 77, 'num_queries': 77, 'average_relevant_docs_per_query': 1.0}, 'russian': {'average_document_length': 985.1076923076923, 'average_query_length': 85.2, 'num_documents': 65, 'num_queries': 65, 'average_relevant_docs_per_query': 1.0}, 'spanish': {'average_document_length': 941.1666666666666, 'average_query_length': 84.67901234567901, 'num_documents': 162, 'num_queries': 162, 'average_relevant_docs_per_query': 1.0}, 'vietnamese': {'average_document_length': 704.5454545454545, 'average_query_length': 71.83116883116882, 'num_documents': 77, 'num_queries': 77, 'average_relevant_docs_per_query': 1.0}}} | -| [PunjabiNewsClassification](https://github.com/goru001/nlp-for-punjabi/) (Anoop Kunchukuttan, 2020) | ['pan'] | Classification | s2s | [News, Written] | {'train': 627, 'test': 157} | {'train': 4222.22, 'test': 4115.14} | +| [PublicHealthQA](https://huggingface.co/datasets/xhluca/publichealth-qa) | ['ara', 'eng', 'fra', 'kor', 'rus', 'spa', 'vie', 'zho'] | Retrieval | s2p | [Medical, Government, Web, Written] | None | None | +| [PunjabiNewsClassification](https://github.com/goru001/nlp-for-punjabi/) (Anoop Kunchukuttan, 2020) | ['pan'] | Classification | s2s | [News, Written] | None | None | | [QBQTC](https://github.com/CLUEbenchmark/QBQTC/tree/main/dataset) | ['cmn'] | STS | s2s | | None | None | -| [Quail](https://text-machine.cs.uml.edu/lab2/projects/quail/) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | {'test': 2720} | {'test': {'average_document_length': 27.50788422240522, 'average_query_length': 1957.3632352941177, 'num_documents': 32787, 'num_queries': 2720, 'average_relevant_docs_per_query': 1.0}} | -| [Quora-PL](https://quoradata.quora.com/First-Quora-Dataset-Release-Question-Pairs) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2s | | None | {'validation': {'average_document_length': 65.82473022253414, 'average_query_length': 54.6006, 'num_documents': 522931, 'num_queries': 5000, 'average_relevant_docs_per_query': 1.5252}, 'test': {'average_document_length': 65.82473022253414, 'average_query_length': 54.5354, 'num_documents': 522931, 'num_queries': 10000, 'average_relevant_docs_per_query': 1.5675}} | -| [Quora-PLHardNegatives](https://quoradata.quora.com/First-Quora-Dataset-Release-Question-Pairs) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2s | | {'test': 1000} | {'test': {'average_document_length': 67.77529631287385, 'average_query_length': 53.846, 'num_documents': 172031, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.641}} | -| [QuoraRetrieval](https://quoradata.quora.com/First-Quora-Dataset-Release-Question-Pairs) (DataCanary et al., 2017) | ['eng'] | Retrieval | s2s | | None | {'dev': {'average_document_length': 62.158154708747425, 'average_query_length': 51.5342, 'num_documents': 522931, 'num_queries': 5000, 'average_relevant_docs_per_query': 1.5252}, 'test': {'average_document_length': 62.158154708747425, 'average_query_length': 51.5396, 'num_documents': 522931, 'num_queries': 10000, 'average_relevant_docs_per_query': 1.5675}} | -| [QuoraRetrievalHardNegatives](https://quoradata.quora.com/First-Quora-Dataset-Release-Question-Pairs) (DataCanary et al., 2017) | ['eng'] | Retrieval | s2s | | {'test': 1000} | {'test': {'average_document_length': 58.96963812985781, 'average_query_length': 51.228, 'num_documents': 177163, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.641}} | -| [RARbCode](https://arxiv.org/abs/2404.06347) (Xiao et al., 2024) | ['eng'] | Retrieval | s2p | [Programming, Written] | {'test': 1484} | {'test': {'average_document_length': 793.6813076734267, 'average_query_length': 375.7506738544474, 'num_documents': 301482, 'num_queries': 1484, 'average_relevant_docs_per_query': 1.0}} | -| [RARbMath](https://arxiv.org/abs/2404.06347) (Xiao et al., 2024) | ['eng'] | Retrieval | s2p | [Encyclopaedic, Written] | {'test': 6319} | {'test': {'average_document_length': 504.0197829347469, 'average_query_length': 210.30732710871973, 'num_documents': 389376, 'num_queries': 6319, 'average_relevant_docs_per_query': 1.0}} | -| [RTE3](https://aclanthology.org/W07-1401/) | ['deu', 'eng', 'fra', 'ita'] | PairClassification | s2s | [News, Web, Encyclopaedic, Written] | {'test': 1923} | {'test': 124.79} | -| [RUParaPhraserSTS](https://aclanthology.org/2020.ngt-1.6) (Pivovarova et al., 2017) | ['rus'] | STS | s2s | [News, Written] | {'test': 1924} | {'test': 61.25} | -| [RedditClustering.v2](https://arxiv.org/abs/2104.07081) (Gregor Geigle, 2021) | ['eng'] | Clustering | s2s | [Web, Social, Written] | {'test': 32768} | {'test': 64.7} | -| [RedditClusteringP2P.v2](https://arxiv.org/abs/2104.07081) (Gregor Geigle, 2021) | ['eng'] | Clustering | p2p | [Web, Social, Written] | {'test': 18375} | {'test': 727.7} | -| [RestaurantReviewSentimentClassification](https://link.springer.com/chapter/10.1007/978-3-319-18117-2_2) (ElSahar et al., 2015) | ['ara'] | Classification | s2s | [Reviews, Written] | {'train': 2048} | {'train': 231.4} | -| [RiaNewsRetrieval](https://arxiv.org/abs/1901.07786) (Gavrilov et al., 2019) | ['rus'] | Retrieval | s2p | [News, Written] | {'test': 10000} | {'test': {'average_document_length': 1165.6429557148213, 'average_query_length': 62.4029, 'num_documents': 704344, 'num_queries': 10000, 'average_relevant_docs_per_query': 1.0}} | -| [RiaNewsRetrievalHardNegatives](https://arxiv.org/abs/1901.07786) (Gavrilov et al., 2019) | ['rus'] | Retrieval | s2p | [News, Written] | {'test': 1000} | {'test': {'average_document_length': 1225.7253146619116, 'average_query_length': 62.338, 'num_documents': 191237, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}} | -| [Robust04InstructionRetrieval](https://arxiv.org/abs/2403.15246) (Orion Weller, 2024) | ['eng'] | InstructionRetrieval | s2p | [News, Written] | {'eng': 95088} | {'eng': 2471.0398058252426} | -| [RomaTalesBitextMining](https://idoc.pub/documents/idocpub-zpnxm9g35ylv) | ['hun', 'rom'] | BitextMining | s2s | [Fiction, Written] | {'test': 215} | {'test': 316.8046511627907} | -| [RomaniBibleClustering](https://romani.global.bible/info) | ['rom'] | Clustering | p2p | [Religious, Written] | {'test': 2048} | {'test': 132.2} | -| [RomanianReviewsSentiment](https://arxiv.org/abs/2101.04197) (Anca Maria Tache, 2021) | ['ron'] | Classification | s2s | [Reviews, Written] | {'test': 2048} | {'test': 588.6} | -| [RomanianSentimentClassification](https://arxiv.org/abs/2009.08712) (Dumitrescu et al., 2020) | ['ron'] | Classification | s2s | [Reviews, Written] | {'test': 2048} | {'test': 67.6} | -| [RonSTS](https://openreview.net/forum?id=JH61CD7afTv) (Dumitrescu et al., 2021) | ['ron'] | STS | s2s | [News, Social, Web, Written] | {'test': 1379} | {'test': 60.5} | -| [RuBQReranking](https://openreview.net/pdf?id=P5UQFFoQ4PJ) (Ivan Rybin, 2021) | ['rus'] | Reranking | s2p | [Encyclopaedic, Written] | {'test': 1551} | {'test': 499.9} | -| [RuBQRetrieval](https://openreview.net/pdf?id=P5UQFFoQ4PJ) (Ivan Rybin, 2021) | ['rus'] | Retrieval | s2p | [Encyclopaedic, Written] | {'test': 2845} | {'test': {'average_document_length': 448.94659134903037, 'average_query_length': 45.29609929078014, 'num_documents': 56826, 'num_queries': 1692, 'average_relevant_docs_per_query': 1.6814420803782506}} | -| [RuReviewsClassification](https://github.com/sismetanin/rureviews) (Sergey Smetanin, 2019) | ['rus'] | Classification | p2p | [Reviews, Written] | {'test': 2048} | {'test': 133.2} | -| [RuSTSBenchmarkSTS](https://github.com/PhilipMay/stsb-multi-mt/) (Philip May, 2021) | ['rus'] | STS | s2s | [News, Social, Web, Written] | {'test': 1264} | {'test': 54.2} | -| [RuSciBenchGRNTIClassification](https://github.com/mlsa-iai-msu-lab/ru_sci_bench/) | ['rus'] | Classification | p2p | [Academic, Written] | {'test': 2048} | {'test': 890.1} | -| [RuSciBenchGRNTIClusteringP2P](https://github.com/mlsa-iai-msu-lab/ru_sci_bench/) | ['rus'] | Clustering | p2p | [Academic, Written] | {'test': 2048} | {'test': {'num_samples': 2048, 'average_text_length': 889.81396484375, 'average_labels_per_text': 1.0, 'unique_labels': 28, 'labels': {'3': {'count': 73}, '4': {'count': 73}, '20': {'count': 73}, '9': {'count': 73}, '21': {'count': 73}, '15': {'count': 73}, '16': {'count': 74}, '2': {'count': 73}, '8': {'count': 73}, '23': {'count': 73}, '6': {'count': 73}, '24': {'count': 73}, '10': {'count': 73}, '1': {'count': 73}, '17': {'count': 74}, '14': {'count': 74}, '18': {'count': 73}, '27': {'count': 73}, '19': {'count': 73}, '22': {'count': 73}, '12': {'count': 73}, '25': {'count': 73}, '5': {'count': 74}, '0': {'count': 73}, '26': {'count': 73}, '11': {'count': 73}, '13': {'count': 73}, '7': {'count': 73}}}} | -| [RuSciBenchOECDClassification](https://github.com/mlsa-iai-msu-lab/ru_sci_bench/) | ['rus'] | Classification | p2p | [Academic, Written] | {'test': 2048} | {'test': 838.9} | -| [RuSciBenchOECDClusteringP2P](https://github.com/mlsa-iai-msu-lab/ru_sci_bench/) | ['rus'] | Clustering | p2p | [Academic, Written] | {'test': 2048} | {'test': 838.9} | -| [SCDBPAccountabilityLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 379} | {'test': 3520} | -| [SCDBPAuditsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 379} | {'test': 3507} | -| [SCDBPCertificationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 378} | {'test': 3507} | -| [SCDBPTrainingLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 379} | {'test': 3506} | -| [SCDBPVerificationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 379} | {'test': 3498} | -| [SCDDAccountabilityLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 378} | {'test': 3522} | -| [SCDDAuditsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 379} | {'test': 3506} | -| [SCDDCertificationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 378} | {'test': 3518} | -| [SCDDTrainingLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 379} | {'test': 3499} | -| [SCDDVerificationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 379} | {'test': 3503} | -| [SCIDOCS](https://allenai.org/data/scidocs) (Arman Cohan, 2020) | ['eng'] | Retrieval | s2p | [Academic, Written, Non-fiction] | None | {'test': {'average_document_length': 1203.3659819932182, 'average_query_length': 71.632, 'num_documents': 25657, 'num_queries': 1000, 'average_relevant_docs_per_query': 4.928}} | -| [SCIDOCS-PL](https://allenai.org/data/scidocs) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | | None | {'test': {'average_document_length': 1270.0791986592353, 'average_query_length': 80.671, 'num_documents': 25657, 'num_queries': 1000, 'average_relevant_docs_per_query': 4.928}} | -| [SIB200Classification](https://arxiv.org/abs/2309.07445) (Adelani et al., 2023) | ['ace', 'acm', 'acq', 'aeb', 'afr', 'ajp', 'aka', 'als', 'amh', 'apc', 'arb', 'ars', 'ary', 'arz', 'asm', 'ast', 'awa', 'ayr', 'azb', 'azj', 'bak', 'bam', 'ban', 'bel', 'bem', 'ben', 'bho', 'bjn', 'bod', 'bos', 'bug', 'bul', 'cat', 'ceb', 'ces', 'cjk', 'ckb', 'crh', 'cym', 'dan', 'deu', 'dik', 'dyu', 'dzo', 'ell', 'eng', 'epo', 'est', 'eus', 'ewe', 'fao', 'fij', 'fin', 'fon', 'fra', 'fur', 'fuv', 'gaz', 'gla', 'gle', 'glg', 'grn', 'guj', 'hat', 'hau', 'heb', 'hin', 'hne', 'hrv', 'hun', 'hye', 'ibo', 'ilo', 'ind', 'isl', 'ita', 'jav', 'jpn', 'kab', 'kac', 'kam', 'kan', 'kas', 'kat', 'kaz', 'kbp', 'kea', 'khk', 'khm', 'kik', 'kin', 'kir', 'kmb', 'kmr', 'knc', 'kon', 'kor', 'lao', 'lij', 'lim', 'lin', 'lit', 'lmo', 'ltg', 'ltz', 'lua', 'lug', 'luo', 'lus', 'lvs', 'mag', 'mai', 'mal', 'mar', 'min', 'mkd', 'mlt', 'mni', 'mos', 'mri', 'mya', 'nld', 'nno', 'nob', 'npi', 'nqo', 'nso', 'nus', 'nya', 'oci', 'ory', 'pag', 'pan', 'pap', 'pbt', 'pes', 'plt', 'pol', 'por', 'prs', 'quy', 'ron', 'run', 'rus', 'sag', 'san', 'sat', 'scn', 'shn', 'sin', 'slk', 'slv', 'smo', 'sna', 'snd', 'som', 'sot', 'spa', 'srd', 'srp', 'ssw', 'sun', 'swe', 'swh', 'szl', 'tam', 'taq', 'tat', 'tel', 'tgk', 'tgl', 'tha', 'tir', 'tpi', 'tsn', 'tso', 'tuk', 'tum', 'tur', 'twi', 'tzm', 'uig', 'ukr', 'umb', 'urd', 'uzn', 'vec', 'vie', 'war', 'wol', 'xho', 'ydd', 'yor', 'yue', 'zho', 'zsm', 'zul'] | Classification | s2s | [News, Written] | {'train': 701, 'validation': 99, 'test': 204} | {'train': 111.24, 'validation': 97.11, 'test': 135.53} | -| [SIB200ClusteringS2S](https://arxiv.org/abs/2309.07445) (Adelani et al., 2023) | ['ace', 'acm', 'acq', 'aeb', 'afr', 'ajp', 'aka', 'als', 'amh', 'apc', 'arb', 'ars', 'ary', 'arz', 'asm', 'ast', 'awa', 'ayr', 'azb', 'azj', 'bak', 'bam', 'ban', 'bel', 'bem', 'ben', 'bho', 'bjn', 'bod', 'bos', 'bug', 'bul', 'cat', 'ceb', 'ces', 'cjk', 'ckb', 'crh', 'cym', 'dan', 'deu', 'dik', 'dyu', 'dzo', 'ell', 'eng', 'epo', 'est', 'eus', 'ewe', 'fao', 'fij', 'fin', 'fon', 'fra', 'fur', 'fuv', 'gaz', 'gla', 'gle', 'glg', 'grn', 'guj', 'hat', 'hau', 'heb', 'hin', 'hne', 'hrv', 'hun', 'hye', 'ibo', 'ilo', 'ind', 'isl', 'ita', 'jav', 'jpn', 'kab', 'kac', 'kam', 'kan', 'kas', 'kat', 'kaz', 'kbp', 'kea', 'khk', 'khm', 'kik', 'kin', 'kir', 'kmb', 'kmr', 'knc', 'kon', 'kor', 'lao', 'lij', 'lim', 'lin', 'lit', 'lmo', 'ltg', 'ltz', 'lua', 'lug', 'luo', 'lus', 'lvs', 'mag', 'mai', 'mal', 'mar', 'min', 'mkd', 'mlt', 'mni', 'mos', 'mri', 'mya', 'nld', 'nno', 'nob', 'npi', 'nqo', 'nso', 'nus', 'nya', 'oci', 'ory', 'pag', 'pan', 'pap', 'pbt', 'pes', 'plt', 'pol', 'por', 'prs', 'quy', 'ron', 'run', 'rus', 'sag', 'san', 'sat', 'scn', 'shn', 'sin', 'slk', 'slv', 'smo', 'sna', 'snd', 'som', 'sot', 'spa', 'srd', 'srp', 'ssw', 'sun', 'swe', 'swh', 'szl', 'tam', 'taq', 'tat', 'tel', 'tgk', 'tgl', 'tha', 'tir', 'tpi', 'tsn', 'tso', 'tuk', 'tum', 'tur', 'twi', 'tzm', 'uig', 'ukr', 'umb', 'urd', 'uzn', 'vec', 'vie', 'war', 'wol', 'xho', 'ydd', 'yor', 'yue', 'zho', 'zsm', 'zul'] | Clustering | s2s | [News, Written] | {'test': 1004} | {'test': 114.78} | -| [SICK-BR-PC](https://linux.ime.usp.br/~thalen/SICK_PT.pdf) | ['por'] | PairClassification | s2s | [Web, Written] | {'test': 1000} | {'test': 54.89} | -| [SICK-BR-STS](https://linux.ime.usp.br/~thalen/SICK_PT.pdf) | ['por'] | STS | s2s | [Web, Written] | {'test': 1000} | {'test': 54.89} | +| [Quail](https://text-machine.cs.uml.edu/lab2/projects/quail/) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | None | None | +| [Quora-PL](https://quoradata.quora.com/First-Quora-Dataset-Release-Question-Pairs) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2s | | None | None | +| [Quora-PLHardNegatives](https://quoradata.quora.com/First-Quora-Dataset-Release-Question-Pairs) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2s | | None | None | +| [QuoraRetrieval](https://quoradata.quora.com/First-Quora-Dataset-Release-Question-Pairs) (DataCanary et al., 2017) | ['eng'] | Retrieval | s2s | | None | None | +| [QuoraRetrievalHardNegatives](https://quoradata.quora.com/First-Quora-Dataset-Release-Question-Pairs) (DataCanary et al., 2017) | ['eng'] | Retrieval | s2s | | None | None | +| [RARbCode](https://arxiv.org/abs/2404.06347) (Xiao et al., 2024) | ['eng'] | Retrieval | s2p | [Programming, Written] | None | None | +| [RARbMath](https://arxiv.org/abs/2404.06347) (Xiao et al., 2024) | ['eng'] | Retrieval | s2p | [Encyclopaedic, Written] | None | None | +| [RTE3](https://aclanthology.org/W07-1401/) | ['deu', 'eng', 'fra', 'ita'] | PairClassification | s2s | [News, Web, Encyclopaedic, Written] | None | None | +| [RUParaPhraserSTS](https://aclanthology.org/2020.ngt-1.6) (Pivovarova et al., 2017) | ['rus'] | STS | s2s | [News, Written] | None | None | +| [RedditClustering.v2](https://arxiv.org/abs/2104.07081) (Gregor Geigle, 2021) | ['eng'] | Clustering | s2s | [Web, Social, Written] | None | None | +| [RedditClusteringP2P.v2](https://arxiv.org/abs/2104.07081) (Gregor Geigle, 2021) | ['eng'] | Clustering | p2p | [Web, Social, Written] | None | None | +| [RestaurantReviewSentimentClassification](https://link.springer.com/chapter/10.1007/978-3-319-18117-2_2) (ElSahar et al., 2015) | ['ara'] | Classification | s2s | [Reviews, Written] | None | None | +| [RiaNewsRetrieval](https://arxiv.org/abs/1901.07786) (Gavrilov et al., 2019) | ['rus'] | Retrieval | s2p | [News, Written] | None | None | +| [RiaNewsRetrievalHardNegatives](https://arxiv.org/abs/1901.07786) (Gavrilov et al., 2019) | ['rus'] | Retrieval | s2p | [News, Written] | None | None | +| [Robust04InstructionRetrieval](https://arxiv.org/abs/2403.15246) (Orion Weller, 2024) | ['eng'] | InstructionRetrieval | s2p | [News, Written] | None | None | +| [RomaTalesBitextMining](https://idoc.pub/documents/idocpub-zpnxm9g35ylv) | ['hun', 'rom'] | BitextMining | s2s | [Fiction, Written] | None | None | +| [RomaniBibleClustering](https://romani.global.bible/info) | ['rom'] | Clustering | p2p | [Religious, Written] | None | None | +| [RomanianReviewsSentiment](https://arxiv.org/abs/2101.04197) (Anca Maria Tache, 2021) | ['ron'] | Classification | s2s | [Reviews, Written] | None | None | +| [RomanianSentimentClassification](https://arxiv.org/abs/2009.08712) (Dumitrescu et al., 2020) | ['ron'] | Classification | s2s | [Reviews, Written] | None | None | +| [RonSTS](https://openreview.net/forum?id=JH61CD7afTv) (Dumitrescu et al., 2021) | ['ron'] | STS | s2s | [News, Social, Web, Written] | None | None | +| [RuBQReranking](https://openreview.net/pdf?id=P5UQFFoQ4PJ) (Ivan Rybin, 2021) | ['rus'] | Reranking | s2p | [Encyclopaedic, Written] | None | None | +| [RuBQRetrieval](https://openreview.net/pdf?id=P5UQFFoQ4PJ) (Ivan Rybin, 2021) | ['rus'] | Retrieval | s2p | [Encyclopaedic, Written] | None | None | +| [RuReviewsClassification](https://github.com/sismetanin/rureviews) (Sergey Smetanin, 2019) | ['rus'] | Classification | p2p | [Reviews, Written] | None | None | +| [RuSTSBenchmarkSTS](https://github.com/PhilipMay/stsb-multi-mt/) (Philip May, 2021) | ['rus'] | STS | s2s | [News, Social, Web, Written] | None | None | +| [RuSciBenchGRNTIClassification](https://github.com/mlsa-iai-msu-lab/ru_sci_bench/) | ['rus'] | Classification | p2p | [Academic, Written] | None | None | +| [RuSciBenchGRNTIClusteringP2P](https://github.com/mlsa-iai-msu-lab/ru_sci_bench/) | ['rus'] | Clustering | p2p | [Academic, Written] | {'test': 2048} | {'test': {'num_samples': 2048, 'number_of_characters': 1822339, 'average_text_length': 889.81, 'average_labels_per_text': 1.0, 'unique_labels': 28, 'labels': {'3': {'count': 73}, '4': {'count': 73}, '20': {'count': 73}, '9': {'count': 73}, '21': {'count': 73}, '15': {'count': 73}, '16': {'count': 74}, '2': {'count': 73}, '8': {'count': 73}, '23': {'count': 73}, '6': {'count': 73}, '24': {'count': 73}, '10': {'count': 73}, '1': {'count': 73}, '17': {'count': 74}, '14': {'count': 74}, '18': {'count': 73}, '27': {'count': 73}, '19': {'count': 73}, '22': {'count': 73}, '12': {'count': 73}, '25': {'count': 73}, '5': {'count': 74}, '0': {'count': 73}, '26': {'count': 73}, '11': {'count': 73}, '13': {'count': 73}, '7': {'count': 73}}}} | +| [RuSciBenchOECDClassification](https://github.com/mlsa-iai-msu-lab/ru_sci_bench/) | ['rus'] | Classification | p2p | [Academic, Written] | None | None | +| [RuSciBenchOECDClusteringP2P](https://github.com/mlsa-iai-msu-lab/ru_sci_bench/) | ['rus'] | Clustering | p2p | [Academic, Written] | None | None | +| [SCDBPAccountabilityLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [SCDBPAuditsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [SCDBPCertificationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [SCDBPTrainingLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [SCDBPVerificationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [SCDDAccountabilityLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [SCDDAuditsLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [SCDDCertificationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [SCDDTrainingLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [SCDDVerificationLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [SCIDOCS](https://allenai.org/data/scidocs) (Arman Cohan, 2020) | ['eng'] | Retrieval | s2p | [Academic, Written, Non-fiction] | None | None | +| [SCIDOCS-PL](https://allenai.org/data/scidocs) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | | None | None | +| [SIB200Classification](https://arxiv.org/abs/2309.07445) (Adelani et al., 2023) | ['ace', 'acm', 'acq', 'aeb', 'afr', 'ajp', 'aka', 'als', 'amh', 'apc', 'arb', 'ars', 'ary', 'arz', 'asm', 'ast', 'awa', 'ayr', 'azb', 'azj', 'bak', 'bam', 'ban', 'bel', 'bem', 'ben', 'bho', 'bjn', 'bod', 'bos', 'bug', 'bul', 'cat', 'ceb', 'ces', 'cjk', 'ckb', 'crh', 'cym', 'dan', 'deu', 'dik', 'dyu', 'dzo', 'ell', 'eng', 'epo', 'est', 'eus', 'ewe', 'fao', 'fij', 'fin', 'fon', 'fra', 'fur', 'fuv', 'gaz', 'gla', 'gle', 'glg', 'grn', 'guj', 'hat', 'hau', 'heb', 'hin', 'hne', 'hrv', 'hun', 'hye', 'ibo', 'ilo', 'ind', 'isl', 'ita', 'jav', 'jpn', 'kab', 'kac', 'kam', 'kan', 'kas', 'kat', 'kaz', 'kbp', 'kea', 'khk', 'khm', 'kik', 'kin', 'kir', 'kmb', 'kmr', 'knc', 'kon', 'kor', 'lao', 'lij', 'lim', 'lin', 'lit', 'lmo', 'ltg', 'ltz', 'lua', 'lug', 'luo', 'lus', 'lvs', 'mag', 'mai', 'mal', 'mar', 'min', 'mkd', 'mlt', 'mni', 'mos', 'mri', 'mya', 'nld', 'nno', 'nob', 'npi', 'nqo', 'nso', 'nus', 'nya', 'oci', 'ory', 'pag', 'pan', 'pap', 'pbt', 'pes', 'plt', 'pol', 'por', 'prs', 'quy', 'ron', 'run', 'rus', 'sag', 'san', 'sat', 'scn', 'shn', 'sin', 'slk', 'slv', 'smo', 'sna', 'snd', 'som', 'sot', 'spa', 'srd', 'srp', 'ssw', 'sun', 'swe', 'swh', 'szl', 'tam', 'taq', 'tat', 'tel', 'tgk', 'tgl', 'tha', 'tir', 'tpi', 'tsn', 'tso', 'tuk', 'tum', 'tur', 'twi', 'tzm', 'uig', 'ukr', 'umb', 'urd', 'uzn', 'vec', 'vie', 'war', 'wol', 'xho', 'ydd', 'yor', 'yue', 'zho', 'zsm', 'zul'] | Classification | s2s | [News, Written] | None | None | +| [SIB200ClusteringS2S](https://arxiv.org/abs/2309.07445) (Adelani et al., 2023) | ['ace', 'acm', 'acq', 'aeb', 'afr', 'ajp', 'aka', 'als', 'amh', 'apc', 'arb', 'ars', 'ary', 'arz', 'asm', 'ast', 'awa', 'ayr', 'azb', 'azj', 'bak', 'bam', 'ban', 'bel', 'bem', 'ben', 'bho', 'bjn', 'bod', 'bos', 'bug', 'bul', 'cat', 'ceb', 'ces', 'cjk', 'ckb', 'crh', 'cym', 'dan', 'deu', 'dik', 'dyu', 'dzo', 'ell', 'eng', 'epo', 'est', 'eus', 'ewe', 'fao', 'fij', 'fin', 'fon', 'fra', 'fur', 'fuv', 'gaz', 'gla', 'gle', 'glg', 'grn', 'guj', 'hat', 'hau', 'heb', 'hin', 'hne', 'hrv', 'hun', 'hye', 'ibo', 'ilo', 'ind', 'isl', 'ita', 'jav', 'jpn', 'kab', 'kac', 'kam', 'kan', 'kas', 'kat', 'kaz', 'kbp', 'kea', 'khk', 'khm', 'kik', 'kin', 'kir', 'kmb', 'kmr', 'knc', 'kon', 'kor', 'lao', 'lij', 'lim', 'lin', 'lit', 'lmo', 'ltg', 'ltz', 'lua', 'lug', 'luo', 'lus', 'lvs', 'mag', 'mai', 'mal', 'mar', 'min', 'mkd', 'mlt', 'mni', 'mos', 'mri', 'mya', 'nld', 'nno', 'nob', 'npi', 'nqo', 'nso', 'nus', 'nya', 'oci', 'ory', 'pag', 'pan', 'pap', 'pbt', 'pes', 'plt', 'pol', 'por', 'prs', 'quy', 'ron', 'run', 'rus', 'sag', 'san', 'sat', 'scn', 'shn', 'sin', 'slk', 'slv', 'smo', 'sna', 'snd', 'som', 'sot', 'spa', 'srd', 'srp', 'ssw', 'sun', 'swe', 'swh', 'szl', 'tam', 'taq', 'tat', 'tel', 'tgk', 'tgl', 'tha', 'tir', 'tpi', 'tsn', 'tso', 'tuk', 'tum', 'tur', 'twi', 'tzm', 'uig', 'ukr', 'umb', 'urd', 'uzn', 'vec', 'vie', 'war', 'wol', 'xho', 'ydd', 'yor', 'yue', 'zho', 'zsm', 'zul'] | Clustering | s2s | [News, Written] | None | None | +| [SICK-BR-PC](https://linux.ime.usp.br/~thalen/SICK_PT.pdf) | ['por'] | PairClassification | s2s | [Web, Written] | None | None | +| [SICK-BR-STS](https://linux.ime.usp.br/~thalen/SICK_PT.pdf) | ['por'] | STS | s2s | [Web, Written] | None | None | | [SICK-E-PL](https://aclanthology.org/2020.lrec-1.207) | ['pol'] | PairClassification | s2s | | None | None | | [SICK-R](https://aclanthology.org/2020.lrec-1.207) | ['eng'] | STS | s2s | | None | None | -| [SICK-R-PL](https://aclanthology.org/2020.lrec-1.207) | ['pol'] | STS | s2s | [Web, Written] | {'test': 9812} | {'test': 42.8} | +| [SICK-R-PL](https://aclanthology.org/2020.lrec-1.207) | ['pol'] | STS | s2s | [Web, Written] | None | None | | [SICKFr](https://huggingface.co/datasets/Lajavaness/SICK-fr) | ['fra'] | STS | s2s | | None | None | -| [SIQA](https://leaderboard.allenai.org/socialiqa/submissions/get-started) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | {'test': 0} | {'test': {'average_document_length': 22.967085695044617, 'average_query_length': 127.75383828045035, 'num_documents': 71276, 'num_queries': 1954, 'average_relevant_docs_per_query': 1.0}} | -| [SKQuadRetrieval](https://huggingface.co/datasets/TUKE-KEMT/retrieval-skquad) | ['slk'] | Retrieval | s2s | [Encyclopaedic] | {'test': 1134} | {'test': {'average_document_length': 1180.5071792496526, 'average_query_length': 53.63403880070547, 'num_documents': 6477, 'num_queries': 1134, 'average_relevant_docs_per_query': 11}} | -| [SNLHierarchicalClusteringP2P](https://huggingface.co/datasets/navjordj/SNL_summarization) (Navjord et al., 2023) | ['nob'] | Clustering | p2p | [Encyclopaedic, Non-fiction, Written] | {'test': 1300} | {'test': 1986.9453846153847} | -| [SNLHierarchicalClusteringS2S](https://huggingface.co/datasets/navjordj/SNL_summarization) (Navjord et al., 2023) | ['nob'] | Clustering | s2s | [Encyclopaedic, Non-fiction, Written] | {'test': 1300} | {'test': 242.22384615384615} | -| [SNLRetrieval](https://huggingface.co/datasets/navjordj/SNL_summarization) (Navjord et al., 2023) | ['nob'] | Retrieval | p2p | [Encyclopaedic, Non-fiction, Written] | {'test': 2048} | {'test': {'average_document_length': 1986.9453846153847, 'average_query_length': 14.906153846153845, 'num_documents': 1300, 'num_queries': 1300, 'average_relevant_docs_per_query': 1.0}} | -| [SRNCorpusBitextMining](https://arxiv.org/abs/2212.06383) (Zwennicker et al., 2022) | ['nld', 'srn'] | BitextMining | s2s | [Social, Web, Written] | {'test': 256} | {'test': 55} | -| [STS12](https://www.aclweb.org/anthology/S12-1051.pdf) (Agirre et al., 2012) | ['eng'] | STS | s2s | [Encyclopaedic, News, Written] | {'test': 6216} | {'test': {'num_samples': 3108, 'average_sentence1_len': 63.78893178893179, 'average_sentence2_len': 65.5926640926641, 'avg_score': 3.5060643500643507}} | -| [STS13](https://www.aclweb.org/anthology/S13-1004/) (Eneko Agirre, 2013) | ['eng'] | STS | s2s | [Web, News, Non-fiction, Written] | {'test': 3000} | {'test': 54.0} | -| [STS14](https://www.aclweb.org/anthology/S14-1002) | ['eng'] | STS | s2s | [Blog, Web, Spoken] | {'test': 7500} | {'test': 54.3} | -| [STS15](https://www.aclweb.org/anthology/S15-2010) | ['eng'] | STS | s2s | [Blog, News, Web, Written, Spoken] | {'test': 6000} | {'test': 57.7} | -| [STS16](https://www.aclweb.org/anthology/S16-1001) | ['eng'] | STS | s2s | [Blog, Web, Spoken] | {'test': 2372} | {'test': 65.3} | -| [STS17](https://alt.qcri.org/semeval2017/task1/) | ['ara', 'deu', 'eng', 'fra', 'ita', 'kor', 'nld', 'spa', 'tur'] | STS | s2s | [News, Web, Written] | {'test': 500} | {'test': {'num_samples': 5346, 'average_sentence1_len': 38.14665170220726, 'average_sentence2_len': 36.72502805836139, 'avg_score': 2.3554804214989464, 'hf_subset_descriptive_stats': {'ko-ko': {'num_samples': 2846, 'average_sentence1_len': 31.991918482080113, 'average_sentence2_len': 32.44483485593816, 'avg_score': 2.469359920356055}, 'ar-ar': {'num_samples': 250, 'average_sentence1_len': 32.208, 'average_sentence2_len': 32.78, 'avg_score': 2.216800000000001}, 'en-ar': {'num_samples': 250, 'average_sentence1_len': 42.36, 'average_sentence2_len': 32.696, 'avg_score': 2.1423999999999994}, 'en-de': {'num_samples': 250, 'average_sentence1_len': 43.952, 'average_sentence2_len': 44.756, 'avg_score': 2.2776000000000014}, 'en-en': {'num_samples': 250, 'average_sentence1_len': 43.952, 'average_sentence2_len': 42.724, 'avg_score': 2.2776000000000014}, 'en-tr': {'num_samples': 250, 'average_sentence1_len': 41.916, 'average_sentence2_len': 41.6, 'avg_score': 2.1335999999999986}, 'es-en': {'num_samples': 250, 'average_sentence1_len': 50.84, 'average_sentence2_len': 42.024, 'avg_score': 2.1464000000000003}, 'es-es': {'num_samples': 250, 'average_sentence1_len': 49.836, 'average_sentence2_len': 51.224, 'avg_score': 2.2312000000000007}, 'fr-en': {'num_samples': 250, 'average_sentence1_len': 49.624, 'average_sentence2_len': 42.724, 'avg_score': 2.2776000000000014}, 'it-en': {'num_samples': 250, 'average_sentence1_len': 50.028, 'average_sentence2_len': 42.724, 'avg_score': 2.2776000000000014}, 'nl-en': {'num_samples': 250, 'average_sentence1_len': 46.816, 'average_sentence2_len': 42.724, 'avg_score': 2.2776000000000014}}}} | -| [STS22.v2](https://competitions.codalab.org/competitions/33835) | ['ara', 'cmn', 'deu', 'eng', 'fra', 'ita', 'pol', 'rus', 'spa', 'tur'] | STS | p2p | [News, Written] | {'test': 3958} | {'test': 1993.6} | +| [SIQA](https://leaderboard.allenai.org/socialiqa/submissions/get-started) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | None | None | +| [SKQuadRetrieval](https://huggingface.co/datasets/TUKE-KEMT/retrieval-skquad) | ['slk'] | Retrieval | s2s | [Encyclopaedic] | None | None | +| [SNLHierarchicalClusteringP2P](https://huggingface.co/datasets/navjordj/SNL_summarization) (Navjord et al., 2023) | ['nob'] | Clustering | p2p | [Encyclopaedic, Non-fiction, Written] | None | None | +| [SNLHierarchicalClusteringS2S](https://huggingface.co/datasets/navjordj/SNL_summarization) (Navjord et al., 2023) | ['nob'] | Clustering | s2s | [Encyclopaedic, Non-fiction, Written] | None | None | +| [SNLRetrieval](https://huggingface.co/datasets/navjordj/SNL_summarization) (Navjord et al., 2023) | ['nob'] | Retrieval | p2p | [Encyclopaedic, Non-fiction, Written] | None | None | +| [SRNCorpusBitextMining](https://arxiv.org/abs/2212.06383) (Zwennicker et al., 2022) | ['nld', 'srn'] | BitextMining | s2s | [Social, Web, Written] | None | None | +| [STS12](https://www.aclweb.org/anthology/S12-1051.pdf) (Agirre et al., 2012) | ['eng'] | STS | s2s | [Encyclopaedic, News, Written] | {'test': 3108} | {'test': {'num_samples': 3108, 'number_of_characters': 402118, 'average_sentence1_len': 63.79, 'average_sentence2_len': 65.59, 'avg_score': 3.51}} | +| [STS13](https://www.aclweb.org/anthology/S13-1004/) (Eneko Agirre, 2013) | ['eng'] | STS | s2s | [Web, News, Non-fiction, Written] | None | None | +| [STS14](https://www.aclweb.org/anthology/S14-1002) | ['eng'] | STS | s2s | [Blog, Web, Spoken] | None | None | +| [STS15](https://www.aclweb.org/anthology/S15-2010) | ['eng'] | STS | s2s | [Blog, News, Web, Written, Spoken] | None | None | +| [STS16](https://www.aclweb.org/anthology/S16-1001) | ['eng'] | STS | s2s | [Blog, Web, Spoken] | None | None | +| [STS17](https://alt.qcri.org/semeval2017/task1/) | ['ara', 'deu', 'eng', 'fra', 'ita', 'kor', 'nld', 'spa', 'tur'] | STS | s2s | [News, Web, Written] | {'test': 5346} | {'test': {'num_samples': 5346, 'number_of_characters': 400264, 'average_sentence1_len': 38.15, 'average_sentence2_len': 36.73, 'avg_score': 2.36, 'hf_subset_descriptive_stats': {'ko-ko': {'num_samples': 2846, 'number_of_characters': 183387, 'average_sentence1_len': 31.99, 'average_sentence2_len': 32.44, 'avg_score': 2.47}, 'ar-ar': {'num_samples': 250, 'number_of_characters': 16247, 'average_sentence1_len': 32.21, 'average_sentence2_len': 32.78, 'avg_score': 2.22}, 'en-ar': {'num_samples': 250, 'number_of_characters': 18764, 'average_sentence1_len': 42.36, 'average_sentence2_len': 32.7, 'avg_score': 2.14}, 'en-de': {'num_samples': 250, 'number_of_characters': 22177, 'average_sentence1_len': 43.95, 'average_sentence2_len': 44.76, 'avg_score': 2.28}, 'en-en': {'num_samples': 250, 'number_of_characters': 21669, 'average_sentence1_len': 43.95, 'average_sentence2_len': 42.72, 'avg_score': 2.28}, 'en-tr': {'num_samples': 250, 'number_of_characters': 20879, 'average_sentence1_len': 41.92, 'average_sentence2_len': 41.6, 'avg_score': 2.13}, 'es-en': {'num_samples': 250, 'number_of_characters': 23216, 'average_sentence1_len': 50.84, 'average_sentence2_len': 42.02, 'avg_score': 2.15}, 'es-es': {'num_samples': 250, 'number_of_characters': 25265, 'average_sentence1_len': 49.84, 'average_sentence2_len': 51.22, 'avg_score': 2.23}, 'fr-en': {'num_samples': 250, 'number_of_characters': 23087, 'average_sentence1_len': 49.62, 'average_sentence2_len': 42.72, 'avg_score': 2.28}, 'it-en': {'num_samples': 250, 'number_of_characters': 23188, 'average_sentence1_len': 50.03, 'average_sentence2_len': 42.72, 'avg_score': 2.28}, 'nl-en': {'num_samples': 250, 'number_of_characters': 22385, 'average_sentence1_len': 46.82, 'average_sentence2_len': 42.72, 'avg_score': 2.28}}}} | +| [STS22.v2](https://competitions.codalab.org/competitions/33835) | ['ara', 'cmn', 'deu', 'eng', 'fra', 'ita', 'pol', 'rus', 'spa', 'tur'] | STS | p2p | [News, Written] | None | None | | [STSB](https://aclanthology.org/2021.emnlp-main.357) (Shitao Xiao, 2024) | ['cmn'] | STS | s2s | | None | None | | [STSBenchmark](https://github.com/PhilipMay/stsb-multi-mt/) (Philip May, 2021) | ['eng'] | STS | s2s | | None | None | -| [STSBenchmarkMultilingualSTS](https://github.com/PhilipMay/stsb-multi-mt/) (Philip May, 2021) | ['cmn', 'deu', 'eng', 'fra', 'ita', 'nld', 'pol', 'por', 'rus', 'spa'] | STS | s2s | [News, Social, Web, Spoken, Written] | {'dev': 30000, 'test': 27580} | {'dev': 66.5, 'test': 56.1} | +| [STSBenchmarkMultilingualSTS](https://github.com/PhilipMay/stsb-multi-mt/) (Philip May, 2021) | ['cmn', 'deu', 'eng', 'fra', 'ita', 'nld', 'pol', 'por', 'rus', 'spa'] | STS | s2s | [News, Social, Web, Spoken, Written] | None | None | | [STSES](https://huggingface.co/datasets/PlanTL-GOB-ES/sts-es) (Agirre et al., 2015) | ['spa'] | STS | s2s | [Written] | None | None | -| [SadeemQuestionRetrieval](https://huggingface.co/datasets/sadeem-ai/sadeem-ar-eval-retrieval-questions) | ['ara'] | Retrieval | s2p | [Written, Written] | {'test': 22979} | {'test': 500.0} | -| [SanskritShlokasClassification](https://github.com/goru001/nlp-for-sanskrit) | ['san'] | Classification | s2s | [Religious, Written] | {'train': 383, 'validation': 96} | {'train': 98.415, 'validation': 96.635} | -| [ScalaClassification](https://aclanthology.org/2023.nodalida-1.20/) | ['dan', 'nno', 'nob', 'swe'] | Classification | s2s | [Fiction, News, Non-fiction, Blog, Spoken, Web, Written] | {'test': 4096} | {'test': 102.72} | -| [SciDocsRR](https://allenai.org/data/scidocs) | ['eng'] | Reranking | s2s | [Academic, Non-fiction, Written] | {'test': 19599} | {'test': 69.0} | -| [SciFact](https://github.com/allenai/scifact) (Arman Cohan, 2020) | ['eng'] | Retrieval | s2p | | None | {'train': {'average_document_length': 1498.4152035500674, 'average_query_length': 88.58838071693448, 'num_documents': 5183, 'num_queries': 809, 'average_relevant_docs_per_query': 1.1359703337453646}, 'test': {'average_document_length': 1498.4152035500674, 'average_query_length': 90.34666666666666, 'num_documents': 5183, 'num_queries': 300, 'average_relevant_docs_per_query': 1.13}} | -| [SciFact-PL](https://github.com/allenai/scifact) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | | None | {'test': {'average_document_length': 1553.5178468068686, 'average_query_length': 95.44, 'num_documents': 5183, 'num_queries': 300, 'average_relevant_docs_per_query': 1.13}} | -| [SemRel24STS](https://huggingface.co/datasets/SemRel/SemRel2024) (Nedjma Ousidhoum, 2024) | ['afr', 'amh', 'arb', 'arq', 'ary', 'eng', 'hau', 'hin', 'ind', 'kin', 'mar', 'tel'] | STS | s2s | [Spoken, Written] | {'dev': 2089, 'test': 7498} | {'dev': 163.1, 'test': 145.9} | -| [SensitiveTopicsClassification](https://aclanthology.org/2021.bsnlp-1.4) | ['rus'] | MultilabelClassification | s2s | [Web, Social, Written] | {'test': 2048} | {'test': 95.3} | -| [SentimentAnalysisHindi](https://huggingface.co/datasets/OdiaGenAI/sentiment_analysis_hindi) (Shantipriya Parida, 2023) | ['hin'] | Classification | s2s | [Reviews, Written] | {'train': 2497} | {'train': 81.29} | -| [SinhalaNewsClassification](https://huggingface.co/datasets/NLPC-UOM/Sinhala-News-Category-classification) (Nisansa de Silva, 2015) | ['sin'] | Classification | s2s | [News, Written] | {'train': 3327} | {'train': 148.04} | -| [SinhalaNewsSourceClassification](https://huggingface.co/datasets/NLPC-UOM/Sinhala-News-Source-classification) (Dhananjaya et al., 2022) | ['sin'] | Classification | s2s | [News, Written] | {'train': 24094} | {'train': 56.08} | -| [SiswatiNewsClassification](https://huggingface.co/datasets/dsfsi/za-isizulu-siswati-news) (Madodonga et al., 2023) | ['ssw'] | Classification | s2s | [News, Written] | {'train': 80} | {'train': 354.2} | -| [SlovakHateSpeechClassification](https://huggingface.co/datasets/TUKE-KEMT/hate_speech_slovak) | ['slk'] | Classification | s2s | [Social, Written] | {'test': 1319} | {'test': 92.71} | -| [SlovakMovieReviewSentimentClassification](https://arxiv.org/pdf/2304.01922) ({ {S, 2023) | ['svk'] | Classification | s2s | [Reviews, Written] | {'test': 2048} | {'test': 366.17} | -| [SlovakSumRetrieval](https://huggingface.co/datasets/NaiveNeuron/slovaksum) | ['slk'] | Retrieval | s2s | [News, Social, Web, Written] | {'test': 600} | {'test': {'average_document_length': 2156.445, 'average_query_length': 143.59833333333333, 'num_documents': 600, 'num_queries': 600, 'average_relevant_docs_per_query': 1.0}} | -| [SouthAfricanLangClassification](https://www.kaggle.com/competitions/south-african-language-identification/) (ExploreAI Academy et al., 2022) | ['afr', 'eng', 'nbl', 'nso', 'sot', 'ssw', 'tsn', 'tso', 'ven', 'xho', 'zul'] | Classification | s2s | [Web, Non-fiction, Written] | {'test': 2048} | {'test': 247.49} | -| [SpanishNewsClassification](https://huggingface.co/datasets/MarcOrfilaCarreras/spanish-news) | ['spa'] | Classification | s2s | [News, Written] | {'train': 2048} | {'train': 4218.2} | +| [SadeemQuestionRetrieval](https://huggingface.co/datasets/sadeem-ai/sadeem-ar-eval-retrieval-questions) | ['ara'] | Retrieval | s2p | [Written, Written] | None | None | +| [SanskritShlokasClassification](https://github.com/goru001/nlp-for-sanskrit) | ['san'] | Classification | s2s | [Religious, Written] | None | None | +| [ScalaClassification](https://aclanthology.org/2023.nodalida-1.20/) | ['dan', 'nno', 'nob', 'swe'] | Classification | s2s | [Fiction, News, Non-fiction, Blog, Spoken, Web, Written] | None | None | +| [SciDocsRR](https://allenai.org/data/scidocs) | ['eng'] | Reranking | s2s | [Academic, Non-fiction, Written] | None | None | +| [SciFact](https://github.com/allenai/scifact) (Arman Cohan, 2020) | ['eng'] | Retrieval | s2p | | None | None | +| [SciFact-PL](https://github.com/allenai/scifact) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | | None | None | +| [SemRel24STS](https://huggingface.co/datasets/SemRel/SemRel2024) (Nedjma Ousidhoum, 2024) | ['afr', 'amh', 'arb', 'arq', 'ary', 'eng', 'hau', 'hin', 'ind', 'kin', 'mar', 'tel'] | STS | s2s | [Spoken, Written] | None | None | +| [SensitiveTopicsClassification](https://aclanthology.org/2021.bsnlp-1.4) | ['rus'] | MultilabelClassification | s2s | [Web, Social, Written] | None | None | +| [SentimentAnalysisHindi](https://huggingface.co/datasets/OdiaGenAI/sentiment_analysis_hindi) (Shantipriya Parida, 2023) | ['hin'] | Classification | s2s | [Reviews, Written] | None | None | +| [SinhalaNewsClassification](https://huggingface.co/datasets/NLPC-UOM/Sinhala-News-Category-classification) (Nisansa de Silva, 2015) | ['sin'] | Classification | s2s | [News, Written] | None | None | +| [SinhalaNewsSourceClassification](https://huggingface.co/datasets/NLPC-UOM/Sinhala-News-Source-classification) (Dhananjaya et al., 2022) | ['sin'] | Classification | s2s | [News, Written] | None | None | +| [SiswatiNewsClassification](https://huggingface.co/datasets/dsfsi/za-isizulu-siswati-news) (Madodonga et al., 2023) | ['ssw'] | Classification | s2s | [News, Written] | None | None | +| [SlovakHateSpeechClassification](https://huggingface.co/datasets/TUKE-KEMT/hate_speech_slovak) | ['slk'] | Classification | s2s | [Social, Written] | {'test': 1319} | {'test': {'num_samples': 1319, 'number_of_characters': 122279, 'average_text_length': 92.71, 'unique_labels': 2, 'labels': {'1': {'count': 360}, '0': {'count': 959}}}} | +| [SlovakMovieReviewSentimentClassification](https://arxiv.org/pdf/2304.01922) ({ {S, 2023) | ['svk'] | Classification | s2s | [Reviews, Written] | None | None | +| [SlovakSumRetrieval](https://huggingface.co/datasets/NaiveNeuron/slovaksum) | ['slk'] | Retrieval | s2s | [News, Social, Web, Written] | None | None | +| [SouthAfricanLangClassification](https://www.kaggle.com/competitions/south-african-language-identification/) (ExploreAI Academy et al., 2022) | ['afr', 'eng', 'nbl', 'nso', 'sot', 'ssw', 'tsn', 'tso', 'ven', 'xho', 'zul'] | Classification | s2s | [Web, Non-fiction, Written] | None | None | +| [SpanishNewsClassification](https://huggingface.co/datasets/MarcOrfilaCarreras/spanish-news) | ['spa'] | Classification | s2s | [News, Written] | None | None | | [SpanishNewsClusteringP2P](https://www.kaggle.com/datasets/kevinmorgado/spanish-news-classification) | ['spa'] | Clustering | p2p | | None | None | -| [SpanishPassageRetrievalS2P](https://mklab.iti.gr/results/spanish-passage-retrieval-dataset/) | ['spa'] | Retrieval | s2p | | None | {'test': {'average_document_length': 2635.217893792966, 'average_query_length': 67.55688622754491, 'num_documents': 10037, 'num_queries': 167, 'average_relevant_docs_per_query': 6.053892215568863}} | -| [SpanishPassageRetrievalS2S](https://mklab.iti.gr/results/spanish-passage-retrieval-dataset/) | ['spa'] | Retrieval | s2s | | None | {'test': {'average_document_length': 434.5924528301887, 'average_query_length': 67.55688622754491, 'num_documents': 265, 'num_queries': 167, 'average_relevant_docs_per_query': 7.718562874251497}} | -| [SpanishSentimentClassification](https://huggingface.co/datasets/sepidmnorozy/Spanish_sentiment) | ['spa'] | Classification | s2s | [Reviews, Written] | {'validation': 147, 'test': 296} | {'validation': 85.02, 'test': 87.91} | -| [SpartQA](https://github.com/HLR/SpartQA_generation) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | {'test': 0} | {'test': {'average_document_length': 50.40829145728643, 'average_query_length': 656.2328881469115, 'num_documents': 1592, 'num_queries': 3594, 'average_relevant_docs_per_query': 1.8786867000556482}} | -| [SprintDuplicateQuestions](https://www.aclweb.org/anthology/D18-1131/) | ['eng'] | PairClassification | s2s | [Programming, Written] | {'validation': 101000, 'test': 101000} | {'validation': 65.2, 'test': 67.9} | -| [StackExchangeClustering.v2](https://arxiv.org/abs/2104.07081) (Gregor Geigle, 2021) | ['eng'] | Clustering | s2s | [Web, Written] | {'test': 32768} | {'test': 57.0} | -| [StackExchangeClusteringP2P.v2](https://arxiv.org/abs/2104.07081) (Gregor Geigle, 2021) | ['eng'] | Clustering | p2p | [Web, Written] | {'test': 2996} | {'test': 1090.7} | -| [StackOverflowDupQuestions](https://www.microsoft.com/en-us/research/uploads/prod/2019/03/nl4se18LinkSO.pdf) (Xueqing Liu, 2018) | ['eng'] | Reranking | s2s | | {'test': 3467} | {'test': 49.8} | -| [StackOverflowQA](https://arxiv.org/abs/2407.02883) (Xiangyang Li, 2024) | ['eng'] | Retrieval | p2p | [Programming, Written] | {'test': 1000} | {'test': {'average_document_length': 1202.4815613867845, 'average_query_length': 1302.6263791374122, 'num_documents': 19931, 'num_queries': 1994, 'average_relevant_docs_per_query': 1.0}} | -| [StatcanDialogueDatasetRetrieval](https://mcgill-nlp.github.io/statcan-dialogue-dataset/) | ['eng', 'fra'] | Retrieval | s2p | [Government, Web, Written] | {'dev': 1000, 'test': 1011, 'corpus': 5907} | {'dev': {'english': {'average_document_length': 6535.865413915693, 'average_query_length': 6.869244935543278, 'num_documents': 5907, 'num_queries': 543, 'average_relevant_docs_per_query': 1.4714548802946592}, 'french': {'average_document_length': 7078.072794988996, 'average_query_length': 6.860655737704918, 'num_documents': 5907, 'num_queries': 122, 'average_relevant_docs_per_query': 1.6475409836065573}}, 'test': {'english': {'average_document_length': 6535.865413915693, 'average_query_length': 7.650994575045208, 'num_documents': 5907, 'num_queries': 553, 'average_relevant_docs_per_query': 1.573236889692586}, 'french': {'average_document_length': 7078.072794988996, 'average_query_length': 5.907407407407407, 'num_documents': 5907, 'num_queries': 108, 'average_relevant_docs_per_query': 1.3055555555555556}}} | -| [SummEvalFrSummarization.v2](https://github.com/Yale-LILY/SummEval) (Fabbri et al., 2020) | ['fra'] | Summarization | p2p | [News, Written] | {'test': 2800} | {'test': 407.1} | -| [SummEvalSummarization.v2](https://github.com/Yale-LILY/SummEval) (Fabbri et al., 2020) | ['eng'] | Summarization | p2p | [News, Written] | {'test': 2800} | {'test': 359.8} | -| [SwahiliNewsClassification](https://huggingface.co/datasets/Mollel/SwahiliNewsClassification) | ['swa'] | Classification | s2s | [News, Written] | {'train': 2048} | {'train': 2438.2308135942326} | -| [SweFaqRetrieval](https://spraakbanken.gu.se/en/resources/superlim) (Berdi{ {c, 2023) | ['swe'] | Retrieval | s2s | [Government, Non-fiction, Written] | {'test': 1024} | {'test': {'average_document_length': 319.8473581213307, 'average_query_length': 70.51461988304094, 'num_documents': 511, 'num_queries': 513, 'average_relevant_docs_per_query': 1.0}} | -| [SweRecClassification](https://aclanthology.org/2023.nodalida-1.20/) | ['swe'] | Classification | s2s | [Reviews, Written] | {'test': 1024} | {'test': 318.8} | -| [SwedishSentimentClassification](https://huggingface.co/datasets/swedish_reviews) | ['swe'] | Classification | s2s | [Reviews, Written] | {'validation': 1024, 'test': 1024} | {'validation': 499.3, 'test': 498.1} | -| [SwednClusteringP2P](https://spraakbanken.gu.se/en/resources/swedn) (Monsen et al., 2021) | ['swe'] | Clustering | p2p | [News, Non-fiction, Written] | {'all': 2048} | {'all': 1619.71} | -| [SwednClusteringS2S](https://spraakbanken.gu.se/en/resources/swedn) (Monsen et al., 2021) | ['swe'] | Clustering | s2s | [News, Non-fiction, Written] | {'all': 2048} | {'all': 1619.71} | -| [SwednRetrieval](https://spraakbanken.gu.se/en/resources/swedn) (Monsen et al., 2021) | ['swe'] | Retrieval | p2p | [News, Non-fiction, Written] | {'test': 2048} | {'test': {'average_document_length': 2896.519550342131, 'average_query_length': 45.876953125, 'num_documents': 2046, 'num_queries': 1024, 'average_relevant_docs_per_query': 2.0}} | -| [SwissJudgementClassification](https://aclanthology.org/2021.nllp-1.3/) (Joel Niklaus, 2022) | ['deu', 'fra', 'ita'] | Classification | s2s | [Legal, Written] | {'test': 2048} | {'test': 3411.72} | +| [SpanishPassageRetrievalS2P](https://mklab.iti.gr/results/spanish-passage-retrieval-dataset/) | ['spa'] | Retrieval | s2p | | None | None | +| [SpanishPassageRetrievalS2S](https://mklab.iti.gr/results/spanish-passage-retrieval-dataset/) | ['spa'] | Retrieval | s2s | | None | None | +| [SpanishSentimentClassification](https://huggingface.co/datasets/sepidmnorozy/Spanish_sentiment) | ['spa'] | Classification | s2s | [Reviews, Written] | None | None | +| [SpartQA](https://github.com/HLR/SpartQA_generation) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | None | None | +| [SprintDuplicateQuestions](https://www.aclweb.org/anthology/D18-1131/) | ['eng'] | PairClassification | s2s | [Programming, Written] | None | None | +| [StackExchangeClustering.v2](https://arxiv.org/abs/2104.07081) (Gregor Geigle, 2021) | ['eng'] | Clustering | s2s | [Web, Written] | None | None | +| [StackExchangeClusteringP2P.v2](https://arxiv.org/abs/2104.07081) (Gregor Geigle, 2021) | ['eng'] | Clustering | p2p | [Web, Written] | None | None | +| [StackOverflowDupQuestions](https://www.microsoft.com/en-us/research/uploads/prod/2019/03/nl4se18LinkSO.pdf) (Xueqing Liu, 2018) | ['eng'] | Reranking | s2s | | None | None | +| [StackOverflowQA](https://arxiv.org/abs/2407.02883) (Xiangyang Li, 2024) | ['eng'] | Retrieval | p2p | [Programming, Written] | {'test': 21925} | {'test': {'number_of_characters': 2506.11, 'num_samples': 21925, 'num_queries': 1994, 'num_documents': 19931, 'average_document_length': 0.06, 'average_query_length': 0.65, 'average_relevant_docs_per_query': 1.0}} | +| [StatcanDialogueDatasetRetrieval](https://mcgill-nlp.github.io/statcan-dialogue-dataset/) | ['eng', 'fra'] | Retrieval | s2p | [Government, Web, Written] | None | None | +| [SummEvalFrSummarization.v2](https://github.com/Yale-LILY/SummEval) (Fabbri et al., 2020) | ['fra'] | Summarization | p2p | [News, Written] | None | None | +| [SummEvalSummarization.v2](https://github.com/Yale-LILY/SummEval) (Fabbri et al., 2020) | ['eng'] | Summarization | p2p | [News, Written] | None | None | +| [SwahiliNewsClassification](https://huggingface.co/datasets/Mollel/SwahiliNewsClassification) | ['swa'] | Classification | s2s | [News, Written] | None | None | +| [SweFaqRetrieval](https://spraakbanken.gu.se/en/resources/superlim) (Berdi{ {c, 2023) | ['swe'] | Retrieval | s2s | [Government, Non-fiction, Written] | None | None | +| [SweRecClassification](https://aclanthology.org/2023.nodalida-1.20/) | ['swe'] | Classification | s2s | [Reviews, Written] | None | None | +| [SwedishSentimentClassification](https://huggingface.co/datasets/swedish_reviews) | ['swe'] | Classification | s2s | [Reviews, Written] | None | None | +| [SwednClusteringP2P](https://spraakbanken.gu.se/en/resources/swedn) (Monsen et al., 2021) | ['swe'] | Clustering | p2p | [News, Non-fiction, Written] | None | None | +| [SwednClusteringS2S](https://spraakbanken.gu.se/en/resources/swedn) (Monsen et al., 2021) | ['swe'] | Clustering | s2s | [News, Non-fiction, Written] | None | None | +| [SwednRetrieval](https://spraakbanken.gu.se/en/resources/swedn) (Monsen et al., 2021) | ['swe'] | Retrieval | p2p | [News, Non-fiction, Written] | None | None | +| [SwissJudgementClassification](https://aclanthology.org/2021.nllp-1.3/) (Joel Niklaus, 2022) | ['deu', 'fra', 'ita'] | Classification | s2s | [Legal, Written] | None | None | | [SyntecReranking](https://huggingface.co/datasets/lyon-nlp/mteb-fr-reranking-syntec-s2p) (Mathieu Ciancone, 2024) | ['fra'] | Reranking | s2p | [Legal, Written] | None | None | -| [SyntecRetrieval](https://huggingface.co/datasets/lyon-nlp/mteb-fr-retrieval-syntec-s2p) (Mathieu Ciancone, 2024) | ['fra'] | Retrieval | s2p | [Legal, Written] | {'test': 90} | {'test': {'average_document_length': 1224.2666666666667, 'average_query_length': 72.82, 'num_documents': 90, 'num_queries': 100, 'average_relevant_docs_per_query': 1.0}} | -| [SyntheticText2SQL](https://huggingface.co/datasets/gretelai/synthetic_text_to_sql) (Meyer et al., 2024) | ['eng', 'sql'] | Retrieval | p2p | [Programming, Written] | {'test': 1000} | {'test': {'average_document_length': 127.07126054548375, 'average_query_length': 82.90582806357888, 'num_documents': 105851, 'num_queries': 5851, 'average_relevant_docs_per_query': 1.0}} | +| [SyntecRetrieval](https://huggingface.co/datasets/lyon-nlp/mteb-fr-retrieval-syntec-s2p) (Mathieu Ciancone, 2024) | ['fra'] | Retrieval | s2p | [Legal, Written] | None | None | +| [SyntheticText2SQL](https://huggingface.co/datasets/gretelai/synthetic_text_to_sql) (Meyer et al., 2024) | ['eng', 'sql'] | Retrieval | p2p | [Programming, Written] | {'test': 111702} | {'test': {'number_of_characters': 210.98, 'num_samples': 111702, 'num_queries': 5851, 'num_documents': 105851, 'average_document_length': 0.0, 'average_query_length': 0.01, 'average_relevant_docs_per_query': 1.0}} | | [T2Reranking](https://arxiv.org/abs/2304.03679) (Xiaohui Xie, 2023) | ['cmn'] | Reranking | s2s | | None | None | -| [T2Retrieval](https://arxiv.org/abs/2304.03679) (Xiaohui Xie, 2023) | ['cmn'] | Retrieval | s2p | | None | {'dev': {'average_document_length': 874.1184182791619, 'average_query_length': 10.938847974750132, 'num_documents': 118605, 'num_queries': 22812, 'average_relevant_docs_per_query': 5.213571804313519}} | -| [TERRa](https://arxiv.org/pdf/2010.15925) (Shavrina et al., 2020) | ['rus'] | PairClassification | s2s | [News, Web, Written] | {'dev': 307} | {'dev': 138.2} | +| [T2Retrieval](https://arxiv.org/abs/2304.03679) (Xiaohui Xie, 2023) | ['cmn'] | Retrieval | s2p | | None | None | +| [TERRa](https://arxiv.org/pdf/2010.15925) (Shavrina et al., 2020) | ['rus'] | PairClassification | s2s | [News, Web, Written] | None | None | | [TNews](https://www.cluebenchmarks.com/introduce.html) | ['cmn'] | Classification | s2s | | None | None | -| [TRECCOVID](https://ir.nist.gov/covidSubmit/index.html) (Kirk Roberts, 2021) | ['eng'] | Retrieval | s2p | | None | {'test': {'average_document_length': 1116.7434221277986, 'average_query_length': 69.24, 'num_documents': 171332, 'num_queries': 50, 'average_relevant_docs_per_query': 493.5}} | -| [TRECCOVID-PL](https://ir.nist.gov/covidSubmit/index.html) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | [Academic, Non-fiction, Written] | None | {'test': {'average_document_length': 1159.8020276422385, 'average_query_length': 69.42, 'num_documents': 171332, 'num_queries': 50, 'average_relevant_docs_per_query': 493.5}} | -| [TV2Nordretrieval](https://huggingface.co/datasets/alexandrainst/nordjylland-news-summarization) | ['dan'] | Retrieval | p2p | [News, Non-fiction, Written] | {'test': 4096} | {'test': {'average_document_length': 1440.66552734375, 'average_query_length': 126.552734375, 'num_documents': 2048, 'num_queries': 2048, 'average_relevant_docs_per_query': 1.0}} | -| [TamilNewsClassification](https://github.com/vanangamudi/tamil-news-classification) (Anoop Kunchukuttan, 2020) | ['tam'] | Classification | s2s | [News, Written] | {'train': 14521, 'test': 3631} | {'train': 56.5, 'test': 56.52} | -| [Tatoeba](https://github.com/facebookresearch/LASER/tree/main/data/tatoeba/v1) (Tatoeba community, 2021) | ['afr', 'amh', 'ang', 'ara', 'arq', 'arz', 'ast', 'awa', 'aze', 'bel', 'ben', 'ber', 'bos', 'bre', 'bul', 'cat', 'cbk', 'ceb', 'ces', 'cha', 'cmn', 'cor', 'csb', 'cym', 'dan', 'deu', 'dsb', 'dtp', 'ell', 'eng', 'epo', 'est', 'eus', 'fao', 'fin', 'fra', 'fry', 'gla', 'gle', 'glg', 'gsw', 'heb', 'hin', 'hrv', 'hsb', 'hun', 'hye', 'ido', 'ile', 'ina', 'ind', 'isl', 'ita', 'jav', 'jpn', 'kab', 'kat', 'kaz', 'khm', 'kor', 'kur', 'kzj', 'lat', 'lfn', 'lit', 'lvs', 'mal', 'mar', 'max', 'mhr', 'mkd', 'mon', 'nds', 'nld', 'nno', 'nob', 'nov', 'oci', 'orv', 'pam', 'pes', 'pms', 'pol', 'por', 'ron', 'rus', 'slk', 'slv', 'spa', 'sqi', 'srp', 'swe', 'swg', 'swh', 'tam', 'tat', 'tel', 'tgl', 'tha', 'tuk', 'tur', 'tzl', 'uig', 'ukr', 'urd', 'uzb', 'vie', 'war', 'wuu', 'xho', 'yid', 'yue', 'zsm'] | BitextMining | s2s | [Written] | {'test': 2000} | {'test': 39.4} | -| [TbilisiCityHallBitextMining](https://huggingface.co/datasets/jupyterjazz/tbilisi-city-hall-titles) | ['eng', 'kat'] | BitextMining | s2s | [News, Written] | {'test': 1820} | {'test': 78} | -| [TelemarketingSalesRuleLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 47} | {'test': 348.29} | -| [TeluguAndhraJyotiNewsClassification](https://github.com/AnushaMotamarri/Telugu-Newspaper-Article-Dataset) | ['tel'] | Classification | s2s | [News, Written] | {'test': 4329} | {'test': 1428.28} | -| [TempReasonL1](https://github.com/DAMO-NLP-SG/TempReason) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | {'test': 4000} | {'test': {'average_document_length': 8.989843250159948, 'average_query_length': 50.22375, 'num_documents': 12504, 'num_queries': 4000, 'average_relevant_docs_per_query': 1.0}} | -| [TempReasonL2Context](https://github.com/DAMO-NLP-SG/TempReason) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | {'test': 0} | {'test': {'average_document_length': 19.823525685690758, 'average_query_length': 11919.25792106726, 'num_documents': 15787, 'num_queries': 5397, 'average_relevant_docs_per_query': 1.0}} | -| [TempReasonL2Fact](https://github.com/DAMO-NLP-SG/TempReason) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | {'test': 5397} | {'test': {'average_document_length': 19.823525685690758, 'average_query_length': 830.7268853066519, 'num_documents': 15787, 'num_queries': 5397, 'average_relevant_docs_per_query': 1.0}} | -| [TempReasonL2Pure](https://github.com/DAMO-NLP-SG/TempReason) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | {'test': 5397} | {'test': {'average_document_length': 19.823525685690758, 'average_query_length': 55.94089308875301, 'num_documents': 15787, 'num_queries': 5397, 'average_relevant_docs_per_query': 1.0}} | -| [TempReasonL3Context](https://github.com/DAMO-NLP-SG/TempReason) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | {'test': 4426} | {'test': {'average_document_length': 19.80534984678243, 'average_query_length': 13424.633077270673, 'num_documents': 15664, 'num_queries': 4426, 'average_relevant_docs_per_query': 1.0}} | -| [TempReasonL3Fact](https://github.com/DAMO-NLP-SG/TempReason) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | {'test': 4426} | {'test': {'average_document_length': 19.80534984678243, 'average_query_length': 896.0754631721645, 'num_documents': 15664, 'num_queries': 4426, 'average_relevant_docs_per_query': 1.0}} | -| [TempReasonL3Pure](https://github.com/DAMO-NLP-SG/TempReason) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | {'test': 4426} | {'test': {'average_document_length': 19.80534984678243, 'average_query_length': 74.44012652507908, 'num_documents': 15664, 'num_queries': 4426, 'average_relevant_docs_per_query': 1.0}} | -| [TenKGnadClassification](https://tblock.github.io/10kGNAD/) | ['deu'] | Classification | p2p | [News, Written] | {'test': 1028} | {'test': 2627.31} | -| [TenKGnadClusteringP2P.v2](https://tblock.github.io/10kGNAD/) | ['deu'] | Clustering | p2p | [News, Non-fiction, Written] | {'test': 10275} | {'test': 2641.03} | -| [TenKGnadClusteringS2S.v2](https://tblock.github.io/10kGNAD/) | ['deu'] | Clustering | s2s | [News, Non-fiction, Written] | {'test': 10267} | {'test': 50.96} | -| [TextualismToolDictionariesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 107} | {'test': 943.23} | -| [TextualismToolPlainLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 165} | {'test': 997.97} | -| [ThuNewsClusteringP2P.v2](http://thuctc.thunlp.org/) (Sun et al., 2016) | ['cmn'] | Clustering | p2p | [News, Written] | {'test': 2048} | {} | -| [ThuNewsClusteringS2S.v2](http://thuctc.thunlp.org/) (Sun et al., 2016) | ['cmn'] | Clustering | s2s | [News, Written] | {'test': 2048} | {} | -| [TopiOCQA](https://mcgill-nlp.github.io/topiocqa) (Vaibhav Adlakha, 2022) | ['eng'] | Retrieval | s2p | [Encyclopaedic, Written] | {'dev': 2514} | {'validation': {'average_document_length': 478.8968086416064, 'average_query_length': 12.579952267303103, 'num_documents': 25700592, 'num_queries': 2514, 'average_relevant_docs_per_query': 1.0}} | -| [TopiOCQAHardNegatives](https://mcgill-nlp.github.io/topiocqa) (Vaibhav Adlakha, 2022) | ['eng'] | Retrieval | s2p | [Encyclopaedic, Written] | {'test': 1000} | {'validation': {'average_document_length': 538.7586536643946, 'average_query_length': 12.85, 'num_documents': 89933, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}} | -| [Touche2020Retrieval.v3](https://github.com/castorini/touche-error-analysis) | ['eng'] | Retrieval | s2p | [Academic] | | | -| [ToxicChatClassification](https://aclanthology.org/2023.findings-emnlp.311/) (Zi Lin, 2023) | ['eng'] | Classification | s2s | [Constructed, Written] | {'test': 1427} | {'test': 189.4} | -| [ToxicConversationsClassification](https://www.kaggle.com/competitions/jigsaw-unintended-bias-in-toxicity-classification/overview) (cjadams, 2019) | ['eng'] | Classification | s2s | [Social, Written] | {'test': 50000} | {'test': 296.6} | -| [TswanaNewsClassification](https://link.springer.com/chapter/10.1007/978-3-031-49002-6_17) (Vukosi Marivate, 2023) | ['tsn'] | Classification | s2s | [News, Written] | {'validation': 487, 'test': 487} | {'validation': 2417.72, 'test': 2369.52} | -| [TurHistQuadRetrieval](https://github.com/okanvk/Turkish-Reading-Comprehension-Question-Answering-Dataset) (Soygazi et al., 2021) | ['tur'] | Retrieval | p2p | [Encyclopaedic, Non-fiction, Academic, Written] | {'test': 1330} | {'test': {'average_document_length': 172.12118713932398, 'average_query_length': 62.5302734375, 'num_documents': 1213, 'num_queries': 1024, 'average_relevant_docs_per_query': 2.0}} | -| [TurkicClassification](https://huggingface.co/datasets/Electrotubbie/classification_Turkic_languages/) | ['bak', 'kaz', 'kir'] | Classification | s2s | [News, Written] | {'train': 193056} | {'train': 1103.13} | -| [TurkishMovieSentimentClassification](https://www.win.tue.nl/~mpechen/publications/pubs/MT_WISDOM2013.pdf) (Erkin Demirtas, 2013) | ['tur'] | Classification | s2s | [Reviews, Written] | {'test': 2644} | {'test': 141.5} | -| [TurkishProductSentimentClassification](https://www.win.tue.nl/~mpechen/publications/pubs/MT_WISDOM2013.pdf) (Erkin Demirtas, 2013) | ['tur'] | Classification | s2s | [Reviews, Written] | {'test': 800} | {'test': 246.85} | -| [TweetEmotionClassification](https://link.springer.com/chapter/10.1007/978-3-319-77116-8_8) (Al-Khatib et al., 2018) | ['ara'] | Classification | s2s | [Social, Written] | {'train': 2048} | {'train': 78.8} | -| [TweetSarcasmClassification](https://aclanthology.org/2020.osact-1.5/) | ['ara'] | Classification | s2s | [Social, Written] | {'test': 2110} | {'test': 102.1} | -| [TweetSentimentClassification](https://aclanthology.org/2022.lrec-1.27) | ['ara', 'deu', 'eng', 'fra', 'hin', 'ita', 'por', 'spa'] | Classification | s2s | [Social, Written] | {'test': 2048} | {'test': 83.51} | -| [TweetSentimentExtractionClassification](https://www.kaggle.com/competitions/tweet-sentiment-extraction/overview) (Maggie et al., 2020) | ['eng'] | Classification | s2s | [Social, Written] | {'test': 3534} | {'test': 67.8} | -| [TweetTopicSingleClassification](https://arxiv.org/abs/2209.09824) | ['eng'] | Classification | s2s | [Social, News, Written] | {'test_2021': 1693} | {'test_2021': 167.66} | -| [TwentyNewsgroupsClustering.v2](https://scikit-learn.org/0.19/datasets/twenty_newsgroups.html) (Ken Lang, 1995) | ['eng'] | Clustering | s2s | [News, Written] | {'test': 2381} | {'test': 32.0} | -| [TwitterHjerneRetrieval](https://huggingface.co/datasets/sorenmulli/da-hashtag-twitterhjerne) (Holm et al., 2024) | ['dan'] | Retrieval | p2p | [Social, Written] | {'train': 340} | {'train': {'average_document_length': 128.85114503816794, 'average_query_length': 166.3846153846154, 'num_documents': 262, 'num_queries': 78, 'average_relevant_docs_per_query': 3.358974358974359}} | -| [TwitterSemEval2015](https://alt.qcri.org/semeval2015/task1/) | ['eng'] | PairClassification | s2s | | {'test': 16777} | {'test': 38.3} | -| [TwitterURLCorpus](https://languagenet.github.io/) | ['eng'] | PairClassification | s2s | | {'test': 51534} | {'test': {'num_samples': 51534, 'avg_sentence1_len': 79.48919160166103, 'avg_sentence2_len': 88.5540419916948, 'unique_labels': 2, 'labels': {'0': {'count': 38546}, '1': {'count': 12988}}}} | -| [UCCVCommonLawLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 94} | {'test': 114.127} | -| [UkrFormalityClassification](https://huggingface.co/datasets/ukr-detect/ukr-formality-dataset-translated-gyafc) | ['ukr'] | Classification | s2s | [News, Written] | {'train': 2048, 'test': 2048} | {'train': 52.1, 'test': 53.07} | -| [UnfairTOSLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | {'test': 2048} | {'test': 184.69} | -| [UrduRomanSentimentClassification](https://archive.ics.uci.edu/dataset/458/roman+urdu+data+set) (Sharf,Zareen, 2018) | ['urd'] | Classification | s2s | [Social, Written] | {'train': 2048} | {'train': 68.248} | -| [VGHierarchicalClusteringP2P](https://huggingface.co/datasets/navjordj/VG_summarization) (Navjord et al., 2023) | ['nob'] | Clustering | p2p | [News, Non-fiction, Written] | {'test': 2048} | {'test': 2670.3243084794544} | -| [VGHierarchicalClusteringS2S](https://huggingface.co/datasets/navjordj/VG_summarization) (Navjord et al., 2023) | ['nob'] | Clustering | p2p | [News, Non-fiction, Written] | {'test': 2048} | {'test': 139.31247668283325} | -| [VideoRetrieval](https://arxiv.org/abs/2203.03367) | ['cmn'] | Retrieval | s2p | | None | {'dev': {'average_document_length': 31.048855642524522, 'average_query_length': 7.365, 'num_documents': 100930, 'num_queries': 1000, 'average_relevant_docs_per_query': 1.0}} | -| [VieMedEVBitextMining](https://aclanthology.org/2015.iwslt-evaluation.11/) (Nhu Vo, 2024) | ['eng', 'vie'] | BitextMining | s2s | [Medical, Written] | {'test': 2048} | {'test': 139.23} | -| [VieQuADRetrieval](https://aclanthology.org/2020.coling-main.233.pdf) | ['vie'] | Retrieval | s2p | [Encyclopaedic, Non-fiction, Written] | {'validation': 2048} | {'validation': {'average_document_length': 222.61244979919678, 'average_query_length': 65.51513671875, 'num_documents': 2490, 'num_queries': 2048, 'average_relevant_docs_per_query': 2.0}} | -| [VieStudentFeedbackClassification](https://ieeexplore.ieee.org/document/8573337) (Nguyen et al., 2018) | ['vie'] | Classification | s2s | [Reviews, Written] | {'test': 2048} | {'test': 14.22} | -| [VoyageMMarcoReranking](https://arxiv.org/abs/2312.16144) (Benjamin Clavié, 2023) | ['jpn'] | Reranking | s2s | [Academic, Non-fiction, Written] | {'test': 2048} | {'test': 162} | -| [WRIMEClassification](https://aclanthology.org/2021.naacl-main.169/) | ['jpn'] | Classification | s2s | [Social, Written] | {'test': 2048} | {'test': 47.78} | +| [TRECCOVID](https://ir.nist.gov/covidSubmit/index.html) (Kirk Roberts, 2021) | ['eng'] | Retrieval | s2p | | None | None | +| [TRECCOVID-PL](https://ir.nist.gov/covidSubmit/index.html) (Konrad Wojtasik, 2024) | ['pol'] | Retrieval | s2p | [Academic, Non-fiction, Written] | None | None | +| [TV2Nordretrieval](https://huggingface.co/datasets/alexandrainst/nordjylland-news-summarization) | ['dan'] | Retrieval | p2p | [News, Non-fiction, Written] | None | None | +| [TamilNewsClassification](https://github.com/vanangamudi/tamil-news-classification) (Anoop Kunchukuttan, 2020) | ['tam'] | Classification | s2s | [News, Written] | None | None | +| [Tatoeba](https://github.com/facebookresearch/LASER/tree/main/data/tatoeba/v1) (Tatoeba community, 2021) | ['afr', 'amh', 'ang', 'ara', 'arq', 'arz', 'ast', 'awa', 'aze', 'bel', 'ben', 'ber', 'bos', 'bre', 'bul', 'cat', 'cbk', 'ceb', 'ces', 'cha', 'cmn', 'cor', 'csb', 'cym', 'dan', 'deu', 'dsb', 'dtp', 'ell', 'eng', 'epo', 'est', 'eus', 'fao', 'fin', 'fra', 'fry', 'gla', 'gle', 'glg', 'gsw', 'heb', 'hin', 'hrv', 'hsb', 'hun', 'hye', 'ido', 'ile', 'ina', 'ind', 'isl', 'ita', 'jav', 'jpn', 'kab', 'kat', 'kaz', 'khm', 'kor', 'kur', 'kzj', 'lat', 'lfn', 'lit', 'lvs', 'mal', 'mar', 'max', 'mhr', 'mkd', 'mon', 'nds', 'nld', 'nno', 'nob', 'nov', 'oci', 'orv', 'pam', 'pes', 'pms', 'pol', 'por', 'ron', 'rus', 'slk', 'slv', 'spa', 'sqi', 'srp', 'swe', 'swg', 'swh', 'tam', 'tat', 'tel', 'tgl', 'tha', 'tuk', 'tur', 'tzl', 'uig', 'ukr', 'urd', 'uzb', 'vie', 'war', 'wuu', 'xho', 'yid', 'yue', 'zsm'] | BitextMining | s2s | [Written] | None | None | +| [TbilisiCityHallBitextMining](https://huggingface.co/datasets/jupyterjazz/tbilisi-city-hall-titles) | ['eng', 'kat'] | BitextMining | s2s | [News, Written] | None | None | +| [TelemarketingSalesRuleLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [TeluguAndhraJyotiNewsClassification](https://github.com/AnushaMotamarri/Telugu-Newspaper-Article-Dataset) | ['tel'] | Classification | s2s | [News, Written] | None | None | +| [TempReasonL1](https://github.com/DAMO-NLP-SG/TempReason) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | None | None | +| [TempReasonL2Context](https://github.com/DAMO-NLP-SG/TempReason) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | None | None | +| [TempReasonL2Fact](https://github.com/DAMO-NLP-SG/TempReason) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | None | None | +| [TempReasonL2Pure](https://github.com/DAMO-NLP-SG/TempReason) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | None | None | +| [TempReasonL3Context](https://github.com/DAMO-NLP-SG/TempReason) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | None | None | +| [TempReasonL3Fact](https://github.com/DAMO-NLP-SG/TempReason) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | None | None | +| [TempReasonL3Pure](https://github.com/DAMO-NLP-SG/TempReason) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | None | None | +| [TenKGnadClassification](https://tblock.github.io/10kGNAD/) | ['deu'] | Classification | p2p | [News, Written] | None | None | +| [TenKGnadClusteringP2P.v2](https://tblock.github.io/10kGNAD/) | ['deu'] | Clustering | p2p | [News, Non-fiction, Written] | None | None | +| [TenKGnadClusteringS2S.v2](https://tblock.github.io/10kGNAD/) | ['deu'] | Clustering | s2s | [News, Non-fiction, Written] | None | None | +| [TextualismToolDictionariesLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [TextualismToolPlainLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [ThuNewsClusteringP2P.v2](http://thuctc.thunlp.org/) (Sun et al., 2016) | ['cmn'] | Clustering | p2p | [News, Written] | None | None | +| [ThuNewsClusteringS2S.v2](http://thuctc.thunlp.org/) (Sun et al., 2016) | ['cmn'] | Clustering | s2s | [News, Written] | None | None | +| [TopiOCQA](https://mcgill-nlp.github.io/topiocqa) (Vaibhav Adlakha, 2022) | ['eng'] | Retrieval | s2p | [Encyclopaedic, Written] | None | None | +| [TopiOCQAHardNegatives](https://mcgill-nlp.github.io/topiocqa) (Vaibhav Adlakha, 2022) | ['eng'] | Retrieval | s2p | [Encyclopaedic, Written] | None | None | +| [Touche2020Retrieval.v3](https://github.com/castorini/touche-error-analysis) | ['eng'] | Retrieval | s2p | [Academic] | {'test': 303781} | {'test': {'number_of_characters': 2140.82, 'num_samples': 303781, 'num_queries': 49, 'num_documents': 303732, 'average_document_length': 0.01, 'average_query_length': 0.89, 'average_relevant_docs_per_query': 34.94}} | +| [ToxicChatClassification](https://aclanthology.org/2023.findings-emnlp.311/) (Zi Lin, 2023) | ['eng'] | Classification | s2s | [Constructed, Written] | None | None | +| [ToxicConversationsClassification](https://www.kaggle.com/competitions/jigsaw-unintended-bias-in-toxicity-classification/overview) (cjadams, 2019) | ['eng'] | Classification | s2s | [Social, Written] | None | None | +| [TswanaNewsClassification](https://link.springer.com/chapter/10.1007/978-3-031-49002-6_17) (Vukosi Marivate, 2023) | ['tsn'] | Classification | s2s | [News, Written] | None | None | +| [TurHistQuadRetrieval](https://github.com/okanvk/Turkish-Reading-Comprehension-Question-Answering-Dataset) (Soygazi et al., 2021) | ['tur'] | Retrieval | p2p | [Encyclopaedic, Non-fiction, Academic, Written] | None | None | +| [TurkicClassification](https://huggingface.co/datasets/Electrotubbie/classification_Turkic_languages/) | ['bak', 'kaz', 'kir'] | Classification | s2s | [News, Written] | None | None | +| [TurkishMovieSentimentClassification](https://www.win.tue.nl/~mpechen/publications/pubs/MT_WISDOM2013.pdf) (Erkin Demirtas, 2013) | ['tur'] | Classification | s2s | [Reviews, Written] | None | None | +| [TurkishProductSentimentClassification](https://www.win.tue.nl/~mpechen/publications/pubs/MT_WISDOM2013.pdf) (Erkin Demirtas, 2013) | ['tur'] | Classification | s2s | [Reviews, Written] | None | None | +| [TweetEmotionClassification](https://link.springer.com/chapter/10.1007/978-3-319-77116-8_8) (Al-Khatib et al., 2018) | ['ara'] | Classification | s2s | [Social, Written] | None | None | +| [TweetSarcasmClassification](https://aclanthology.org/2020.osact-1.5/) | ['ara'] | Classification | s2s | [Social, Written] | None | None | +| [TweetSentimentClassification](https://aclanthology.org/2022.lrec-1.27) | ['ara', 'deu', 'eng', 'fra', 'hin', 'ita', 'por', 'spa'] | Classification | s2s | [Social, Written] | None | None | +| [TweetSentimentExtractionClassification](https://www.kaggle.com/competitions/tweet-sentiment-extraction/overview) (Maggie et al., 2020) | ['eng'] | Classification | s2s | [Social, Written] | None | None | +| [TweetTopicSingleClassification](https://arxiv.org/abs/2209.09824) | ['eng'] | Classification | s2s | [Social, News, Written] | None | None | +| [TwentyNewsgroupsClustering.v2](https://scikit-learn.org/0.19/datasets/twenty_newsgroups.html) (Ken Lang, 1995) | ['eng'] | Clustering | s2s | [News, Written] | None | None | +| [TwitterHjerneRetrieval](https://huggingface.co/datasets/sorenmulli/da-hashtag-twitterhjerne) (Holm et al., 2024) | ['dan'] | Retrieval | p2p | [Social, Written] | None | None | +| [TwitterSemEval2015](https://alt.qcri.org/semeval2015/task1/) | ['eng'] | PairClassification | s2s | | None | None | +| [TwitterURLCorpus](https://languagenet.github.io/) | ['eng'] | PairClassification | s2s | | {'test': 51534} | {'test': {'num_samples': 51534, 'number_of_characters': 8659940, 'avg_sentence1_len': 79.49, 'avg_sentence2_len': 88.55, 'unique_labels': 2, 'labels': {'0': {'count': 38546}, '1': {'count': 12988}}}} | +| [UCCVCommonLawLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [UkrFormalityClassification](https://huggingface.co/datasets/ukr-detect/ukr-formality-dataset-translated-gyafc) | ['ukr'] | Classification | s2s | [News, Written] | None | None | +| [UnfairTOSLegalBenchClassification](https://huggingface.co/datasets/nguha/legalbench) (Neel Guha, 2023) | ['eng'] | Classification | s2s | [Legal, Written] | None | None | +| [UrduRomanSentimentClassification](https://archive.ics.uci.edu/dataset/458/roman+urdu+data+set) (Sharf,Zareen, 2018) | ['urd'] | Classification | s2s | [Social, Written] | None | None | +| [VGHierarchicalClusteringP2P](https://huggingface.co/datasets/navjordj/VG_summarization) (Navjord et al., 2023) | ['nob'] | Clustering | p2p | [News, Non-fiction, Written] | None | None | +| [VGHierarchicalClusteringS2S](https://huggingface.co/datasets/navjordj/VG_summarization) (Navjord et al., 2023) | ['nob'] | Clustering | p2p | [News, Non-fiction, Written] | None | None | +| [VideoRetrieval](https://arxiv.org/abs/2203.03367) | ['cmn'] | Retrieval | s2p | | None | None | +| [VieMedEVBitextMining](https://aclanthology.org/2015.iwslt-evaluation.11/) (Nhu Vo, 2024) | ['eng', 'vie'] | BitextMining | s2s | [Medical, Written] | None | None | +| [VieQuADRetrieval](https://aclanthology.org/2020.coling-main.233.pdf) | ['vie'] | Retrieval | s2p | [Encyclopaedic, Non-fiction, Written] | None | None | +| [VieStudentFeedbackClassification](https://ieeexplore.ieee.org/document/8573337) (Nguyen et al., 2018) | ['vie'] | Classification | s2s | [Reviews, Written] | None | None | +| [VoyageMMarcoReranking](https://arxiv.org/abs/2312.16144) (Benjamin Clavié, 2023) | ['jpn'] | Reranking | s2s | [Academic, Non-fiction, Written] | None | None | +| [WRIMEClassification](https://aclanthology.org/2021.naacl-main.169/) | ['jpn'] | Classification | s2s | [Social, Written] | None | None | | [Waimai](https://aclanthology.org/2023.nodalida-1.20/) (Xiao et al., 2023) | ['cmn'] | Classification | s2s | | None | None | -| [WebLINXCandidatesReranking](https://mcgill-nlp.github.io/weblinx) (Xing Han Lù, 2024) | ['eng'] | Reranking | p2p | [Academic, Web, Written] | {'validation': 1301, 'test_iid': 1438, 'test_cat': 3560, 'test_web': 3144, 'test_vis': 5298, 'test_geo': 4916} | {'validation': 1647.52, 'test_iid': 1722.63, 'test_cat': 2149.66, 'test_web': 1831.46, 'test_vis': 1737.26, 'test_geo': 1742.66} | +| [WebLINXCandidatesReranking](https://mcgill-nlp.github.io/weblinx) (Xing Han Lù, 2024) | ['eng'] | Reranking | p2p | [Academic, Web, Written] | None | None | | [WikiCitiesClustering](https://huggingface.co/datasets/wikipedia) | ['eng'] | Clustering | p2p | [Encyclopaedic, Written] | None | None | -| [WikiClusteringP2P.v2](https://github.com/Rysias/wiki-clustering) | ['bos', 'cat', 'ces', 'dan', 'eus', 'glv', 'ilo', 'kur', 'lav', 'min', 'mlt', 'sco', 'sqi', 'wln'] | Clustering | p2p | [Encyclopaedic, Written] | {'test': 2048} | {'test': {'num_samples': 28672, 'average_text_length': 629.7426409040179, 'average_labels_per_text': 1.0, 'unique_labels': 39, 'labels': {'16': {'count': 541}, '3': {'count': 1607}, '12': {'count': 846}, '0': {'count': 2410}, '15': {'count': 878}, '11': {'count': 864}, '6': {'count': 787}, '9': {'count': 654}, '14': {'count': 966}, '8': {'count': 1389}, '2': {'count': 2428}, '10': {'count': 839}, '1': {'count': 1370}, '4': {'count': 2942}, '7': {'count': 2514}, '5': {'count': 1490}, '13': {'count': 918}, '19': {'count': 315}, '17': {'count': 711}, '20': {'count': 345}, '18': {'count': 800}, '24': {'count': 467}, '25': {'count': 928}, '21': {'count': 62}, '26': {'count': 270}, '22': {'count': 186}, '23': {'count': 36}, '27': {'count': 465}, '28': {'count': 62}, '36': {'count': 139}, '32': {'count': 57}, '38': {'count': 43}, '30': {'count': 52}, '34': {'count': 80}, '33': {'count': 75}, '35': {'count': 62}, '31': {'count': 63}, '37': {'count': 8}, '29': {'count': 3}}, 'hf_subset_descriptive_stats': {'bs': {'num_samples': 2048, 'average_text_length': 1046.25732421875, 'average_labels_per_text': 1.0, 'unique_labels': 17, 'labels': {'16': {'count': 268}, '3': {'count': 89}, '12': {'count': 597}, '0': {'count': 202}, '15': {'count': 113}, '11': {'count': 11}, '6': {'count': 142}, '9': {'count': 181}, '14': {'count': 179}, '8': {'count': 33}, '2': {'count': 172}, '10': {'count': 12}, '1': {'count': 7}, '4': {'count': 25}, '7': {'count': 6}, '5': {'count': 9}, '13': {'count': 2}}}, 'ca': {'num_samples': 2048, 'average_text_length': 600.73291015625, 'average_labels_per_text': 1.0, 'unique_labels': 8, 'labels': {'6': {'count': 257}, '1': {'count': 737}, '2': {'count': 284}, '4': {'count': 394}, '0': {'count': 162}, '7': {'count': 151}, '5': {'count': 55}, '3': {'count': 8}}}, 'cs': {'num_samples': 2048, 'average_text_length': 659.2294921875, 'average_labels_per_text': 1.0, 'unique_labels': 21, 'labels': {'19': {'count': 35}, '5': {'count': 624}, '17': {'count': 126}, '10': {'count': 155}, '1': {'count': 231}, '7': {'count': 215}, '11': {'count': 128}, '0': {'count': 57}, '13': {'count': 75}, '2': {'count': 83}, '3': {'count': 38}, '9': {'count': 8}, '6': {'count': 14}, '12': {'count': 9}, '16': {'count': 16}, '20': {'count': 73}, '18': {'count': 38}, '4': {'count': 60}, '15': {'count': 14}, '14': {'count': 38}, '8': {'count': 11}}}, 'da': {'num_samples': 2048, 'average_text_length': 767.58935546875, 'average_labels_per_text': 1.0, 'unique_labels': 20, 'labels': {'14': {'count': 212}, '4': {'count': 74}, '15': {'count': 16}, '8': {'count': 165}, '13': {'count': 115}, '0': {'count': 79}, '1': {'count': 34}, '9': {'count': 114}, '7': {'count': 364}, '10': {'count': 32}, '17': {'count': 66}, '18': {'count': 32}, '12': {'count': 129}, '11': {'count': 159}, '2': {'count': 66}, '3': {'count': 185}, '19': {'count': 103}, '16': {'count': 33}, '5': {'count': 56}, '6': {'count': 14}}}, 'eu': {'num_samples': 2048, 'average_text_length': 405.16015625, 'average_labels_per_text': 1.0, 'unique_labels': 5, 'labels': {'4': {'count': 383}, '0': {'count': 995}, '3': {'count': 282}, '2': {'count': 344}, '1': {'count': 44}}}, 'gv': {'num_samples': 2048, 'average_text_length': 368.01123046875, 'average_labels_per_text': 1.0, 'unique_labels': 28, 'labels': {'6': {'count': 32}, '1': {'count': 83}, '24': {'count': 13}, '17': {'count': 152}, '2': {'count': 534}, '25': {'count': 76}, '5': {'count': 198}, '15': {'count': 100}, '21': {'count': 22}, '26': {'count': 188}, '13': {'count': 230}, '20': {'count': 11}, '3': {'count': 107}, '19': {'count': 88}, '16': {'count': 55}, '22': {'count': 29}, '14': {'count': 12}, '8': {'count': 61}, '0': {'count': 5}, '10': {'count': 4}, '4': {'count': 9}, '23': {'count': 6}, '7': {'count': 3}, '9': {'count': 20}, '18': {'count': 4}, '12': {'count': 3}, '27': {'count': 1}, '11': {'count': 2}}}, 'ilo': {'num_samples': 2048, 'average_text_length': 617.90771484375, 'average_labels_per_text': 1.0, 'unique_labels': 29, 'labels': {'3': {'count': 562}, '0': {'count': 373}, '18': {'count': 521}, '8': {'count': 129}, '13': {'count': 123}, '11': {'count': 54}, '25': {'count': 8}, '27': {'count': 5}, '17': {'count': 13}, '15': {'count': 4}, '4': {'count': 28}, '7': {'count': 83}, '10': {'count': 15}, '1': {'count': 11}, '24': {'count': 15}, '14': {'count': 8}, '16': {'count': 4}, '19': {'count': 9}, '23': {'count': 10}, '26': {'count': 4}, '28': {'count': 8}, '12': {'count': 29}, '21': {'count': 12}, '6': {'count': 5}, '20': {'count': 6}, '5': {'count': 4}, '22': {'count': 2}, '9': {'count': 2}, '2': {'count': 1}}}, 'ku': {'num_samples': 2048, 'average_text_length': 421.17333984375, 'average_labels_per_text': 1.0, 'unique_labels': 39, 'labels': {'14': {'count': 14}, '36': {'count': 139}, '20': {'count': 108}, '22': {'count': 27}, '15': {'count': 102}, '32': {'count': 55}, '8': {'count': 431}, '17': {'count': 210}, '38': {'count': 43}, '30': {'count': 51}, '4': {'count': 60}, '2': {'count': 111}, '6': {'count': 95}, '34': {'count': 70}, '27': {'count': 15}, '5': {'count': 174}, '26': {'count': 37}, '0': {'count': 11}, '25': {'count': 50}, '16': {'count': 2}, '12': {'count': 16}, '24': {'count': 2}, '11': {'count': 17}, '21': {'count': 9}, '13': {'count': 20}, '1': {'count': 7}, '33': {'count': 33}, '35': {'count': 28}, '10': {'count': 11}, '31': {'count': 51}, '18': {'count': 4}, '3': {'count': 4}, '28': {'count': 8}, '37': {'count': 8}, '23': {'count': 2}, '19': {'count': 7}, '7': {'count': 6}, '9': {'count': 8}, '29': {'count': 2}}}, 'lv': {'num_samples': 2048, 'average_text_length': 770.67138671875, 'average_labels_per_text': 1.0, 'unique_labels': 16, 'labels': {'15': {'count': 288}, '2': {'count': 110}, '6': {'count': 74}, '12': {'count': 50}, '0': {'count': 171}, '14': {'count': 188}, '10': {'count': 351}, '5': {'count': 142}, '4': {'count': 300}, '13': {'count': 60}, '11': {'count': 48}, '1': {'count': 165}, '8': {'count': 53}, '7': {'count': 5}, '3': {'count': 9}, '9': {'count': 34}}}, 'min': {'num_samples': 2048, 'average_text_length': 631.74072265625, 'average_labels_per_text': 1.0, 'unique_labels': 15, 'labels': {'7': {'count': 1595}, '9': {'count': 9}, '4': {'count': 48}, '3': {'count': 83}, '2': {'count': 160}, '0': {'count': 19}, '5': {'count': 74}, '6': {'count': 12}, '10': {'count': 12}, '13': {'count': 10}, '8': {'count': 5}, '11': {'count': 13}, '12': {'count': 2}, '1': {'count': 5}, '14': {'count': 1}}}, 'mt': {'num_samples': 2048, 'average_text_length': 821.22265625, 'average_labels_per_text': 1.0, 'unique_labels': 27, 'labels': {'12': {'count': 8}, '10': {'count': 147}, '14': {'count': 180}, '17': {'count': 117}, '25': {'count': 654}, '19': {'count': 35}, '0': {'count': 77}, '3': {'count': 12}, '16': {'count': 44}, '15': {'count': 108}, '24': {'count': 267}, '6': {'count': 43}, '26': {'count': 32}, '4': {'count': 79}, '22': {'count': 67}, '9': {'count': 16}, '8': {'count': 16}, '2': {'count': 55}, '5': {'count': 6}, '11': {'count': 30}, '18': {'count': 12}, '21': {'count': 12}, '20': {'count': 15}, '23': {'count': 7}, '13': {'count': 6}, '7': {'count': 1}, '1': {'count': 2}}}, 'sco': {'num_samples': 2048, 'average_text_length': 1065.21044921875, 'average_labels_per_text': 1.0, 'unique_labels': 23, 'labels': {'18': {'count': 178}, '6': {'count': 92}, '9': {'count': 28}, '15': {'count': 106}, '8': {'count': 432}, '2': {'count': 95}, '11': {'count': 104}, '1': {'count': 42}, '13': {'count': 248}, '16': {'count': 118}, '20': {'count': 130}, '3': {'count': 171}, '22': {'count': 57}, '7': {'count': 83}, '10': {'count': 74}, '5': {'count': 6}, '4': {'count': 17}, '17': {'count': 24}, '14': {'count': 14}, '0': {'count': 7}, '19': {'count': 18}, '21': {'count': 3}, '12': {'count': 1}}}, 'sq': {'num_samples': 2048, 'average_text_length': 425.486328125, 'average_labels_per_text': 1.0, 'unique_labels': 36, 'labels': {'27': {'count': 444}, '9': {'count': 234}, '14': {'count': 120}, '0': {'count': 128}, '15': {'count': 27}, '11': {'count': 298}, '24': {'count': 170}, '28': {'count': 46}, '19': {'count': 20}, '25': {'count': 140}, '3': {'count': 47}, '2': {'count': 87}, '35': {'count': 34}, '8': {'count': 53}, '31': {'count': 12}, '17': {'count': 3}, '23': {'count': 11}, '20': {'count': 2}, '33': {'count': 42}, '10': {'count': 26}, '34': {'count': 10}, '7': {'count': 2}, '13': {'count': 29}, '4': {'count': 4}, '6': {'count': 7}, '26': {'count': 9}, '5': {'count': 16}, '30': {'count': 1}, '21': {'count': 4}, '22': {'count': 4}, '18': {'count': 11}, '32': {'count': 2}, '12': {'count': 2}, '16': {'count': 1}, '1': {'count': 1}, '29': {'count': 1}}}, 'wa': {'num_samples': 2048, 'average_text_length': 216.00390625, 'average_labels_per_text': 1.0, 'unique_labels': 6, 'labels': {'5': {'count': 126}, '4': {'count': 1461}, '0': {'count': 124}, '2': {'count': 326}, '3': {'count': 10}, '1': {'count': 1}}}}}} | -| [WikipediaRerankingMultilingual](https://huggingface.co/datasets/ellamind/wikipedia-2023-11-reranking-multilingual) | ['ben', 'bul', 'ces', 'dan', 'deu', 'eng', 'fas', 'fin', 'hin', 'ita', 'nld', 'nor', 'por', 'ron', 'srp', 'swe'] | Reranking | s2p | [Encyclopaedic, Written] | {'en': 1500, 'de': 1500, 'it': 1500, 'pt': 1500, 'nl': 1500, 'cs': 1500, 'ro': 1500, 'bg': 1500, 'sr': 1500, 'fi': 1500, 'da': 1500, 'fa': 1500, 'hi': 1500, 'bn': 1500, 'no': 1500, 'sv': 1500} | {'test': {'num_samples': 24000, 'num_positive': 24000, 'num_negative': 24000, 'avg_query_len': 59.091208333333334, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0, 'hf_subset_descriptive_stats': {'bg': {'num_samples': 1500, 'num_positive': 1500, 'num_negative': 1500, 'avg_query_len': 60.82666666666667, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0}, 'bn': {'num_samples': 1500, 'num_positive': 1500, 'num_negative': 1500, 'avg_query_len': 47.266666666666666, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0}, 'cs': {'num_samples': 1500, 'num_positive': 1500, 'num_negative': 1500, 'avg_query_len': 56.272, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0}, 'da': {'num_samples': 1500, 'num_positive': 1500, 'num_negative': 1500, 'avg_query_len': 56.75066666666667, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0}, 'de': {'num_samples': 1500, 'num_positive': 1500, 'num_negative': 1500, 'avg_query_len': 70.004, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0}, 'en': {'num_samples': 1500, 'num_positive': 1500, 'num_negative': 1500, 'avg_query_len': 68.372, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0}, 'fa': {'num_samples': 1500, 'num_positive': 1500, 'num_negative': 1500, 'avg_query_len': 48.66733333333333, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0}, 'fi': {'num_samples': 1500, 'num_positive': 1500, 'num_negative': 1500, 'avg_query_len': 55.343333333333334, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0}, 'hi': {'num_samples': 1500, 'num_positive': 1500, 'num_negative': 1500, 'avg_query_len': 50.77733333333333, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0}, 'it': {'num_samples': 1500, 'num_positive': 1500, 'num_negative': 1500, 'avg_query_len': 70.05466666666666, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0}, 'nl': {'num_samples': 1500, 'num_positive': 1500, 'num_negative': 1500, 'avg_query_len': 65.34466666666667, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0}, 'pt': {'num_samples': 1500, 'num_positive': 1500, 'num_negative': 1500, 'avg_query_len': 65.11933333333333, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0}, 'ro': {'num_samples': 1500, 'num_positive': 1500, 'num_negative': 1500, 'avg_query_len': 61.973333333333336, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0}, 'sr': {'num_samples': 1500, 'num_positive': 1500, 'num_negative': 1500, 'avg_query_len': 55.669333333333334, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0}, 'no': {'num_samples': 1500, 'num_positive': 1500, 'num_negative': 1500, 'avg_query_len': 55.288, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0}, 'sv': {'num_samples': 1500, 'num_positive': 1500, 'num_negative': 1500, 'avg_query_len': 57.73, 'avg_positive_len': 1.0, 'avg_negative_len': 8.0}}}} | -| [WikipediaRetrievalMultilingual](https://huggingface.co/datasets/ellamind/wikipedia-2023-11-retrieval-multilingual-queries) | ['ben', 'bul', 'ces', 'dan', 'deu', 'eng', 'fas', 'fin', 'hin', 'ita', 'nld', 'nor', 'por', 'ron', 'srp', 'swe'] | Retrieval | s2p | [Encyclopaedic, Written] | {'en': 1500, 'de': 1500, 'it': 1500, 'pt': 1500, 'nl': 1500, 'cs': 1500, 'ro': 1500, 'bg': 1500, 'sr': 1500, 'fi': 1500, 'da': 1500, 'fa': 1500, 'hi': 1500, 'bn': 1500, 'no': 1500, 'sv': 1500} | {'test': {'bg': {'average_document_length': 374.376, 'average_query_length': 1.0, 'num_documents': 13500, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}, 'bn': {'average_document_length': 394.05044444444445, 'average_query_length': 1.0, 'num_documents': 13500, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}, 'cs': {'average_document_length': 369.9831111111111, 'average_query_length': 1.0, 'num_documents': 13500, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}, 'da': {'average_document_length': 345.2597037037037, 'average_query_length': 1.0, 'num_documents': 13500, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}, 'de': {'average_document_length': 398.4137777777778, 'average_query_length': 1.0, 'num_documents': 13500, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}, 'en': {'average_document_length': 452.9871111111111, 'average_query_length': 1.0, 'num_documents': 13500, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}, 'fa': {'average_document_length': 345.1568888888889, 'average_query_length': 1.0, 'num_documents': 13500, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}, 'fi': {'average_document_length': 379.71237037037037, 'average_query_length': 1.0, 'num_documents': 13500, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}, 'hi': {'average_document_length': 410.72540740740743, 'average_query_length': 1.0, 'num_documents': 13500, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}, 'it': {'average_document_length': 393.73437037037036, 'average_query_length': 1.0, 'num_documents': 13500, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}, 'nl': {'average_document_length': 375.6695555555556, 'average_query_length': 1.0, 'num_documents': 13500, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}, 'pt': {'average_document_length': 398.27237037037037, 'average_query_length': 1.0, 'num_documents': 13500, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}, 'ro': {'average_document_length': 348.3817037037037, 'average_query_length': 1.0, 'num_documents': 13500, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}, 'sr': {'average_document_length': 384.3131851851852, 'average_query_length': 1.0, 'num_documents': 13500, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}, 'no': {'average_document_length': 366.93733333333336, 'average_query_length': 1.0, 'num_documents': 13500, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}, 'sv': {'average_document_length': 369.340962962963, 'average_query_length': 1.0, 'num_documents': 13500, 'num_queries': 1500, 'average_relevant_docs_per_query': 1.0}}} | -| [WinoGrande](https://winogrande.allenai.org/) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | {'test': 0} | {'test': {'average_document_length': 7.68243375858685, 'average_query_length': 111.78216258879242, 'num_documents': 5095, 'num_queries': 1267, 'average_relevant_docs_per_query': 1.0}} | -| [WisesightSentimentClassification](https://github.com/PyThaiNLP/wisesight-sentiment) | ['tha'] | Classification | s2s | [Social, News, Written] | {'train': 2048} | {'train': 103.42} | -| XMarket (Bonab et al., 2021) | ['deu', 'eng', 'spa'] | Retrieval | s2p | | None | {'test': {'de': {'average_document_length': 187.4061197288943, 'average_query_length': 15.717612088184294, 'num_documents': 70526, 'num_queries': 4037, 'average_relevant_docs_per_query': 54.3522417636859}, 'en': {'average_document_length': 452.792089662076, 'average_query_length': 15.881635344543357, 'num_documents': 218777, 'num_queries': 9099, 'average_relevant_docs_per_query': 85.43719090009891}, 'es': {'average_document_length': 279.67909262759923, 'average_query_length': 19.97062937062937, 'num_documents': 39675, 'num_queries': 3575, 'average_relevant_docs_per_query': 36.01006993006993}}} | -| [XNLI](https://aclanthology.org/D18-1269/) (Conneau et al., 2018) | ['ara', 'bul', 'deu', 'ell', 'eng', 'fra', 'hin', 'rus', 'spa', 'swa', 'tha', 'tur', 'vie', 'zho'] | PairClassification | s2s | [Non-fiction, Fiction, Government, Written] | {'validation': 2163, 'test': 2460} | {'test': {'num_samples': 19110, 'avg_sentence1_len': 103.23793825222397, 'avg_sentence2_len': 48.88895866038723, 'unique_labels': 2, 'labels': {'0': {'count': 9562}, '1': {'count': 9548}}, 'hf_subset_descriptive_stats': {'ar': {'num_samples': 1365, 'avg_sentence1_len': 89.57362637362637, 'avg_sentence2_len': 41.99487179487179, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'bg': {'num_samples': 1365, 'avg_sentence1_len': 110.01611721611722, 'avg_sentence2_len': 51.62930402930403, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'de': {'num_samples': 1365, 'avg_sentence1_len': 119.92600732600732, 'avg_sentence2_len': 56.794871794871796, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'el': {'num_samples': 1365, 'avg_sentence1_len': 119.05421245421246, 'avg_sentence2_len': 56.93260073260073, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'en': {'num_samples': 1365, 'avg_sentence1_len': 105.67032967032966, 'avg_sentence2_len': 49.8043956043956, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'es': {'num_samples': 1365, 'avg_sentence1_len': 115.43296703296703, 'avg_sentence2_len': 54.68205128205128, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'fr': {'num_samples': 1365, 'avg_sentence1_len': 121.0967032967033, 'avg_sentence2_len': 58.58021978021978, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'hi': {'num_samples': 1365, 'avg_sentence1_len': 104.63443223443224, 'avg_sentence2_len': 50.17289377289377, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'ru': {'num_samples': 1365, 'avg_sentence1_len': 110.76923076923077, 'avg_sentence2_len': 52.452014652014654, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'sw': {'num_samples': 1365, 'avg_sentence1_len': 104.43956043956044, 'avg_sentence2_len': 49.48205128205128, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'th': {'num_samples': 1365, 'avg_sentence1_len': 96.6923076923077, 'avg_sentence2_len': 44.544322344322346, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'tr': {'num_samples': 1365, 'avg_sentence1_len': 103.67765567765568, 'avg_sentence2_len': 49.18534798534799, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'vi': {'num_samples': 1365, 'avg_sentence1_len': 111.31208791208792, 'avg_sentence2_len': 52.46007326007326, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'zh': {'num_samples': 1365, 'avg_sentence1_len': 33.03589743589744, 'avg_sentence2_len': 15.73040293040293, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}}}, 'validation': {'num_samples': 19110, 'avg_sentence1_len': 103.20790162218734, 'avg_sentence2_len': 49.01909994767138, 'unique_labels': 2, 'labels': {'0': {'count': 9562}, '1': {'count': 9548}}, 'hf_subset_descriptive_stats': {'ar': {'num_samples': 1365, 'avg_sentence1_len': 88.31868131868131, 'avg_sentence2_len': 41.61172161172161, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'bg': {'num_samples': 1365, 'avg_sentence1_len': 109.196336996337, 'avg_sentence2_len': 51.967032967032964, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'de': {'num_samples': 1365, 'avg_sentence1_len': 119.81172161172161, 'avg_sentence2_len': 57.36923076923077, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'el': {'num_samples': 1365, 'avg_sentence1_len': 119.87545787545787, 'avg_sentence2_len': 56.88278388278388, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'en': {'num_samples': 1365, 'avg_sentence1_len': 105.71648351648352, 'avg_sentence2_len': 49.87619047619047, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'es': {'num_samples': 1365, 'avg_sentence1_len': 115.17289377289377, 'avg_sentence2_len': 55.120879120879124, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'fr': {'num_samples': 1365, 'avg_sentence1_len': 121.75897435897436, 'avg_sentence2_len': 59.08864468864469, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'hi': {'num_samples': 1365, 'avg_sentence1_len': 105.06446886446886, 'avg_sentence2_len': 50.44395604395604, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'ru': {'num_samples': 1365, 'avg_sentence1_len': 109.74725274725274, 'avg_sentence2_len': 52.26886446886447, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'sw': {'num_samples': 1365, 'avg_sentence1_len': 104.32234432234432, 'avg_sentence2_len': 49.87692307692308, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'th': {'num_samples': 1365, 'avg_sentence1_len': 97.28498168498169, 'avg_sentence2_len': 43.843223443223444, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'tr': {'num_samples': 1365, 'avg_sentence1_len': 102.96630036630036, 'avg_sentence2_len': 49.63809523809524, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'vi': {'num_samples': 1365, 'avg_sentence1_len': 112.26373626373626, 'avg_sentence2_len': 52.432967032967035, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'zh': {'num_samples': 1365, 'avg_sentence1_len': 33.41098901098901, 'avg_sentence2_len': 15.846886446886447, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}}}} | -| [XNLIV2](https://arxiv.org/pdf/2301.06527) (Upadhyay et al., 2023) | ['asm', 'ben', 'bho', 'ell', 'guj', 'kan', 'mar', 'ory', 'pan', 'rus', 'san', 'tam', 'tur'] | PairClassification | s2s | [Non-fiction, Fiction, Government, Written] | {'test': 5010} | {'test': 80.06} | -| [XPQARetrieval](https://arxiv.org/abs/2305.09249) (Shen et al., 2023) | ['ara', 'cmn', 'deu', 'eng', 'fra', 'hin', 'ita', 'jpn', 'kor', 'pol', 'por', 'spa', 'tam'] | Retrieval | s2p | [Reviews, Written] | {'test': 19801} | {'test': {'ara-ara': {'average_document_length': 61.88361204013378, 'average_query_length': 29.688, 'num_documents': 1495, 'num_queries': 750, 'average_relevant_docs_per_query': 2.004}, 'eng-ara': {'average_document_length': 125.26940639269407, 'average_query_length': 29.688, 'num_documents': 1533, 'num_queries': 750, 'average_relevant_docs_per_query': 2.058666666666667}, 'ara-eng': {'average_document_length': 61.88361204013378, 'average_query_length': 39.5188679245283, 'num_documents': 1495, 'num_queries': 742, 'average_relevant_docs_per_query': 2.024258760107817}, 'deu-deu': {'average_document_length': 69.54807692307692, 'average_query_length': 55.51827676240209, 'num_documents': 1248, 'num_queries': 766, 'average_relevant_docs_per_query': 1.6318537859007833}, 'eng-deu': {'average_document_length': 115.77118078719145, 'average_query_length': 55.51827676240209, 'num_documents': 1499, 'num_queries': 766, 'average_relevant_docs_per_query': 1.9634464751958225}, 'deu-eng': {'average_document_length': 69.54807692307692, 'average_query_length': 51.88903394255875, 'num_documents': 1248, 'num_queries': 766, 'average_relevant_docs_per_query': 1.6318537859007833}, 'spa-spa': {'average_document_length': 68.27511591962906, 'average_query_length': 46.711223203026485, 'num_documents': 1941, 'num_queries': 793, 'average_relevant_docs_per_query': 2.4489281210592684}, 'eng-spa': {'average_document_length': 123.43698347107438, 'average_query_length': 46.711223203026485, 'num_documents': 1936, 'num_queries': 793, 'average_relevant_docs_per_query': 2.472887767969735}, 'spa-eng': {'average_document_length': 68.27511591962906, 'average_query_length': 47.21059268600252, 'num_documents': 1941, 'num_queries': 793, 'average_relevant_docs_per_query': 2.4489281210592684}, 'fra-fra': {'average_document_length': 76.99354005167959, 'average_query_length': 56.0520694259012, 'num_documents': 1548, 'num_queries': 749, 'average_relevant_docs_per_query': 2.069425901201602}, 'eng-fra': {'average_document_length': 137.31242532855435, 'average_query_length': 56.0520694259012, 'num_documents': 1674, 'num_queries': 749, 'average_relevant_docs_per_query': 2.248331108144192}, 'fra-eng': {'average_document_length': 76.99354005167959, 'average_query_length': 49.58744993324433, 'num_documents': 1548, 'num_queries': 749, 'average_relevant_docs_per_query': 2.069425901201602}, 'hin-hin': {'average_document_length': 47.20783373301359, 'average_query_length': 33.47783783783784, 'num_documents': 1251, 'num_queries': 925, 'average_relevant_docs_per_query': 1.3902702702702703}, 'eng-hin': {'average_document_length': 106.67662682602922, 'average_query_length': 33.47783783783784, 'num_documents': 1506, 'num_queries': 925, 'average_relevant_docs_per_query': 1.8054054054054054}, 'hin-eng': {'average_document_length': 47.20783373301359, 'average_query_length': 34.98574561403509, 'num_documents': 1251, 'num_queries': 912, 'average_relevant_docs_per_query': 1.4100877192982457}, 'ita-ita': {'average_document_length': 59.778301886792455, 'average_query_length': 49.14932126696833, 'num_documents': 1272, 'num_queries': 663, 'average_relevant_docs_per_query': 1.9245852187028658}, 'eng-ita': {'average_document_length': 123.07302075326672, 'average_query_length': 49.14932126696833, 'num_documents': 1301, 'num_queries': 663, 'average_relevant_docs_per_query': 1.9849170437405732}, 'ita-eng': {'average_document_length': 59.778301886792455, 'average_query_length': 49.040723981900456, 'num_documents': 1272, 'num_queries': 663, 'average_relevant_docs_per_query': 1.9245852187028658}, 'jpn-jpn': {'average_document_length': 41.030605871330415, 'average_query_length': 23.296969696969697, 'num_documents': 1601, 'num_queries': 825, 'average_relevant_docs_per_query': 1.9406060606060607}, 'eng-jpn': {'average_document_length': 126.2647564469914, 'average_query_length': 23.296969696969697, 'num_documents': 1745, 'num_queries': 825, 'average_relevant_docs_per_query': 2.1187878787878787}, 'jpn-eng': {'average_document_length': 41.030605871330415, 'average_query_length': 51.416058394160586, 'num_documents': 1601, 'num_queries': 822, 'average_relevant_docs_per_query': 1.9476885644768855}, 'kor-kor': {'average_document_length': 31.22722159730034, 'average_query_length': 21.81804281345566, 'num_documents': 889, 'num_queries': 654, 'average_relevant_docs_per_query': 1.5642201834862386}, 'eng-kor': {'average_document_length': 112.41231822070145, 'average_query_length': 21.81804281345566, 'num_documents': 1169, 'num_queries': 654, 'average_relevant_docs_per_query': 1.952599388379205}, 'kor-eng': {'average_document_length': 31.22722159730034, 'average_query_length': 43.9527687296417, 'num_documents': 889, 'num_queries': 614, 'average_relevant_docs_per_query': 1.6661237785016287}, 'pol-pol': {'average_document_length': 50.66814439518683, 'average_query_length': 53.72101910828025, 'num_documents': 1579, 'num_queries': 785, 'average_relevant_docs_per_query': 2.080254777070064}, 'eng-pol': {'average_document_length': 112.96919566457501, 'average_query_length': 53.72101910828025, 'num_documents': 1753, 'num_queries': 785, 'average_relevant_docs_per_query': 2.385987261146497}, 'pol-eng': {'average_document_length': 50.66814439518683, 'average_query_length': 54.1994851994852, 'num_documents': 1579, 'num_queries': 777, 'average_relevant_docs_per_query': 2.101673101673102}, 'por-por': {'average_document_length': 75.9845869297164, 'average_query_length': 42.58875, 'num_documents': 1622, 'num_queries': 800, 'average_relevant_docs_per_query': 2.14}, 'eng-por': {'average_document_length': 111.42525930445393, 'average_query_length': 42.58875, 'num_documents': 1639, 'num_queries': 800, 'average_relevant_docs_per_query': 2.21875}, 'por-eng': {'average_document_length': 75.9845869297164, 'average_query_length': 46.57967377666248, 'num_documents': 1622, 'num_queries': 797, 'average_relevant_docs_per_query': 2.148055207026349}, 'tam-tam': {'average_document_length': 64.89019607843137, 'average_query_length': 33.267263427109974, 'num_documents': 1275, 'num_queries': 782, 'average_relevant_docs_per_query': 1.6994884910485935}, 'eng-tam': {'average_document_length': 96.96361185983828, 'average_query_length': 33.267263427109974, 'num_documents': 1484, 'num_queries': 782, 'average_relevant_docs_per_query': 2.0255754475703327}, 'tam-eng': {'average_document_length': 64.89019607843137, 'average_query_length': 34.777633289986994, 'num_documents': 1275, 'num_queries': 769, 'average_relevant_docs_per_query': 1.728218465539662}, 'cmn-cmn': {'average_document_length': 20.958944281524925, 'average_query_length': 12.21116504854369, 'num_documents': 1705, 'num_queries': 824, 'average_relevant_docs_per_query': 2.0716019417475726}, 'eng-cmn': {'average_document_length': 108.31593874078276, 'average_query_length': 12.21116504854369, 'num_documents': 1763, 'num_queries': 824, 'average_relevant_docs_per_query': 2.2633495145631066}, 'cmn-eng': {'average_document_length': 20.958944281524925, 'average_query_length': 41.24390243902439, 'num_documents': 1705, 'num_queries': 820, 'average_relevant_docs_per_query': 2.0817073170731706}}} | -| [XQuADRetrieval](https://huggingface.co/datasets/xquad) (Mikel Artetxe, 2019) | ['arb', 'deu', 'ell', 'eng', 'hin', 'ron', 'rus', 'spa', 'tha', 'tur', 'vie', 'zho'] | Retrieval | s2p | [Web, Written] | {'test': 1190} | {'validation': {'ar': {'average_document_length': 683.4666666666667, 'average_query_length': 53.327993254637434, 'num_documents': 240, 'num_queries': 1186, 'average_relevant_docs_per_query': 1.0}, 'de': {'average_document_length': 894.0666666666667, 'average_query_length': 69.04318374259103, 'num_documents': 240, 'num_queries': 1181, 'average_relevant_docs_per_query': 1.0}, 'el': {'average_document_length': 894.3791666666667, 'average_query_length': 68.61317567567568, 'num_documents': 240, 'num_queries': 1184, 'average_relevant_docs_per_query': 1.0}, 'en': {'average_document_length': 784.8333333333334, 'average_query_length': 61.25063291139241, 'num_documents': 240, 'num_queries': 1185, 'average_relevant_docs_per_query': 1.0}, 'es': {'average_document_length': 883.8041666666667, 'average_query_length': 68.23817567567568, 'num_documents': 240, 'num_queries': 1184, 'average_relevant_docs_per_query': 1.0}, 'hi': {'average_document_length': 764.9416666666667, 'average_query_length': 59.684699915469146, 'num_documents': 240, 'num_queries': 1183, 'average_relevant_docs_per_query': 1.0}, 'ro': {'average_document_length': 878.4458333333333, 'average_query_length': 67.17229729729729, 'num_documents': 240, 'num_queries': 1184, 'average_relevant_docs_per_query': 1.0}, 'ru': {'average_document_length': 850.1875, 'average_query_length': 64.94261603375527, 'num_documents': 240, 'num_queries': 1185, 'average_relevant_docs_per_query': 1.0}, 'th': {'average_document_length': 736.7583333333333, 'average_query_length': 55.103389830508476, 'num_documents': 240, 'num_queries': 1180, 'average_relevant_docs_per_query': 1.0}, 'tr': {'average_document_length': 788.3, 'average_query_length': 60.876689189189186, 'num_documents': 240, 'num_queries': 1184, 'average_relevant_docs_per_query': 1.0}, 'vi': {'average_document_length': 803.9083333333333, 'average_query_length': 61.62859560067682, 'num_documents': 240, 'num_queries': 1182, 'average_relevant_docs_per_query': 1.0}, 'zh': {'average_document_length': 252.4, 'average_query_length': 18.460626587637595, 'num_documents': 240, 'num_queries': 1181, 'average_relevant_docs_per_query': 1.0}}} | -| [XStance](https://github.com/ZurichNLP/xstance) | ['deu', 'fra', 'ita'] | PairClassification | s2s | [Social, Written] | {'test': 2048} | {'test': 152.41} | -| [YahooAnswersTopicsClassification](https://huggingface.co/datasets/yahoo_answers_topics) (Zhang et al., 2015) | ['eng'] | Classification | s2s | [Web, Written] | {'test': 60000} | {'test': 346.35} | -| [YelpReviewFullClassification](https://arxiv.org/abs/1509.01626) (Zhang et al., 2015) | ['eng'] | Classification | s2s | [Reviews, Written] | {'test': 50000} | {} | -| [YueOpenriceReviewClassification](https://github.com/Christainx/Dataset_Cantonese_Openrice) (Xiang et al., 2019) | ['yue'] | Classification | s2s | [Reviews, Spoken] | {'test': 6161} | {'test': 173.0} | -| [indonli](https://link.springer.com/chapter/10.1007/978-3-030-41505-1_39) | ['ind'] | PairClassification | s2s | [Encyclopaedic, Web, News, Written] | {'test_expert': 2040} | {'test_expert': 145.88} | -| [mFollowIRCrossLingualInstructionRetrieval](https://neuclir.github.io/) (Weller et al., 2024) | ['eng', 'fas', 'rus', 'zho'] | Retrieval | s2p | [News, Written] | {'eng-fas': 80, 'eng-rus': 80, 'eng-zho': 86} | {'test': {'num_docs': 121635, 'num_queries': 123, 'average_document_length': 2331.0777818884367, 'average_query_length': 81.8780487804878, 'average_instruction_length': 389.9512195121951, 'average_changed_instruction_length': 450.5528455284553, 'average_relevant_docs_per_query': 10.30952380952381, 'average_top_ranked_per_query': 1024.3902439024391, 'hf_subset_descriptive_stats': {'eng-fas': {'num_docs': 41189, 'num_queries': 40, 'average_document_length': 3145.4990895627475, 'average_query_length': 80.075, 'average_instruction_length': 396.875, 'average_changed_instruction_length': 463.175, 'average_relevant_docs_per_query': 10.465116279069768, 'average_top_ranked_per_query': 1075}, 'eng-rus': {'num_docs': 39326, 'num_queries': 40, 'average_document_length': 2784.0813456746173, 'average_query_length': 81.875, 'average_instruction_length': 371.125, 'average_changed_instruction_length': 431.8, 'average_relevant_docs_per_query': 9.775, 'average_top_ranked_per_query': 1000}, 'eng-zho': {'num_docs': 41120, 'num_queries': 43, 'average_document_length': 1082.0501215953307, 'average_query_length': 83.55813953488372, 'average_instruction_length': 401.0232558139535, 'average_changed_instruction_length': 456.25581395348837, 'average_relevant_docs_per_query': 10.651162790697674, 'average_top_ranked_per_query': 1000}}}} | -| [mFollowIRInstructionRetrieval](https://neuclir.github.io/) (Weller et al., 2024) | ['fas', 'rus', 'zho'] | Retrieval | s2p | [News, Written] | {'fas': 80, 'rus': 80, 'zho': 86} | {'test': {'num_docs': 121635, 'num_queries': 123, 'average_document_length': 2331.0777818884367, 'average_query_length': 57.113821138211385, 'average_instruction_length': 281.0650406504065, 'average_changed_instruction_length': 326.9430894308943, 'average_relevant_docs_per_query': 10.30952380952381, 'average_top_ranked_per_query': 1024.3902439024391, 'hf_subset_descriptive_stats': {'fas': {'num_docs': 41189, 'num_queries': 40, 'average_document_length': 3145.4990895627475, 'average_query_length': 72.65, 'average_instruction_length': 358.925, 'average_changed_instruction_length': 415.325, 'average_relevant_docs_per_query': 10.465116279069768, 'average_top_ranked_per_query': 1075}, 'rus': {'num_docs': 39326, 'num_queries': 40, 'average_document_length': 2784.0813456746173, 'average_query_length': 77.5, 'average_instruction_length': 387, 'average_changed_instruction_length': 458, 'average_relevant_docs_per_query': 9.775, 'average_top_ranked_per_query': 1000}, 'zho': {'num_docs': 41120, 'num_queries': 43, 'average_document_length': 1082.0501215953307, 'average_query_length': 23.697674418604652, 'average_instruction_length': 110.09302325581395, 'average_changed_instruction_length': 122.81395348837209, 'average_relevant_docs_per_query': 10.651162790697674, 'average_top_ranked_per_query': 1000}}}} | +| [WikiClusteringP2P.v2](https://github.com/Rysias/wiki-clustering) | ['bos', 'cat', 'ces', 'dan', 'eus', 'glv', 'ilo', 'kur', 'lav', 'min', 'mlt', 'sco', 'sqi', 'wln'] | Clustering | p2p | [Encyclopaedic, Written] | None | None | +| [WikipediaRerankingMultilingual](https://huggingface.co/datasets/ellamind/wikipedia-2023-11-reranking-multilingual) | ['ben', 'bul', 'ces', 'dan', 'deu', 'eng', 'fas', 'fin', 'hin', 'ita', 'nld', 'nor', 'por', 'ron', 'srp', 'swe'] | Reranking | s2p | [Encyclopaedic, Written] | {'test': 24000} | {'test': {'num_samples': 24000, 'number_of_characters': 83866932, 'num_positive': 24000, 'num_negative': 192000, 'avg_query_len': 59.09, 'avg_positive_len': 385.45, 'avg_negative_len': 381.24, 'hf_subset_descriptive_stats': {'bg': {'num_samples': 1500, 'number_of_characters': 5145316, 'num_positive': 1500, 'num_negative': 12000, 'avg_query_len': 60.83, 'avg_positive_len': 375.89, 'avg_negative_len': 374.19}, 'bn': {'num_samples': 1500, 'number_of_characters': 5390581, 'num_positive': 1500, 'num_negative': 12000, 'avg_query_len': 47.27, 'avg_positive_len': 394.59, 'avg_negative_len': 393.98}, 'cs': {'num_samples': 1500, 'number_of_characters': 5079180, 'num_positive': 1500, 'num_negative': 12000, 'avg_query_len': 56.27, 'avg_positive_len': 383.84, 'avg_negative_len': 368.25}, 'da': {'num_samples': 1500, 'number_of_characters': 4746132, 'num_positive': 1500, 'num_negative': 12000, 'avg_query_len': 56.75, 'avg_positive_len': 351.68, 'avg_negative_len': 344.46}, 'de': {'num_samples': 1500, 'number_of_characters': 5483592, 'num_positive': 1500, 'num_negative': 12000, 'avg_query_len': 70.0, 'avg_positive_len': 391.54, 'avg_negative_len': 399.27}, 'en': {'num_samples': 1500, 'number_of_characters': 6217884, 'num_positive': 1500, 'num_negative': 12000, 'avg_query_len': 68.37, 'avg_positive_len': 451.73, 'avg_negative_len': 453.14}, 'fa': {'num_samples': 1500, 'number_of_characters': 4732619, 'num_positive': 1500, 'num_negative': 12000, 'avg_query_len': 48.67, 'avg_positive_len': 347.7, 'avg_negative_len': 344.84}, 'fi': {'num_samples': 1500, 'number_of_characters': 5209132, 'num_positive': 1500, 'num_negative': 12000, 'avg_query_len': 55.34, 'avg_positive_len': 394.71, 'avg_negative_len': 377.84}, 'hi': {'num_samples': 1500, 'number_of_characters': 5620959, 'num_positive': 1500, 'num_negative': 12000, 'avg_query_len': 50.78, 'avg_positive_len': 420.38, 'avg_negative_len': 409.52}, 'it': {'num_samples': 1500, 'number_of_characters': 5420496, 'num_positive': 1500, 'num_negative': 12000, 'avg_query_len': 70.05, 'avg_positive_len': 396.97, 'avg_negative_len': 393.33}, 'nl': {'num_samples': 1500, 'number_of_characters': 5169556, 'num_positive': 1500, 'num_negative': 12000, 'avg_query_len': 65.34, 'avg_positive_len': 380.79, 'avg_negative_len': 375.03}, 'pt': {'num_samples': 1500, 'number_of_characters': 5474356, 'num_positive': 1500, 'num_negative': 12000, 'avg_query_len': 65.12, 'avg_positive_len': 404.02, 'avg_negative_len': 397.55}, 'ro': {'num_samples': 1500, 'number_of_characters': 4796113, 'num_positive': 1500, 'num_negative': 12000, 'avg_query_len': 61.97, 'avg_positive_len': 346.71, 'avg_negative_len': 348.59}, 'sr': {'num_samples': 1500, 'number_of_characters': 5271732, 'num_positive': 1500, 'num_negative': 12000, 'avg_query_len': 55.67, 'avg_positive_len': 386.35, 'avg_negative_len': 384.06}, 'no': {'num_samples': 1500, 'number_of_characters': 5036586, 'num_positive': 1500, 'num_negative': 12000, 'avg_query_len': 55.29, 'avg_positive_len': 367.72, 'avg_negative_len': 366.84}, 'sv': {'num_samples': 1500, 'number_of_characters': 5072698, 'num_positive': 1500, 'num_negative': 12000, 'avg_query_len': 57.73, 'avg_positive_len': 372.59, 'avg_negative_len': 368.94}}}} | +| [WikipediaRetrievalMultilingual](https://huggingface.co/datasets/ellamind/wikipedia-2023-11-retrieval-multilingual-queries) | ['ben', 'bul', 'ces', 'dan', 'deu', 'eng', 'fas', 'fin', 'hin', 'ita', 'nld', 'nor', 'por', 'ron', 'srp', 'swe'] | Retrieval | s2p | [Encyclopaedic, Written] | None | None | +| [WinoGrande](https://winogrande.allenai.org/) (Xiao et al., 2024) | ['eng'] | Retrieval | s2s | [Encyclopaedic, Written] | None | None | +| [WisesightSentimentClassification](https://github.com/PyThaiNLP/wisesight-sentiment) | ['tha'] | Classification | s2s | [Social, News, Written] | None | None | +| XMarket (Bonab et al., 2021) | ['deu', 'eng', 'spa'] | Retrieval | s2p | | None | None | +| [XNLI](https://aclanthology.org/D18-1269/) (Conneau et al., 2018) | ['ara', 'bul', 'deu', 'ell', 'eng', 'fra', 'hin', 'rus', 'spa', 'swa', 'tha', 'tur', 'vie', 'zho'] | PairClassification | s2s | [Non-fiction, Fiction, Government, Written] | {'test': 19110, 'validation': 19110} | {'test': {'num_samples': 19110, 'number_of_characters': 2907145, 'avg_sentence1_len': 103.24, 'avg_sentence2_len': 48.89, 'unique_labels': 2, 'labels': {'0': {'count': 9562}, '1': {'count': 9548}}, 'hf_subset_descriptive_stats': {'ar': {'num_samples': 1365, 'number_of_characters': 179591, 'avg_sentence1_len': 89.57, 'avg_sentence2_len': 41.99, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'bg': {'num_samples': 1365, 'number_of_characters': 220646, 'avg_sentence1_len': 110.02, 'avg_sentence2_len': 51.63, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'de': {'num_samples': 1365, 'number_of_characters': 241224, 'avg_sentence1_len': 119.93, 'avg_sentence2_len': 56.79, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'el': {'num_samples': 1365, 'number_of_characters': 240222, 'avg_sentence1_len': 119.05, 'avg_sentence2_len': 56.93, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'en': {'num_samples': 1365, 'number_of_characters': 212223, 'avg_sentence1_len': 105.67, 'avg_sentence2_len': 49.8, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'es': {'num_samples': 1365, 'number_of_characters': 232207, 'avg_sentence1_len': 115.43, 'avg_sentence2_len': 54.68, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'fr': {'num_samples': 1365, 'number_of_characters': 245259, 'avg_sentence1_len': 121.1, 'avg_sentence2_len': 58.58, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'hi': {'num_samples': 1365, 'number_of_characters': 211312, 'avg_sentence1_len': 104.63, 'avg_sentence2_len': 50.17, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'ru': {'num_samples': 1365, 'number_of_characters': 222797, 'avg_sentence1_len': 110.77, 'avg_sentence2_len': 52.45, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'sw': {'num_samples': 1365, 'number_of_characters': 210103, 'avg_sentence1_len': 104.44, 'avg_sentence2_len': 49.48, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'th': {'num_samples': 1365, 'number_of_characters': 192788, 'avg_sentence1_len': 96.69, 'avg_sentence2_len': 44.54, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'tr': {'num_samples': 1365, 'number_of_characters': 208658, 'avg_sentence1_len': 103.68, 'avg_sentence2_len': 49.19, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'vi': {'num_samples': 1365, 'number_of_characters': 223549, 'avg_sentence1_len': 111.31, 'avg_sentence2_len': 52.46, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'zh': {'num_samples': 1365, 'number_of_characters': 66566, 'avg_sentence1_len': 33.04, 'avg_sentence2_len': 15.73, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}}}, 'validation': {'num_samples': 19110, 'number_of_characters': 2909058, 'avg_sentence1_len': 103.21, 'avg_sentence2_len': 49.02, 'unique_labels': 2, 'labels': {'0': {'count': 9562}, '1': {'count': 9548}}, 'hf_subset_descriptive_stats': {'ar': {'num_samples': 1365, 'number_of_characters': 177355, 'avg_sentence1_len': 88.32, 'avg_sentence2_len': 41.61, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'bg': {'num_samples': 1365, 'number_of_characters': 219988, 'avg_sentence1_len': 109.2, 'avg_sentence2_len': 51.97, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'de': {'num_samples': 1365, 'number_of_characters': 241852, 'avg_sentence1_len': 119.81, 'avg_sentence2_len': 57.37, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'el': {'num_samples': 1365, 'number_of_characters': 241275, 'avg_sentence1_len': 119.88, 'avg_sentence2_len': 56.88, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'en': {'num_samples': 1365, 'number_of_characters': 212384, 'avg_sentence1_len': 105.72, 'avg_sentence2_len': 49.88, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'es': {'num_samples': 1365, 'number_of_characters': 232451, 'avg_sentence1_len': 115.17, 'avg_sentence2_len': 55.12, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'fr': {'num_samples': 1365, 'number_of_characters': 246857, 'avg_sentence1_len': 121.76, 'avg_sentence2_len': 59.09, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'hi': {'num_samples': 1365, 'number_of_characters': 212269, 'avg_sentence1_len': 105.06, 'avg_sentence2_len': 50.44, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'ru': {'num_samples': 1365, 'number_of_characters': 221152, 'avg_sentence1_len': 109.75, 'avg_sentence2_len': 52.27, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'sw': {'num_samples': 1365, 'number_of_characters': 210482, 'avg_sentence1_len': 104.32, 'avg_sentence2_len': 49.88, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'th': {'num_samples': 1365, 'number_of_characters': 192640, 'avg_sentence1_len': 97.28, 'avg_sentence2_len': 43.84, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'tr': {'num_samples': 1365, 'number_of_characters': 208305, 'avg_sentence1_len': 102.97, 'avg_sentence2_len': 49.64, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'vi': {'num_samples': 1365, 'number_of_characters': 224811, 'avg_sentence1_len': 112.26, 'avg_sentence2_len': 52.43, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}, 'zh': {'num_samples': 1365, 'number_of_characters': 67237, 'avg_sentence1_len': 33.41, 'avg_sentence2_len': 15.85, 'unique_labels': 2, 'labels': {'0': {'count': 683}, '1': {'count': 682}}}}}} | +| [XNLIV2](https://arxiv.org/pdf/2301.06527) (Upadhyay et al., 2023) | ['asm', 'ben', 'bho', 'ell', 'guj', 'kan', 'mar', 'ory', 'pan', 'rus', 'san', 'tam', 'tur'] | PairClassification | s2s | [Non-fiction, Fiction, Government, Written] | None | None | +| [XPQARetrieval](https://arxiv.org/abs/2305.09249) (Shen et al., 2023) | ['ara', 'cmn', 'deu', 'eng', 'fra', 'hin', 'ita', 'jpn', 'kor', 'pol', 'por', 'spa', 'tam'] | Retrieval | s2p | [Reviews, Written] | None | None | +| [XQuADRetrieval](https://huggingface.co/datasets/xquad) (Mikel Artetxe, 2019) | ['arb', 'deu', 'ell', 'eng', 'hin', 'ron', 'rus', 'spa', 'tha', 'tur', 'vie', 'zho'] | Retrieval | s2p | [Web, Written] | None | None | +| [XStance](https://github.com/ZurichNLP/xstance) | ['deu', 'fra', 'ita'] | PairClassification | s2s | [Social, Written] | None | None | +| [YahooAnswersTopicsClassification](https://huggingface.co/datasets/yahoo_answers_topics) (Zhang et al., 2015) | ['eng'] | Classification | s2s | [Web, Written] | None | None | +| [YelpReviewFullClassification](https://arxiv.org/abs/1509.01626) (Zhang et al., 2015) | ['eng'] | Classification | s2s | [Reviews, Written] | None | None | +| [YueOpenriceReviewClassification](https://github.com/Christainx/Dataset_Cantonese_Openrice) (Xiang et al., 2019) | ['yue'] | Classification | s2s | [Reviews, Spoken] | None | None | +| [indonli](https://link.springer.com/chapter/10.1007/978-3-030-41505-1_39) | ['ind'] | PairClassification | s2s | [Encyclopaedic, Web, News, Written] | None | None | +| [mFollowIRCrossLingualInstructionRetrieval](https://neuclir.github.io/) (Weller et al., 2024) | ['eng', 'fas', 'rus', 'zho'] | Retrieval | s2p | [News, Written] | {'test': 121758} | {'test': {'num_samples': 121758, 'num_docs': 121635, 'num_queries': 123, 'number_of_characters': 283654099, 'average_document_length': 2331.08, 'average_query_length': 81.88, 'average_instruction_length': 389.95, 'average_changed_instruction_length': 450.55, 'average_relevant_docs_per_query': 10.43, 'average_top_ranked_per_query': 1000.0, 'hf_subset_descriptive_stats': {'eng-fas': {'num_samples': 41229, 'num_docs': 41189, 'num_queries': 40, 'number_of_characters': 129597567, 'average_document_length': 3145.5, 'average_query_length': 80.08, 'average_instruction_length': 396.88, 'average_changed_instruction_length': 463.18, 'average_relevant_docs_per_query': 10.85, 'average_top_ranked_per_query': 1000.0}, 'eng-rus': {'num_samples': 39366, 'num_docs': 39326, 'num_queries': 40, 'number_of_characters': 109522175, 'average_document_length': 2784.08, 'average_query_length': 81.88, 'average_instruction_length': 371.12, 'average_changed_instruction_length': 431.8, 'average_relevant_docs_per_query': 9.78, 'average_top_ranked_per_query': 1000.0}, 'eng-zho': {'num_samples': 41163, 'num_docs': 41120, 'num_queries': 43, 'number_of_characters': 44534357, 'average_document_length': 1082.05, 'average_query_length': 83.56, 'average_instruction_length': 401.02, 'average_changed_instruction_length': 456.26, 'average_relevant_docs_per_query': 10.65, 'average_top_ranked_per_query': 1000.0}}}} | +| [mFollowIRInstructionRetrieval](https://neuclir.github.io/) (Weller et al., 2024) | ['fas', 'rus', 'zho'] | Retrieval | s2p | [News, Written] | {'test': 121758} | {'test': {'num_samples': 121758, 'num_docs': 121635, 'num_queries': 123, 'number_of_characters': 283622456, 'average_document_length': 2331.08, 'average_query_length': 57.11, 'average_instruction_length': 281.07, 'average_changed_instruction_length': 326.94, 'average_relevant_docs_per_query': 10.43, 'average_top_ranked_per_query': 1000.0, 'hf_subset_descriptive_stats': {'fas': {'num_samples': 41229, 'num_docs': 41189, 'num_queries': 40, 'number_of_characters': 129593838, 'average_document_length': 3145.5, 'average_query_length': 72.65, 'average_instruction_length': 358.93, 'average_changed_instruction_length': 415.32, 'average_relevant_docs_per_query': 10.85, 'average_top_ranked_per_query': 1000.0}, 'rus': {'num_samples': 39366, 'num_docs': 39326, 'num_queries': 40, 'number_of_characters': 109523683, 'average_document_length': 2784.08, 'average_query_length': 77.5, 'average_instruction_length': 387.0, 'average_changed_instruction_length': 458.0, 'average_relevant_docs_per_query': 9.78, 'average_top_ranked_per_query': 1000.0}, 'zho': {'num_samples': 41163, 'num_docs': 41120, 'num_queries': 43, 'number_of_characters': 44504935, 'average_document_length': 1082.05, 'average_query_length': 23.7, 'average_instruction_length': 110.09, 'average_changed_instruction_length': 122.81, 'average_relevant_docs_per_query': 10.65, 'average_top_ranked_per_query': 1000.0}}}} |