[{"_id":"67d6c9b5ba56e14eeb14fab6","id":"mueller91/MLAAD","author":"mueller91","disabled":false,"gated":"auto","lastModified":"2026-08-28T12:53:44.000Z","likes":41,"trendingScore":3,"private":false,"sha":"30c3dec763fa4803f11c2df1557d3b0828395366","description":"\n  \n\n\n\n\t\n\t\t\n\t\n\t\n\t\tIntroduction\n\t\n\nWelcome to MLAAD: The Multi-Language Audio Anti-Spoofing Dataset -- a dataset to train, test and evaluate audio deepfake detection. See\nthe paper for more information.\n\n\t\n\t\t\n\t\n\t\n\t\tLicense\n\t\n\nMLAAD is published strictly for non-commercial academic research use, under the CC-BY-NC 4.0 license. Commercial use is not permitted.\n\n\t\n\t\t\n\t\n\t\n\t\tBibtex\n\t\n\nIf you use this dataset, please consider citing it as follows.\n@article{muller2024mlaad,\n  title={MLAAD: The… See the full description on the dataset page: https://huggingface.co/datasets/mueller91/MLAAD.","downloads":41836,"tags":["task_categories:audio-classification","language:en","language:de","language:fr","language:es","language:uk","language:pl","language:ru","language:it","license:cc-by-nc-4.0","size_categories:100K<n<1M","modality:audio","arxiv:2401.09512","region:us","audio","deepfake","audio-deepfake-detection","anti-spoofing","voice","voice-antispoofing","MLAAD"],"createdAt":"2025-03-16T12:53:09.000Z","key":""},{"_id":"66561c5d5b8ab1ed4f7a21af","id":"mlabonne/harmful_behaviors","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-06-04T10:45:47.000Z","likes":154,"trendingScore":2,"private":false,"sha":"01cead01398926d81f7c52bdb790ee8cf77ebba7","downloads":22885,"tags":["language:en","size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-05-28T18:03:09.000Z","key":""},{"_id":"62b4794beb8d44abbfeedc8f","id":"CShorten/ML-ArXiv-Papers","author":"CShorten","disabled":false,"gated":false,"lastModified":"2022-06-27T12:15:11.000Z","likes":71,"trendingScore":1,"private":false,"sha":"c878972daa0a5ec5f0d684354b6c8018f27d1316","description":"This dataset contains the subset of ArXiv papers with the \"cs.LG\" tag to indicate the paper is about Machine Learning.\nThe core dataset is filtered from the full ArXiv dataset hosted on Kaggle: https://www.kaggle.com/datasets/Cornell-University/arxiv. The original dataset contains roughly 2 million papers. This dataset contains roughly 100,000 papers following the category filtering.\nThe dataset is maintained by with requests to the ArXiv API.\nThe current iteration of the dataset only contains… See the full description on the dataset page: https://huggingface.co/datasets/CShorten/ML-ArXiv-Papers.","downloads":4770,"tags":["license:afl-3.0","size_categories:100K<n<1M","format:csv","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2022-06-23T14:31:39.000Z","key":""},{"_id":"670befa7623c91990f914eb6","id":"mlabonne/open-perfectblend","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-01-15T20:01:32.000Z","likes":183,"trendingScore":1,"private":false,"sha":"af60f3c18201652a83a93f46fcfee1b646ba3df7","description":"\n\n\t\n\t\t\n\t\n\t\n\t\t🎨 Open-PerfectBlend\n\t\n\nOpen-PerfectBlend is an open-source reproduction of the instruction dataset introduced in the paper \"The Perfect Blend: Redefining RLHF with Mixture of Judges\".\nIt's a solid general-purpose instruction dataset with chat, math, code, and instruction-following data.\n\n\t\n\t\t\n\t\n\t\n\t\tData source\n\t\n\n\nHere is the list of the datasets used in this mix:\n\n\t\n\t\t\nDataset\n# Samples\n\n\n\t\t\nmeta-math/MetaMathQA\n395,000\n\n\nopenbmb/UltraInteract_sft\n288,579… See the full description on the dataset page: https://huggingface.co/datasets/mlabonne/open-perfectblend.","downloads":4561,"tags":["license:apache-2.0","arxiv:2409.20370","region:us"],"createdAt":"2024-10-13T16:04:55.000Z","key":""},{"_id":"621ffdd236468d709f181ead","id":"cis-lmu/m_lama","author":"cis-lmu","disabled":false,"gated":false,"lastModified":"2025-05-14T08:05:50.000Z","likes":6,"trendingScore":0,"private":false,"sha":"426625cc2d078843e98117e8a183bff0904e8724","citation":"@article{kassner2021multilingual,\n  author    = {Nora Kassner and\n               Philipp Dufter and\n               Hinrich Sch{\\\"{u}}tze},\n  title     = {Multilingual {LAMA:} Investigating Knowledge in Multilingual Pretrained\n               Language Models},\n  journal   = {CoRR},\n  volume    = {abs/2102.00894},\n  year      = {2021},\n  url       = {https://arxiv.org/abs/2102.00894},\n  archivePrefix = {arXiv},\n  eprint    = {2102.00894},\n  timestamp = {Tue, 09 Feb 2021 13:35:56 +0100},\n  biburl    = {https://dblp.org/rec/journals/corr/abs-2102-00894.bib},\n  bibsource = {dblp computer science bibliography, https://dblp.org},\n  note      = {to appear in EACL2021}\n}","description":"mLAMA: a multilingual version of the LAMA benchmark (T-REx and GoogleRE) covering 53 languages.","downloads":253,"tags":["task_categories:question-answering","task_categories:text-classification","task_ids:open-domain-qa","task_ids:text-scoring","annotations_creators:crowdsourced","annotations_creators:expert-generated","annotations_creators:machine-generated","language_creators:crowdsourced","language_creators:expert-generated","language_creators:machine-generated","multilinguality:translation","source_datasets:extended|lama","language:af","language:ar","language:az","language:be","language:bg","language:bn","language:ca","language:ceb","language:cs","language:cy","language:da","language:de","language:el","language:en","language:es","language:et","language:eu","language:fa","language:fi","language:fr","language:ga","language:gl","language:he","language:hi","language:hr","language:hu","language:hy","language:id","language:it","language:ja","language:ka","language:ko","language:la","language:lt","language:lv","language:ms","language:nl","language:pl","language:pt","language:ro","language:ru","language:sk","language:sl","language:sq","language:sr","language:sv","language:ta","language:th","language:tr","language:uk","language:ur","language:vi","language:zh","license:cc-by-nc-sa-4.0","size_categories:100K<n<1M","arxiv:2102.00894","region:us","probing"],"createdAt":"2022-03-02T23:29:22.000Z","key":""},{"_id":"642f3db5654a3f7660010625","id":"bakhitovd/ML_arxiv","author":"bakhitovd","disabled":false,"gated":false,"lastModified":"2023-05-19T21:47:33.000Z","likes":2,"trendingScore":0,"private":false,"sha":"cab38dadfa71d230be0022ec0b0478b19357239d","description":"\n\t\n\t\t\n\t\tDataset Card for 'ML Articles Subset of Scientific Papers' Dataset\n\t\n\n\n\t\n\t\t\n\t\tDataset Summary\n\t\n\nThe dataset consists of 32,621 instances from the 'Scientific papers' dataset, a selection of scientific papers and summaries from ArXiv repository. This subset focuses on articles that are semantically, vocabulary-wise, structurally, and meaningfully closest to articles describing machine learning. This subset was created using sentence embeddings and K-means clustering.\n\n\t\n\t\t\n\t\tSupported… See the full description on the dataset page: https://huggingface.co/datasets/bakhitovd/ML_arxiv.","downloads":156,"tags":["task_categories:summarization","language:en","license:cc0-1.0","size_categories:10K<n<100K","format:json","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2023-04-06T21:46:29.000Z","key":""},{"_id":"6464b6ee06cd98685a9ff60e","id":"aalksii/ml-arxiv-papers","author":"aalksii","disabled":false,"gated":false,"lastModified":"2023-05-19T11:47:18.000Z","likes":8,"trendingScore":0,"private":false,"sha":"01067b176059af39ee388c0f6106e6fd7eaf1e19","description":"\n\t\n\t\t\n\t\tDataset Card for \"ml-arxiv-papers\"\n\t\n\nThis is a dataset containing ML ArXiv papers. The dataset is a version of the original one from CShorten, which is a part of the ArXiv papers dataset from Kaggle. \nThree steps are made to process the source data:\n\nuseless columns removal;\ntrain-test split;\n'\\n' removal and trimming spaces on sides of the text.\n\nMore Information needed\n","downloads":66,"tags":["task_categories:summarization","language:en","size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us","arxiv","ML"],"createdAt":"2023-05-17T11:13:50.000Z","key":""},{"_id":"64b916e62fccad9f5f12dc99","id":"mlabonne/CodeLlama-2-20k","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2023-07-30T10:45:33.000Z","likes":16,"trendingScore":0,"private":false,"sha":"6409e8b0a784f699f4980c2ac03f2dcda4452b9b","description":"\n\t\n\t\t\n\t\tCodeLlama-2-20k: A Llama 2 Version of CodeAlpaca\n\t\n\nThis dataset is the sahil2801/CodeAlpaca-20k dataset with the Llama 2 prompt format described here.\nHere is the code I used to format it:\nfrom datasets import load_dataset\n\n# Load the dataset\ndataset = load_dataset('sahil2801/CodeAlpaca-20k')\n\n# Define a function to merge the three columns into one\ndef merge_columns(example):\n    if example['input']:\n        merged = f\"<s>[INST] <<SYS>>\\nBelow is an instruction that describes a task… See the full description on the dataset page: https://huggingface.co/datasets/mlabonne/CodeLlama-2-20k.","downloads":106,"tags":["task_categories:text-generation","language:en","license:cc-by-4.0","size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us","code"],"createdAt":"2023-07-20T11:13:42.000Z","key":""},{"_id":"64bd30c62e66dc7b8bbeb356","id":"mlabonne/guanaco-llama2","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2023-07-26T14:49:17.000Z","likes":27,"trendingScore":0,"private":false,"sha":"9c91764721483249b6f9eab28a6b84a625713735","description":"\n\t\n\t\t\n\t\tGuanaco: Lazy Llama 2 Formatting\n\t\n\nThis is the excellent timdettmers/openassistant-guanaco dataset, processed to match Llama 2's prompt format as described in this article.\nUseful if you don't want to reformat it by yourself (e.g., using a script). It was designed for this article about fine-tuning a Llama 2 model in a Google Colab.\n","downloads":77,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2023-07-23T13:53:10.000Z","key":""},{"_id":"64bd424636eb058cd9f783a2","id":"mlabonne/guanaco-llama2-1k","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2023-08-25T16:49:41.000Z","likes":167,"trendingScore":0,"private":false,"sha":"27db19ecdc5fa7cd983280d1019ac2f1f96c96c1","description":"\n\t\n\t\t\n\t\tGuanaco-1k: Lazy Llama 2 Formatting\n\t\n\nThis is a subset (1000 samples) of the excellent timdettmers/openassistant-guanaco dataset, processed to match Llama 2's prompt format as described in this article. It was created using the following colab notebook.\nUseful if you don't want to reformat it by yourself (e.g., using a script). It was designed for this article about fine-tuning a Llama 2 (chat) model in a Google Colab.\n","downloads":1240,"tags":["size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2023-07-23T15:07:50.000Z","key":""},{"_id":"64cc14cea81988d0735bbb75","id":"mlabonne/alpagasus","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2023-08-03T21:18:52.000Z","likes":9,"trendingScore":0,"private":false,"sha":"39501309e1534293f03908b3a1e2f3fc5f4c0337","description":"\n\t\n\t\t\n\t\tAlpagasus (unofficial)\n\t\n\n📝 Paper | 📄 Blog | 💻 Code | 🤗 Model (unofficial)\nDataset of the unofficial implementation of AlpaGasus made by gpt4life. It is a filtered version of the original Alpaca dataset with GPT-4 acting as a judge.\n\n\nThe authors showed that models trained on this version with only 9k samples outperform models trained on the original 52k samples.\n","downloads":122,"tags":["task_categories:text-generation","license:gpl-3.0","size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","arxiv:2307.08701","region:us","alpaca","llama"],"createdAt":"2023-08-03T20:57:50.000Z","key":""},{"_id":"64d60a8e050438e3b945f6ee","id":"mlabonne/ministack-preferences","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2023-08-25T16:26:06.000Z","likes":3,"trendingScore":0,"private":false,"sha":"5496cfd7c6555884c6b1a0f3cf67a30905b67e8d","description":"\n\t\n\t\t\n\t\tMinistack-preferences\n\t\n\nSubset (1000 training samples and 1000 test samples) of the lvwerra/stack-exchange-paired dataset. The original dataset is really heavy and long to process, so hopefully this will help you to try RLHF a little faster.\n","downloads":21,"tags":["size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2023-08-11T10:16:46.000Z","key":""},{"_id":"64e8abce31442d6e30412de5","id":"mlabonne/Evol-Instruct-Python-26k","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2023-08-25T16:29:36.000Z","likes":15,"trendingScore":0,"private":false,"sha":"df386066db6cb020cce14ce43aabb1fb5719727e","description":"\n\t\n\t\t\n\t\tEvol-Instruct-Python-26k\n\t\n\nFiltered version of the nickrosh/Evol-Instruct-Code-80k-v1 dataset that only keeps Python code (26,588 samples). You can find a smaller version of it here mlabonne/Evol-Instruct-Python-1k.\nHere is the distribution of the number of tokens in each row (instruction + output) using Llama's tokenizer:\n\n","downloads":8490,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2023-08-25T13:25:34.000Z","key":""},{"_id":"64e8d6a7f494f8b2a0475477","id":"mlabonne/Evol-Instruct-Python-1k","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2023-08-25T16:31:50.000Z","likes":9,"trendingScore":0,"private":false,"sha":"670687a18d9a239632156cd7e0d606f65522ad5e","description":"\n\t\n\t\t\n\t\tEvol-Instruct-Python-1k\n\t\n\nSubset of the mlabonne/Evol-Instruct-Python-26k dataset with only 1000 samples.\nIt was made by filtering out a few rows (instruction + output) with more than 2048 tokens, and then by keeping the 1000 longest samples.\nHere is the distribution of the number of tokens in each row using Llama's tokenizer:\n\n","downloads":363,"tags":["size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2023-08-25T16:28:23.000Z","key":""},{"_id":"64fc6b0799123d7698aa5d8a","id":"mlabonne/medical-mqca-fr","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2023-09-09T16:18:56.000Z","likes":1,"trendingScore":0,"private":false,"sha":"31d03cb69e0393c087839245b6b91af41c7514a8","description":"\n\t\n\t\t\n\t\tDataset Card for \"medical-mqca-fr\"\n\t\n\nMore Information needed\n","downloads":39,"tags":["size_categories:1K<n<10K","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2023-09-09T12:54:31.000Z","key":""},{"_id":"64fc6c4b66a284440bae6cb0","id":"mlabonne/medical-cases-fr","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2023-09-09T16:24:18.000Z","likes":7,"trendingScore":0,"private":false,"sha":"b175f2a5c54359cc3db204a86e5f6b88016f95bb","description":"\n\t\n\t\t\n\t\tDataset Card for \"medical-cases-fr\"\n\t\n\nMore Information needed\n","downloads":43,"tags":["size_categories:1K<n<10K","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2023-09-09T12:59:55.000Z","key":""},{"_id":"64fc6c538c21ebb3db9d6b11","id":"mlabonne/MedText","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2023-09-09T16:24:24.000Z","likes":8,"trendingScore":0,"private":false,"sha":"c2348193978ac89b345f5df9315a789f9f5f7e26","description":"\n\t\n\t\t\n\t\tDataset Card for \"MedText\"\n\t\n\nMore Information needed\n","downloads":50,"tags":["size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2023-09-09T13:00:03.000Z","key":""},{"_id":"65566cec8076124b8d6a9425","id":"mlabonne/mini-platypus","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2023-11-17T16:39:56.000Z","likes":6,"trendingScore":0,"private":false,"sha":"f818ff74e728872b8212285a4f59a2dd701e0c73","downloads":23,"tags":["size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2023-11-16T19:26:36.000Z","key":""},{"_id":"65610240133861018472ec7c","id":"mlabonne/medical_meadow_medqa","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2023-11-24T21:01:07.000Z","likes":0,"trendingScore":0,"private":false,"sha":"e19208003691af5038bf606776d50461afe06d4d","downloads":81,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2023-11-24T20:06:24.000Z","key":""},{"_id":"656103328fb38d71f72371ef","id":"mlabonne/know_medical_dialogue_v2","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2023-11-24T20:10:27.000Z","likes":6,"trendingScore":0,"private":false,"sha":"ff40238b004c83574d39bce4bf4d758c8bc6003b","downloads":263,"tags":["size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2023-11-24T20:10:26.000Z","key":""},{"_id":"6561039b8631d43d2b8692d2","id":"mlabonne/MedQuad-MedicalQnADataset","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2023-11-24T20:45:49.000Z","likes":4,"trendingScore":0,"private":false,"sha":"2a7a9baeb5753eb4dc311e68d6a465c50469b1d9","downloads":29,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2023-11-24T20:12:11.000Z","key":""},{"_id":"65610446f5532ac1bdd69a13","id":"mlabonne/bactrian-fr","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2023-11-24T20:15:04.000Z","likes":2,"trendingScore":0,"private":false,"sha":"10392e74a4aac6f180cb5129296b139350698f61","downloads":25,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2023-11-24T20:15:02.000Z","key":""},{"_id":"6566763e7b5ed0735815c6cc","id":"mlabonne/chatml_dpo_pairs","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-04-11T09:09:03.000Z","likes":54,"trendingScore":0,"private":false,"sha":"83b4615b4141d3f1b3e68de21f62994d9cfd7e12","description":"\n\t\n\t\t\n\t\tChatML DPO Pairs\n\t\n\nThis is a preprocessed version of Intel/orca_dpo_pairs using the ChatML format.\nLike the original dataset, it contains 12k examples from Orca style dataset Open-Orca/OpenOrca.\nHere is the code used to preprocess it:\ndef chatml_format(example):\n    # Format system\n    if len(example['system']) > 0:\n        message = {\"role\": \"system\", \"content\": example['system']}\n        system = tokenizer.apply_chat_template([message], tokenize=False)\n    else:\n        system = \"\"… See the full description on the dataset page: https://huggingface.co/datasets/mlabonne/chatml_dpo_pairs.","downloads":122,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","arxiv:2306.02707","region:us","dpo"],"createdAt":"2023-11-28T23:22:38.000Z","key":""},{"_id":"656dbdde8a37acfa3f33db03","id":"open-llm-leaderboard-old/details_mlabonne__NeuralHermes-2.5-Mistral-7B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2023-12-04T11:54:49.000Z","likes":0,"trendingScore":0,"private":false,"sha":"e12a7202d71d391e2120c367a508665f09174bb8","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/NeuralHermes-2.5-Mistral-7B\n\t\n\n\n\t\n\t\t\n\t\tDataset Summary\n\t\n\nDataset automatically created during the evaluation run of model mlabonne/NeuralHermes-2.5-Mistral-7B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__NeuralHermes-2.5-Mistral-7B.","downloads":23,"tags":["region:us"],"createdAt":"2023-12-04T11:54:06.000Z","key":""},{"_id":"657f6650416635415f3614e7","id":"LeonJia/mlab-synopsys-commitpack-sample","author":"LeonJia","disabled":false,"gated":false,"lastModified":"2023-12-17T22:07:14.000Z","likes":0,"trendingScore":0,"private":false,"sha":"efdc285a827013092064a7bf5a1a93406f82e1a1","downloads":157,"tags":["size_categories:n<1K","format:json","modality:text","library:datasets","library:dask","library:mlcroissant","region:us"],"createdAt":"2023-12-17T21:21:20.000Z","key":""},{"_id":"6586c1a2fb9c2bdfaefddd9a","id":"atutej/m_lama","author":"atutej","disabled":false,"gated":false,"lastModified":"2024-03-14T19:36:35.000Z","likes":0,"trendingScore":0,"private":false,"sha":"bc27b0d37fd47eacd9934b9763d06f8baa1700e6","description":"Extension/Modification of the original m_lama dataset\n","downloads":271,"tags":["size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2023-12-23T11:16:50.000Z","key":""},{"_id":"658ea40d35c41262d634a905","id":"open-llm-leaderboard-old/details_mlabonne__GML-Mistral-merged-v1","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2023-12-29T10:49:08.000Z","likes":0,"trendingScore":0,"private":false,"sha":"5855fd551ae3a9cc871a540ad30046cf0e184d5d","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/GML-Mistral-merged-v1\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/GML-Mistral-merged-v1 on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__GML-Mistral-merged-v1.","downloads":18,"tags":["region:us"],"createdAt":"2023-12-29T10:48:45.000Z","key":""},{"_id":"658f062311f68f12ea0cd29b","id":"open-llm-leaderboard-old/details_mlabonne__NeuralPipe-7B-slerp","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-01-05T12:35:40.000Z","likes":0,"trendingScore":0,"private":false,"sha":"9f3ed49c439ebeb3ac77833bdd4b196364dc88ec","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/NeuralPipe-7B-slerp\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/NeuralPipe-7B-slerp on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 2 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__NeuralPipe-7B-slerp.","downloads":39,"tags":["region:us"],"createdAt":"2023-12-29T17:47:15.000Z","key":""},{"_id":"658f0f1852dc1046cac93552","id":"open-llm-leaderboard-old/details_mlabonne__NeuralPipe-7B-ties","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2023-12-29T18:25:50.000Z","likes":0,"trendingScore":0,"private":false,"sha":"0069f23a233e7bbf460b28a075db9700fe66d5a3","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/NeuralPipe-7B-ties\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/NeuralPipe-7B-ties on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__NeuralPipe-7B-ties.","downloads":26,"tags":["region:us"],"createdAt":"2023-12-29T18:25:28.000Z","key":""},{"_id":"659760a8929cd840d8eb71a3","id":"open-llm-leaderboard-old/details_mlabonne__NeuralHermes-2.5-Mistral-7B-laser","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-01-05T01:51:58.000Z","likes":0,"trendingScore":0,"private":false,"sha":"57c4bd83cca4f5dcf4932c6ae7d5b5ca5f4d89a4","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/NeuralHermes-2.5-Mistral-7B-laser\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/NeuralHermes-2.5-Mistral-7B-laser on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\"… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__NeuralHermes-2.5-Mistral-7B-laser.","downloads":40,"tags":["region:us"],"createdAt":"2024-01-05T01:51:36.000Z","key":""},{"_id":"659b227a21a74316430b58de","id":"open-llm-leaderboard-old/details_mlabonne__NeuralMarcoro14-7B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-01-07T22:15:41.000Z","likes":0,"trendingScore":0,"private":false,"sha":"9e7d482a7acfeb51db0d0a73b9cb88f23ae65c9a","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/NeuralMarcoro14-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/NeuralMarcoro14-7B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__NeuralMarcoro14-7B.","downloads":33,"tags":["region:us"],"createdAt":"2024-01-07T22:15:22.000Z","key":""},{"_id":"659f87acd8a112a4b9a1d326","id":"trl-internal-testing/mlabonne-chatml-dpo-pairs-copy","author":"trl-internal-testing","disabled":false,"gated":false,"lastModified":"2024-01-11T06:17:53.000Z","likes":0,"trendingScore":0,"private":false,"sha":"771ec252531a5c754c2037aaa3561f8489dcb0f7","description":"This is a copy and unmaintained version of mlabonne/chatml_dpo_pairs that we use in TRL CI for testing purpose. Please refer to the original dataset for usage and more details\n","downloads":22,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-01-11T06:16:12.000Z","key":""},{"_id":"65a44b096e52f8334000a979","id":"mlabonne/chessllm","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-03-08T16:17:15.000Z","likes":13,"trendingScore":0,"private":false,"sha":"3841467b8a6b528d897956b577e1c579816d6c7f","downloads":192,"tags":["size_categories:1M<n<10M","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-01-14T20:58:49.000Z","key":""},{"_id":"65a56d44215aabac48b94caf","id":"open-llm-leaderboard-old/details_mlabonne__Beagle14-7B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-01-15T17:37:33.000Z","likes":0,"trendingScore":0,"private":false,"sha":"e2ae5382069e8e3684db4958e303cf88ffcf2d2e","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Beagle14-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Beagle14-7B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__Beagle14-7B.","downloads":37,"tags":["region:us"],"createdAt":"2024-01-15T17:37:08.000Z","key":""},{"_id":"65a5d7972823ba72ed34021b","id":"open-llm-leaderboard-old/details_mlabonne__NeuralBeagle14-7B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-01-16T01:11:09.000Z","likes":0,"trendingScore":0,"private":false,"sha":"69efeef0beecbbda4065825025ab8ceebd0e7743","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/NeuralBeagle14-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/NeuralBeagle14-7B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__NeuralBeagle14-7B.","downloads":32,"tags":["region:us"],"createdAt":"2024-01-16T01:10:47.000Z","key":""},{"_id":"65a5daaee1e787bdecbed948","id":"open-llm-leaderboard-old/details_mlabonne__NeuralDaredevil-7B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-01-16T01:24:21.000Z","likes":1,"trendingScore":0,"private":false,"sha":"ce9afe4015caafa3f7341964a72bcf4bb630f0d5","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/NeuralDaredevil-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/NeuralDaredevil-7B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__NeuralDaredevil-7B.","downloads":21,"tags":["region:us"],"createdAt":"2024-01-16T01:23:58.000Z","key":""},{"_id":"65a7cfd5a6fe31817bbefb0f","id":"reza-alipour/m-landmark-generated","author":"reza-alipour","disabled":false,"gated":"manual","lastModified":"2024-01-17T15:36:11.000Z","likes":0,"trendingScore":0,"private":false,"sha":"3ef2a4a8880bd74c0517eb2ba4b17137ad33115e","downloads":6,"tags":["size_categories:1K<n<10K","format:parquet","modality:image","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-01-17T13:02:13.000Z","key":""},{"_id":"65b0c7589d4b4d79309f2723","id":"open-llm-leaderboard-old/details_mlabonne__Darewin-7B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-01-24T08:16:50.000Z","likes":0,"trendingScore":0,"private":false,"sha":"e278d2df7402723c0b46786cbacfecd26894e24f","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Darewin-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Darewin-7B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__Darewin-7B.","downloads":23,"tags":["region:us"],"createdAt":"2024-01-24T08:16:24.000Z","key":""},{"_id":"65b2d07bb0a5a381b6efe6d4","id":"open-llm-leaderboard-old/details_mlabonne__Darewin-7B-v2","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-01-25T21:20:23.000Z","likes":0,"trendingScore":0,"private":false,"sha":"031676f5c78b7fff27d602e62802afd5b248764a","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Darewin-7B-v2\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Darewin-7B-v2 on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__Darewin-7B-v2.","downloads":18,"tags":["region:us"],"createdAt":"2024-01-25T21:19:55.000Z","key":""},{"_id":"65ba4d1311162bf34fd39778","id":"mlabonne/distilabel-intel-orca-dpo-pairs","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-02-06T10:57:48.000Z","likes":3,"trendingScore":0,"private":false,"sha":"ff4a91c37ed9a301f5ff26532ec2019276229ceb","downloads":41,"tags":["size_categories:1K<n<10K","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","library:distilabel","region:us","synthetic","distilabel"],"createdAt":"2024-01-31T13:37:23.000Z","key":""},{"_id":"65bbda15b281d14183be2cef","id":"open-llm-leaderboard-old/details_mlabonne__NeuralDarewin-7B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-02-01T17:51:44.000Z","likes":0,"trendingScore":0,"private":false,"sha":"71b0e9709f4970c51bfee69ad6ae49aa19149f37","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/NeuralDarewin-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/NeuralDarewin-7B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__NeuralDarewin-7B.","downloads":49,"tags":["region:us"],"createdAt":"2024-02-01T17:51:17.000Z","key":""},{"_id":"65bc378f31e7709efb92fc38","id":"open-llm-leaderboard-old/details_mlabonne__OmniBeagle-7B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-02-02T00:30:44.000Z","likes":0,"trendingScore":0,"private":false,"sha":"8f3e05a74874797f9e7c08c3b898b4cda3a246eb","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/OmniBeagle-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/OmniBeagle-7B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__OmniBeagle-7B.","downloads":21,"tags":["region:us"],"createdAt":"2024-02-02T00:30:07.000Z","key":""},{"_id":"65ca1d88dc38a2858aa0fe73","id":"mlabonne/chatml-OpenHermes2.5-dpo-binarized-alpha","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-03-21T12:46:44.000Z","likes":20,"trendingScore":0,"private":false,"sha":"ad729cf9ac1541e0f5014143472e2b858dc163a3","description":"\n\t\n\t\t\n\t\tchatml-OpenHermes2.5-dpo-binarized-alpha\n\t\n\nThis is a DPO dataset based on argilla/OpenHermes2.5-dpo-binarized-alpha. It implements the following features:\n\nIntel format: you can directly use this dataset in Axolotl with \"type: chatml.intel\"\nFilter out low scores: removed samples with delta scores < 1 (530 in the training set, 66 in the test set).\nCurriculum learning: sort the dataset by the 'delta_score' column in descending order.\n\n\n\t\n\t\t\n\t\n\t\n\t\t💻 Code\n\t\n\nCode to reproduce this… See the full description on the dataset page: https://huggingface.co/datasets/mlabonne/chatml-OpenHermes2.5-dpo-binarized-alpha.","downloads":53,"tags":["size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-02-12T13:30:48.000Z","key":""},{"_id":"65ca423209b8df351f860d82","id":"mlabonne/distilabel-truthy-dpo-v0.1","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-02-12T19:59:01.000Z","likes":5,"trendingScore":0,"private":false,"sha":"834ee78e41f2d68c0161fa2a93f308fc4cbf8983","description":"\n\t\n\t\t\n\t\tdistilabel-truthy-dpo-v0.1\n\t\n\nA DPO dataset built with distilabel on top of Jon Durbin's jondurbin/truthy-dpo-v0.1 dataset.\nInterestingly, it swaps a lot of chosen and rejected answers.\n\n  \n    \n  \n","downloads":32,"tags":["size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-02-12T16:07:14.000Z","key":""},{"_id":"65ca5007d6c974694f9a10dd","id":"mlabonne/distilabel-truthy-dpo-v0.1-filtered","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-02-15T11:00:37.000Z","likes":2,"trendingScore":0,"private":false,"sha":"765be433c2ecadc75534f8a9cec10a6f6ad881b2","downloads":58,"tags":["language:en","size_categories:n<1K","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-02-12T17:06:15.000Z","key":""},{"_id":"65ca8257d6c974694faba6de","id":"mlabonne/truthy-dpo-v0.1","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-02-18T12:23:50.000Z","likes":1,"trendingScore":0,"private":false,"sha":"005895a6048b972fdbdd88287c38a491a22106d5","downloads":37,"tags":["language:en","size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-02-12T20:40:55.000Z","key":""},{"_id":"65cbe4b120448f0d128a5fd5","id":"open-llm-leaderboard-old/details_mlabonne__Monarch-7B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-02-13T21:53:14.000Z","likes":0,"trendingScore":0,"private":false,"sha":"1eaeb60c75e6c8af06d2a3990ec0429a0e6da419","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Monarch-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Monarch-7B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__Monarch-7B.","downloads":26,"tags":["region:us"],"createdAt":"2024-02-13T21:52:49.000Z","key":""},{"_id":"65cc9a0077a2d5c196d1add8","id":"open-llm-leaderboard-old/details_mlabonne__NeuralMonarch-7B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-02-14T10:46:47.000Z","likes":0,"trendingScore":0,"private":false,"sha":"616157785995d8fe7b8a76a797a2d9e5fcd53b7c","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/NeuralMonarch-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/NeuralMonarch-7B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__NeuralMonarch-7B.","downloads":26,"tags":["region:us"],"createdAt":"2024-02-14T10:46:24.000Z","key":""},{"_id":"65cce62c89823f76d2dede75","id":"open-llm-leaderboard-old/details_mlabonne__AlphaMonarch-7B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-02-14T16:11:46.000Z","likes":0,"trendingScore":0,"private":false,"sha":"b83c5f0c18d3f39f3ed13b25229476bb209377b4","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/AlphaMonarch-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/AlphaMonarch-7B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__AlphaMonarch-7B.","downloads":28,"tags":["region:us"],"createdAt":"2024-02-14T16:11:24.000Z","key":""},{"_id":"65d224b51e4ed7404591258f","id":"open-llm-leaderboard-old/details_Kukedlc__neuronal-7b-Mlab","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-02-18T15:39:55.000Z","likes":0,"trendingScore":0,"private":false,"sha":"c2c7dc08230b39289a6caf6fd4bf4afa2606b0f9","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of Kukedlc/neuronal-7b-Mlab\n\t\n\n\n\nDataset automatically created during the evaluation run of model Kukedlc/neuronal-7b-Mlab on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_Kukedlc__neuronal-7b-Mlab.","downloads":29,"tags":["region:us"],"createdAt":"2024-02-18T15:39:33.000Z","key":""},{"_id":"65d8ef76d6961d2df402fc5b","id":"open-llm-leaderboard-old/details_mlabonne__Gemmalpaca-2B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-02-23T19:18:37.000Z","likes":0,"trendingScore":0,"private":false,"sha":"0aa7bb98f50670daad47413734a28dbfcd6590d4","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Gemmalpaca-2B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Gemmalpaca-2B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__Gemmalpaca-2B.","downloads":26,"tags":["region:us"],"createdAt":"2024-02-23T19:18:14.000Z","key":""},{"_id":"65fc2aac45ed38e79a2a32cd","id":"mlabonne/ultrafeedback-binarized-preferences-cleaned","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-03-27T10:31:23.000Z","likes":4,"trendingScore":0,"private":false,"sha":"034503e608bdc7db8d54cf616e6333f33f3f0d76","description":"\n\t\n\t\t\n\t\tultrafeedback-binarized-preferences-cleaned\n\t\n\nThis is a DPO dataset based on argilla/ultrafeedback-binarized-preferences-cleaned. It implements the following features:\n\nIntel format: you can directly use this dataset in Axolotl with \"type: chatml.intel\"\nFilter out low scores: removed samples with delta scores < 1 (530 in the training set, 66 in the test set).\nCurriculum learning: sort the dataset by the 'delta_score' column in descending order.\n\n\n\t\n\t\t\n\t\n\t\n\t\t💻 Code\n\t\n\nCode to… See the full description on the dataset page: https://huggingface.co/datasets/mlabonne/ultrafeedback-binarized-preferences-cleaned.","downloads":18,"tags":["language:en","size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-03-21T12:40:12.000Z","key":""},{"_id":"65fd44274314eab9c7c7c0a1","id":"open-llm-leaderboard-old/details_mlabonne__UltraMerge-7B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-03-22T08:41:34.000Z","likes":0,"trendingScore":0,"private":false,"sha":"2fa1e6d1ccf921ce0e6cd297b29d207afc041868","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/UltraMerge-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/UltraMerge-7B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__UltraMerge-7B.","downloads":17,"tags":["region:us"],"createdAt":"2024-03-22T08:41:11.000Z","key":""},{"_id":"65fd774fb9f7730551954110","id":"open-llm-leaderboard-old/details_mlabonne__Beyonder-4x7B-v3","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-03-22T12:19:50.000Z","likes":0,"trendingScore":0,"private":false,"sha":"5814512adb53e95ca181637f188a090c2ad63d5b","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Beyonder-4x7B-v3\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Beyonder-4x7B-v3 on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__Beyonder-4x7B-v3.","downloads":18,"tags":["region:us"],"createdAt":"2024-03-22T12:19:27.000Z","key":""},{"_id":"65fdbb064de2daf8dc4afa9c","id":"open-llm-leaderboard-old/details_mlabonne__Mistralpaca-7B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-03-22T17:08:47.000Z","likes":0,"trendingScore":0,"private":false,"sha":"2552c359c4f8b26d3caf5e8b5c0e861c19e9f90d","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Mistralpaca-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Mistralpaca-7B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__Mistralpaca-7B.","downloads":32,"tags":["region:us"],"createdAt":"2024-03-22T17:08:22.000Z","key":""},{"_id":"65fdbe621004cb321ce1282b","id":"open-llm-leaderboard-old/details_mlabonne__FrankenMonarch-7B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-03-22T17:23:08.000Z","likes":0,"trendingScore":0,"private":false,"sha":"37fe466380e73e9e966ca1c1fc9b74fb30640435","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/FrankenMonarch-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/FrankenMonarch-7B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__FrankenMonarch-7B.","downloads":48,"tags":["region:us"],"createdAt":"2024-03-22T17:22:42.000Z","key":""},{"_id":"660c43df01eba685240aa742","id":"open-llm-leaderboard-old/details_mlabonne__Zebrafish-7B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-04-02T17:44:48.000Z","likes":0,"trendingScore":0,"private":false,"sha":"5755c722f633d0b71a69168759b55602aa0a917a","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Zebrafish-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Zebrafish-7B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__Zebrafish-7B.","downloads":34,"tags":["region:us"],"createdAt":"2024-04-02T17:43:59.000Z","key":""},{"_id":"661e98951ef8d816b65715cd","id":"mlabonne/synthetic_text_to_sql-ShareGPT","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-04-17T10:38:33.000Z","likes":4,"trendingScore":0,"private":false,"sha":"9548c84635c492f8473de9f96f2c8d0f0933904e","description":"\n\t\n\t\t\n\t\tsynthetic_text_to_sql\n\t\n\nShareGPT version of gretelai/synthetic_text_to_sql using the following code:\nfrom datasets import load_dataset, DatasetDict\n\n# Load the dataset\ndataset = load_dataset('gretelai/synthetic_text_to_sql', split='all')\n\ndef format_sample(sample):\n    conversations = [\n        {\n            \"from\": \"human\",\n            \"value\": f\"{sample['sql_context']}\\n\\n{sample['sql_prompt']}\"\n        },\n        {\n            \"from\": \"gpt\",\n            \"value\":… See the full description on the dataset page: https://huggingface.co/datasets/mlabonne/synthetic_text_to_sql-ShareGPT.","downloads":10,"tags":["size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-04-16T15:26:13.000Z","key":""},{"_id":"661faaedcc26dfa68e555b3c","id":"mlabonne/WizardLM_evol_instruct_70k-ShareGPT","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-04-17T10:56:49.000Z","likes":4,"trendingScore":0,"private":false,"sha":"8be09eec1f4315bf34881ce91207a61f288afe86","downloads":17,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-04-17T10:56:45.000Z","key":""},{"_id":"661fac22809b1dc3d4c5197b","id":"mlabonne/WizardLM_evol_instruct_v2_196K-ShareGPT","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-04-17T11:02:03.000Z","likes":3,"trendingScore":0,"private":false,"sha":"a49141af338fd4bd7ed1656a13e1d06c4379f283","downloads":5,"tags":["size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-04-17T11:01:54.000Z","key":""},{"_id":"662005a74360f44332b11379","id":"mlabonne/orpo-dpo-mix-40k","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-10-17T21:44:52.000Z","likes":309,"trendingScore":0,"private":false,"sha":"0f72511202b8f093e9be60e1683d84b046062e36","description":"\n\t\n\t\t\n\t\tORPO-DPO-mix-40k v1.2\n\t\n\n\nThis dataset is designed for ORPO or DPO training.\nSee Fine-tune Llama 3 with ORPO for more information about how to use it.\nIt is a combination of the following high-quality DPO datasets:\n\nargilla/Capybara-Preferences: highly scored chosen answers >=5 (7,424 samples)argilla/distilabel-intel-orca-dpo-pairs: highly scored chosen answers >=9, not in GSM8K (2,299 samples)\nargilla/ultrafeedback-binarized-preferences-cleaned: highly scored chosen answers >=5 (22… See the full description on the dataset page: https://huggingface.co/datasets/mlabonne/orpo-dpo-mix-40k.","downloads":1096,"tags":["task_categories:text-generation","language:en","license:apache-2.0","size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us","dpo","rlhf","preference","orpo"],"createdAt":"2024-04-17T17:23:51.000Z","key":""},{"_id":"6624167ac1b44fa73584bf9e","id":"open-llm-leaderboard-old/details_mlabonne__Chimera-8B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-04-20T19:25:00.000Z","likes":0,"trendingScore":0,"private":false,"sha":"5078a10d93037e0a9b77d8a0487d1995aa063bb2","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Chimera-8B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Chimera-8B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__Chimera-8B.","downloads":19,"tags":["region:us"],"createdAt":"2024-04-20T19:24:42.000Z","key":""},{"_id":"6624e7ec44f862018ffa0715","id":"hanyueshf/ml-arxiv-papers-qa","author":"hanyueshf","disabled":false,"gated":false,"lastModified":"2025-02-12T12:22:03.000Z","likes":6,"trendingScore":0,"private":false,"sha":"6a93d7a9beac46c1d74cb28811afaa0dde18e433","description":"This ML Q&A dataset contains 43,713 samples, where each includes three fields - question, context(title + abstract) and answer. \nIt is created based on the original dataset aalksii/ml-arxiv-papers, which contains the titles and abstracts of ML ArXiv papers. \nTo create question-answer pairs, the gpt-3.5-turbo API is called with the following prompt:messages = [\n    {\"role\": \"system\", \"content\": \"You are a helpful assistant.\"},\n    {\"role\": \"user\", \"content\": f\"Title: \"{title}\". Abstract:… See the full description on the dataset page: https://huggingface.co/datasets/hanyueshf/ml-arxiv-papers-qa.","downloads":38,"tags":["task_categories:question-answering","license:mit","size_categories:10K<n<100K","format:json","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-04-21T10:18:20.000Z","key":""},{"_id":"66266ff13b3cc996fef04849","id":"mlabonne/arena-preferences","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-04-27T16:04:59.000Z","likes":10,"trendingScore":0,"private":false,"sha":"57806d0d3de9f4d3c0cbcf8d40ee2029c799f86a","description":"\n\t\n\t\t\n\t\t⚔️ Arena Preferences\n\t\n\nThis is a preference dataset based on lmsys/chatbot_arena_conversations.\nIt contains multi-turn conversations (up to 11 turns) and original samples in 39 different languages (no translation).\n\nChosen answers are answers where GPT-4 was the winner (33k => 2,868 samples)\nDuplicates were removed (13 samples)\nGPTisms were removed (166 samples)\n\n\n\t\n\t\t\n\t\n\t\n\t\t📊 Plots\n\t\n\nHere's breakdown of the four most represented languages + an \"other\" bin in the dataset.\n\nHere's… See the full description on the dataset page: https://huggingface.co/datasets/mlabonne/arena-preferences.","downloads":27,"tags":["language:en","license:apache-2.0","size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-04-22T14:10:57.000Z","key":""},{"_id":"6626cb051d8f82a4717426bc","id":"mlabonne/oasst-preferences","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-04-22T21:16:06.000Z","likes":3,"trendingScore":0,"private":false,"sha":"39354415d74e056b2cfea35daa2b85731c24bb6c","description":"\n\t\n\t\t\n\t\tOpenAssistant Conversations Preferences\n\t\n\nThis is a preference dataset based on OpenAssistant/oasst1 and OpenAssistant/oasst2.\nIt is processed using a pipeline based on alexredna/oasst2_dpo_pairs.\nNote that there are outliers in user ratings.\n","downloads":10,"tags":["license:apache-2.0","size_categories:10K<n<100K","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-04-22T20:39:33.000Z","key":""},{"_id":"662720ffd6781c2164c81d34","id":"open-llm-leaderboard-old/details_mlabonne__ChimeraLlama-3-8B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-04-23T02:46:49.000Z","likes":0,"trendingScore":0,"private":false,"sha":"2cf4af66fba5b141674192f47e1fee6c6ac9925a","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/ChimeraLlama-3-8B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/ChimeraLlama-3-8B on the Open LLM Leaderboard.\nThe dataset is composed of 63 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard-old/details_mlabonne__ChimeraLlama-3-8B.","downloads":21,"tags":["region:us"],"createdAt":"2024-04-23T02:46:23.000Z","key":""},{"_id":"662aef7b72e3e3e4d8e0b7a0","id":"NickyNicky/mlabonne_orpo-dpo-mix-40k","author":"NickyNicky","disabled":false,"gated":false,"lastModified":"2024-04-26T00:08:08.000Z","likes":0,"trendingScore":0,"private":false,"sha":"a3cbcb7a52593b4797893f7121544a99efd93568","description":"https://huggingface.co/datasets/mlabonne/orpo-dpo-mix-40k\n\n","downloads":9,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-04-26T00:04:11.000Z","key":""},{"_id":"66325a3382a4b66b9eaf07f4","id":"open-llm-leaderboard-old/details_mlabonne__ChimeraLlama-3-8B-v2","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-05-01T15:05:42.000Z","likes":0,"trendingScore":0,"private":false,"sha":"20802081f11ca4fe8ab04a6a9c6f29ae97c9cc9a","downloads":31,"tags":["region:us"],"createdAt":"2024-05-01T15:05:23.000Z","key":""},{"_id":"6632614b31487f646e11d1b3","id":"open-llm-leaderboard-old/details_mlabonne__ChimeraLlama-3-8B-v3","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-05-01T15:35:58.000Z","likes":0,"trendingScore":0,"private":false,"sha":"6970fe81ba27ac30748ce899b2019c1935ed3e6e","downloads":20,"tags":["region:us"],"createdAt":"2024-05-01T15:35:39.000Z","key":""},{"_id":"6634fee77731add2c7808108","id":"open-llm-leaderboard-old/details_mlabonne__OrpoLlama-3-8B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-05-03T15:12:58.000Z","likes":0,"trendingScore":0,"private":false,"sha":"f18cf9a3c9e03cacffede74c9335399ae5ca20cf","downloads":20,"tags":["region:us"],"createdAt":"2024-05-03T15:12:39.000Z","key":""},{"_id":"663b46f19a5d20bd3a424192","id":"mlabonne/WildGPT-4","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-05-08T09:34:56.000Z","likes":8,"trendingScore":0,"private":false,"sha":"84d85ef3d3ba4d5fd0a9eb6b0289b9502147d3a0","description":"\n\t\n\t\t\n\t\tWildGPT-4\n\t\n\nHigh-quality conversations with GPT-4 models extracted from the excellent allenai/WildChat-1M dataset and reformatted in ShareGPT format.\n","downloads":31,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-05-08T09:33:37.000Z","key":""},{"_id":"6640a11328538eae74dab0f6","id":"mlabonne/SafeBeaverTails","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-05-12T11:05:56.000Z","likes":2,"trendingScore":0,"private":false,"sha":"46af80ea0ae94d195fbee23d501cdb1ad50f1b10","description":"\n\t\n\t\t\n\t\tSafeBeaverTails\n\t\n\nDeduped cleaned version of PKU-Alignment/BeaverTails where is_safe==Truein ShareGPT format.\n","downloads":12,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-05-12T10:59:31.000Z","key":""},{"_id":"66471f3d9861c9ff960f52a7","id":"OALL/details_mlabonne__NeuralMonarch-7B","author":"OALL","disabled":false,"gated":false,"lastModified":"2024-05-17T09:11:37.000Z","likes":0,"trendingScore":0,"private":false,"sha":"ab24d5ee959cefa5cd555e6a8927b977d7935a6f","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/NeuralMonarch-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/NeuralMonarch-7B.\nThe dataset is composed of 136 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest results.\nAn… See the full description on the dataset page: https://huggingface.co/datasets/OALL/details_mlabonne__NeuralMonarch-7B.","downloads":16,"tags":["size_categories:100K<n<1M","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-05-17T09:11:25.000Z","key":""},{"_id":"6647f168691370727c2070d1","id":"OALL/details_mlabonne__Beyonder-4x7B-v3","author":"OALL","disabled":false,"gated":false,"lastModified":"2024-05-18T00:08:21.000Z","likes":0,"trendingScore":0,"private":false,"sha":"2b1d6b08d326c77ddc9c820bcb5bba3600ecb797","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Beyonder-4x7B-v3\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Beyonder-4x7B-v3.\nThe dataset is composed of 136 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest results.\nAn… See the full description on the dataset page: https://huggingface.co/datasets/OALL/details_mlabonne__Beyonder-4x7B-v3.","downloads":64,"tags":["size_categories:100K<n<1M","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-05-18T00:08:08.000Z","key":""},{"_id":"664b68707a71afa504c1daa3","id":"OALL/details_mlabonne__AlphaMonarch-7B","author":"OALL","disabled":false,"gated":false,"lastModified":"2024-05-20T15:13:01.000Z","likes":0,"trendingScore":0,"private":false,"sha":"8b59f94b3006420cd2f6d0a6a27c9a855cdf2bfe","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/AlphaMonarch-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/AlphaMonarch-7B.\nThe dataset is composed of 136 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest results.\nAn additional… See the full description on the dataset page: https://huggingface.co/datasets/OALL/details_mlabonne__AlphaMonarch-7B.","downloads":53,"tags":["size_categories:100K<n<1M","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-05-20T15:12:48.000Z","key":""},{"_id":"66516ff565502c3cc22e56a7","id":"open-llm-leaderboard-old/details_mlabonne__Meta-Llama-3-12B-Instruct","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-05-25T04:58:47.000Z","likes":0,"trendingScore":0,"private":false,"sha":"ee36afdec999d6524d1729c2804eedf5e8b9dd56","downloads":21,"tags":["region:us"],"createdAt":"2024-05-25T04:58:29.000Z","key":""},{"_id":"6653f18c07cc2255eabd4f1e","id":"open-llm-leaderboard-old/details_mlabonne__Daredevil-8B","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-05-27T02:36:20.000Z","likes":0,"trendingScore":0,"private":false,"sha":"5d6a816bd31bf32a99168689f95b49d6455110d9","downloads":12,"tags":["region:us"],"createdAt":"2024-05-27T02:35:56.000Z","key":""},{"_id":"66546504f104a43fe1b41622","id":"open-llm-leaderboard-old/details_mlabonne__Llama-3-8B-Instruct-abliterated-dpomix","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-05-27T10:48:54.000Z","likes":0,"trendingScore":0,"private":false,"sha":"1264875542665f123bf18b651779376dbea90c43","downloads":22,"tags":["region:us"],"createdAt":"2024-05-27T10:48:36.000Z","key":""},{"_id":"66546531965ea394ee4230d0","id":"open-llm-leaderboard-old/details_mlabonne__Daredevil-8B-abliterated","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-05-27T10:49:40.000Z","likes":0,"trendingScore":0,"private":false,"sha":"bf8b01e949f491756e855223d062382b90af6bb8","downloads":17,"tags":["region:us"],"createdAt":"2024-05-27T10:49:21.000Z","key":""},{"_id":"6654fcf09ccb17d967da2def","id":"OALL/details_mlabonne__Daredevil-8B-abliterated","author":"OALL","disabled":false,"gated":false,"lastModified":"2024-05-27T21:36:59.000Z","likes":0,"trendingScore":0,"private":false,"sha":"2adc9bef1976f877233ebb2ca3cf0b0605a326e4","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Daredevil-8B-abliterated\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Daredevil-8B-abliterated.\nThe dataset is composed of 136 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest… See the full description on the dataset page: https://huggingface.co/datasets/OALL/details_mlabonne__Daredevil-8B-abliterated.","downloads":13,"tags":["size_categories:100K<n<1M","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-05-27T21:36:48.000Z","key":""},{"_id":"6655a282191c117e8185fbba","id":"PJMixers/mlabonne_orpo-dpo-mix-40k-PreferenceShareGPT","author":"PJMixers","disabled":false,"gated":false,"lastModified":"2024-05-30T15:47:22.000Z","likes":4,"trendingScore":0,"private":false,"sha":"40c37353c77e38743529d37a94a3a98f327bd5a2","downloads":20,"tags":["task_categories:reinforcement-learning","size_categories:10K<n<100K","format:json","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us","preference","preferences"],"createdAt":"2024-05-28T09:23:14.000Z","key":""},{"_id":"6655b7a93e89ad7f387933e6","id":"OALL/details_mlabonne__NeuralDaredevil-8B-abliterated","author":"OALL","disabled":false,"gated":false,"lastModified":"2024-05-28T10:53:39.000Z","likes":1,"trendingScore":0,"private":false,"sha":"ba3c9345dda41871c6e8e489415761a996a0aec2","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/NeuralDaredevil-8B-abliterated\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/NeuralDaredevil-8B-abliterated.\nThe dataset is composed of 136 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the… See the full description on the dataset page: https://huggingface.co/datasets/OALL/details_mlabonne__NeuralDaredevil-8B-abliterated.","downloads":32,"tags":["size_categories:100K<n<1M","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-05-28T10:53:29.000Z","key":""},{"_id":"66561cbaa57d0c38363003a7","id":"mlabonne/harmless_alpaca","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-05-30T09:03:22.000Z","likes":47,"trendingScore":0,"private":false,"sha":"02c6a92cfcf11bb0c387334f8146d149d65b587f","downloads":22099,"tags":["language:en","size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-05-28T18:04:42.000Z","key":""},{"_id":"6656bff5ed45f790fbcdcc3f","id":"open-llm-leaderboard-old/details_mlabonne__Daredevil-8B-abliterated-dpomix","author":"open-llm-leaderboard-old","disabled":false,"gated":false,"lastModified":"2024-05-29T05:41:28.000Z","likes":0,"trendingScore":0,"private":false,"sha":"2e51bf4ec6d7f42975a1c2968eeed38a10df4324","downloads":15,"tags":["region:us"],"createdAt":"2024-05-29T05:41:09.000Z","key":""},{"_id":"665c4c6ad94d2b10679f0e66","id":"mlabonne/french_alpaca","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-06-02T10:41:47.000Z","likes":1,"trendingScore":0,"private":false,"sha":"134687a14cea8ecb667d7c445fcadaf84a32f965","downloads":4,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-06-02T10:41:46.000Z","key":""},{"_id":"6662dd34b1fff5575ff2db32","id":"mlabonne/orpo-dpo-mix-40k-flat","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-06-07T10:19:22.000Z","likes":16,"trendingScore":0,"private":false,"sha":"fbf33b6432e9da27d1097e78180abae7d5b48754","description":"\n\t\n\t\t\n\t\tORPO-DPO-mix-40k-flat\n\t\n\n\nThis dataset is designed for ORPO or DPO training.\nSee Uncensor any LLM with Abliteration for more information about how to use it.\nThis is version with raw text instead of lists of dicts as in the original version here.\nIt makes easier to parse in Axolotl, especially for DPO.ORPO-DPO-mix-40k-flat is a combination of the following high-quality DPO datasets:\n\nargilla/Capybara-Preferences: highly scored chosen answers >=5 (7,424 samples)… See the full description on the dataset page: https://huggingface.co/datasets/mlabonne/orpo-dpo-mix-40k-flat.","downloads":44,"tags":["task_categories:text-generation","language:en","license:apache-2.0","size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us","dpo","rlhf","preference","orpo"],"createdAt":"2024-06-07T10:13:08.000Z","key":""},{"_id":"666b5c3dc636d47bd567a9d0","id":"open-llm-leaderboard/mlabonne__AlphaMonarch-7B-details","author":"open-llm-leaderboard","disabled":false,"gated":"auto","lastModified":"2025-02-13T11:39:10.000Z","likes":0,"trendingScore":0,"private":false,"sha":"8f92dc1a08d9075f996942bffbfbc9bab8e3fe2e","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/AlphaMonarch-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/AlphaMonarch-7B\nThe dataset is composed of 44 configuration(s), each one corresponding to one of the evaluated task.\nThe dataset has been created from 2 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest results.\nAn… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard/mlabonne__AlphaMonarch-7B-details.","downloads":23,"tags":["size_categories:10K<n<100K","format:json","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-06-13T20:53:17.000Z","key":""},{"_id":"666b621ad03be697a6fc75c5","id":"open-llm-leaderboard/mlabonne__Beyonder-4x7B-v3-details","author":"open-llm-leaderboard","disabled":false,"gated":"auto","lastModified":"2025-02-13T16:51:11.000Z","likes":0,"trendingScore":0,"private":false,"sha":"ec6aac757d1c099b5b3d4161eae78f0adb889916","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Beyonder-4x7B-v3\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Beyonder-4x7B-v3\nThe dataset is composed of 44 configuration(s), each one corresponding to one of the evaluated task.\nThe dataset has been created from 2 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest results.\nAn… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard/mlabonne__Beyonder-4x7B-v3-details.","downloads":16,"tags":["size_categories:10K<n<100K","format:json","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-06-13T21:18:18.000Z","key":""},{"_id":"666dade3067382b3e9cf79f2","id":"open-llm-leaderboard/mlabonne__NeuralDaredevil-8B-abliterated-details","author":"open-llm-leaderboard","disabled":false,"gated":"auto","lastModified":"2025-02-13T16:50:32.000Z","likes":0,"trendingScore":0,"private":false,"sha":"771cb44618e73f189b2425d060203dc0dee4af40","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/NeuralDaredevil-8B-abliterated\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/NeuralDaredevil-8B-abliterated\nThe dataset is composed of 39 configuration(s), each one corresponding to one of the evaluated task.\nThe dataset has been created from 5 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard/mlabonne__NeuralDaredevil-8B-abliterated-details.","downloads":25,"tags":["size_categories:10K<n<100K","format:json","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-06-15T15:06:11.000Z","key":""},{"_id":"6670056ea86e65e61e17da31","id":"open-llm-leaderboard/mlabonne__OrpoLlama-3-8B-details","author":"open-llm-leaderboard","disabled":false,"gated":"auto","lastModified":"2025-02-13T16:50:47.000Z","likes":0,"trendingScore":0,"private":false,"sha":"f1385df5030707c8bd53e7c51b9324e7e0503677","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/OrpoLlama-3-8B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/OrpoLlama-3-8B\nThe dataset is composed of 44 configuration(s), each one corresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest results.\nAn additional… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard/mlabonne__OrpoLlama-3-8B-details.","downloads":20,"tags":["size_categories:10K<n<100K","format:json","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-06-17T09:44:14.000Z","key":""},{"_id":"66700585df8cad47933541e4","id":"open-llm-leaderboard/mlabonne__phixtral-2x2_8-details","author":"open-llm-leaderboard","disabled":false,"gated":"auto","lastModified":"2025-02-13T11:34:46.000Z","likes":0,"trendingScore":0,"private":false,"sha":"0eccbb6f995403f0db7b9710c3ac6570b3e0a620","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/phixtral-2x2_8\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/phixtral-2x2_8\nThe dataset is composed of 44 configuration(s), each one corresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest results.\nAn additional… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard/mlabonne__phixtral-2x2_8-details.","downloads":18,"tags":["size_categories:10K<n<100K","format:json","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-06-17T09:44:37.000Z","key":""},{"_id":"6671d6952ff693dab0fd79b1","id":"arcee-ai/cleaned-mlabonne-distilabel-intel-orca-dpo-pairs","author":"arcee-ai","disabled":false,"gated":false,"lastModified":"2024-06-18T18:49:01.000Z","likes":0,"trendingScore":0,"private":false,"sha":"6371d62258ade7f4280acd399082a7a8c07a8333","downloads":6,"tags":["size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-06-18T18:48:53.000Z","key":""},{"_id":"6671d6eecbf550c42acbd94d","id":"arcee-ai/cleaned-mlabonne-distilabel-truthy-dpo-v0.1-filtered","author":"arcee-ai","disabled":false,"gated":false,"lastModified":"2024-06-18T18:50:24.000Z","likes":0,"trendingScore":0,"private":false,"sha":"dc6a6a50675d86d2f737f675ade6a9cd35e9caf2","downloads":7,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-06-18T18:50:22.000Z","key":""},{"_id":"6671d72e15d9d099c5b9b3d6","id":"arcee-ai/cleaned-mlabonne-distilabel-truthy-dpo-v0.1","author":"arcee-ai","disabled":false,"gated":false,"lastModified":"2024-06-18T18:51:28.000Z","likes":0,"trendingScore":0,"private":false,"sha":"ba9c5b5d6e1da6906b92ed4629f5f727dc79b047","downloads":3,"tags":["size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-06-18T18:51:26.000Z","key":""},{"_id":"6671d7f00f6ec76f3316455e","id":"arcee-ai/cleaned-mlabonne-orpo-dpo-mix-40k","author":"arcee-ai","disabled":false,"gated":false,"lastModified":"2024-06-18T18:55:33.000Z","likes":1,"trendingScore":0,"private":false,"sha":"c576fd8d127c4de5991a745bbc2ab88dc67968f9","downloads":8,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-06-18T18:54:40.000Z","key":""},{"_id":"6671d86fc974c7f5ea848976","id":"arcee-ai/cleaned-mlabonne-truthy-dpo-v0.1","author":"arcee-ai","disabled":false,"gated":false,"lastModified":"2024-06-18T18:56:50.000Z","likes":0,"trendingScore":0,"private":false,"sha":"26b0d3a1462be03448153fab81266648c66458b4","downloads":9,"tags":["size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-06-18T18:56:47.000Z","key":""},{"_id":"667766dcb3882fd587a03180","id":"arcee-ai/categorized-mlabonne-orpo-dpo-mix-40k","author":"arcee-ai","disabled":false,"gated":false,"lastModified":"2024-06-24T21:14:26.000Z","likes":2,"trendingScore":0,"private":false,"sha":"ae5f463d42a978b7947e7c0cc1c0c2ce76ec6a0c","downloads":11,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-06-23T00:05:48.000Z","key":""},{"_id":"66926e8b04f927126fa44524","id":"mlabonne/llmtwin","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-08-27T13:22:29.000Z","likes":21,"trendingScore":0,"private":false,"sha":"a28c03cf3372b1657c04f0d7bacd315626b06ba4","downloads":89,"tags":["language:en","size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-07-13T12:09:47.000Z","key":""},{"_id":"6698fd00a188ffb7e41280c1","id":"open-llm-leaderboard/mlabonne__Daredevil-8B-details","author":"open-llm-leaderboard","disabled":false,"gated":"auto","lastModified":"2025-02-13T11:37:23.000Z","likes":0,"trendingScore":0,"private":false,"sha":"3b7b045e21aff60b6bb801b284cae23f95f9df18","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Daredevil-8B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Daredevil-8B\nThe dataset is composed of 38 configuration(s), each one corresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest results.\nAn additional… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard/mlabonne__Daredevil-8B-details.","downloads":21,"tags":["size_categories:10K<n<100K","format:json","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-07-18T11:31:12.000Z","key":""},{"_id":"669cc4d9f21b09fdce89e667","id":"MLamateur/sdxltitian","author":"MLamateur","disabled":false,"gated":false,"lastModified":"2024-07-21T09:50:19.000Z","likes":0,"trendingScore":0,"private":false,"sha":"2bc7a668b25828b867b580521a4fc18ddb605a66","downloads":16,"tags":["size_categories:1K<n<10K","format:imagefolder","modality:image","library:datasets","library:mlcroissant","region:us"],"createdAt":"2024-07-21T08:20:41.000Z","key":""},{"_id":"669cdfa10bc10b346053d50f","id":"mlabonne/MiniInstruct-20k","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-07-23T10:53:17.000Z","likes":11,"trendingScore":0,"private":false,"sha":"93e7a0f154dcaac55d6ca7da1ad5df93afb941f1","downloads":18,"tags":["language:en","size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-07-21T10:14:57.000Z","key":""},{"_id":"669e35b3b5aa6ec3f9817caa","id":"open-llm-leaderboard/mlabonne__NeuralBeagle14-7B-details","author":"open-llm-leaderboard","disabled":false,"gated":"auto","lastModified":"2025-02-13T16:51:31.000Z","likes":0,"trendingScore":0,"private":false,"sha":"feaa120b68467b9e37a2145b8450074e47505930","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/NeuralBeagle14-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/NeuralBeagle14-7B\nThe dataset is composed of 38 configuration(s), each one corresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest results.\nAn… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard/mlabonne__NeuralBeagle14-7B-details.","downloads":17,"tags":["size_categories:10K<n<100K","format:json","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-07-22T10:34:27.000Z","key":""},{"_id":"669e3afdfbc9ba8a1299d2dc","id":"open-llm-leaderboard/mlabonne__Daredevil-8B-abliterated-details","author":"open-llm-leaderboard","disabled":false,"gated":"auto","lastModified":"2025-02-13T11:42:54.000Z","likes":0,"trendingScore":0,"private":false,"sha":"6f0f691ce6fa0fc6b24daf80e3d06fce93c0352d","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Daredevil-8B-abliterated\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Daredevil-8B-abliterated\nThe dataset is composed of 38 configuration(s), each one corresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard/mlabonne__Daredevil-8B-abliterated-details.","downloads":17,"tags":["size_categories:10K<n<100K","format:json","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-07-22T10:57:01.000Z","key":""},{"_id":"66a173095f58258df71cde84","id":"mlabonne/Numini-20k","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-07-26T13:53:29.000Z","likes":4,"trendingScore":0,"private":false,"sha":"0d3b4fc2833f9b9ce5a0bab47dda2c123c42bf72","downloads":17,"tags":["language:en","size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-07-24T21:32:57.000Z","key":""},{"_id":"66a53dc7d40a13036c5f2ebe","id":"mlabonne/FineTome-100k","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-07-29T09:52:30.000Z","likes":277,"trendingScore":0,"private":false,"sha":"c2343c1372ff31f51aa21248db18bffa3193efdb","description":"\n\t\n\t\t\n\t\tFineTome-100k\n\t\n\n\nThe FineTome dataset is a subset of arcee-ai/The-Tome (without arcee-ai/qwen2-72b-magpie-en), re-filtered using HuggingFaceFW/fineweb-edu-classifier.\nIt was made for my article \"Fine-tune Llama 3.1 Ultra-Efficiently with Unsloth\".\n","downloads":22256,"tags":["size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-07-27T18:34:47.000Z","key":""},{"_id":"66aa6d34f3ae6c4ba05fb9fc","id":"mlabonne/FineTome-Alpaca-100k","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-08-02T09:50:08.000Z","likes":9,"trendingScore":0,"private":false,"sha":"de34e2b111d8729d6ef2d14607931b918a116273","downloads":45,"tags":["language:en","size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-07-31T16:58:28.000Z","key":""},{"_id":"66b2cb73bc1b41c5f0be17a2","id":"OALL/details_mlabonne__ChimeraLlama-3-8B-v3","author":"OALL","disabled":false,"gated":false,"lastModified":"2024-08-07T01:18:54.000Z","likes":0,"trendingScore":0,"private":false,"sha":"9b6f6de1e5ba82b3b0e1c362c7e2418593c1d9bb","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/ChimeraLlama-3-8B-v3\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/ChimeraLlama-3-8B-v3.\nThe dataset is composed of 136 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest results.\nAn… See the full description on the dataset page: https://huggingface.co/datasets/OALL/details_mlabonne__ChimeraLlama-3-8B-v3.","downloads":5,"tags":["size_categories:100K<n<1M","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-08-07T01:18:43.000Z","key":""},{"_id":"66bb1d1206775d74907b21e0","id":"mlabonne/lmsys-arena-human-preference-filtered-19k","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-08-15T10:44:30.000Z","likes":5,"trendingScore":0,"private":false,"sha":"5767a98e42ddc2de8177d5b0b644e2759176da82","description":"\n\t\n\t\t\n\t\tlmsys-arena-human-preference-filtered-19k\n\t\n\nFiltered version of the lmsys/lmsys-arena-human-preference-55k.\nI removed the ties and samples where the winner isn't a GPT or Claude model.\n","downloads":40,"tags":["language:en","size_categories:10K<n<100K","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-08-13T08:45:06.000Z","key":""},{"_id":"66bc6ef7ec2404f45d525e44","id":"mlao01/km-news","author":"mlao01","disabled":false,"gated":false,"lastModified":"2024-08-21T07:35:16.000Z","likes":0,"trendingScore":0,"private":false,"sha":"9e0b63b07f09d31fde4997a6321b71ba5b814ed0","downloads":2,"tags":["size_categories:1M<n<10M","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-08-14T08:46:47.000Z","key":""},{"_id":"66cebdbed3d85430b7d9f524","id":"OALL/details_mlabonne__ChimeraLlama-3-8B-v2","author":"OALL","disabled":false,"gated":false,"lastModified":"2024-08-28T06:03:54.000Z","likes":0,"trendingScore":0,"private":false,"sha":"8780d6bee14eccc469bc22837c7c01cc70847267","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/ChimeraLlama-3-8B-v2\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/ChimeraLlama-3-8B-v2.\nThe dataset is composed of 136 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest results.\nAn… See the full description on the dataset page: https://huggingface.co/datasets/OALL/details_mlabonne__ChimeraLlama-3-8B-v2.","downloads":5,"tags":["size_categories:100K<n<1M","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-08-28T06:03:42.000Z","key":""},{"_id":"66cf7c13e63883817d7257c5","id":"mlabonne/llmtwin-dpo","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-08-30T11:00:32.000Z","likes":6,"trendingScore":0,"private":false,"sha":"e7031b61de65df3a3e6702b1da9c3f5fa3b18094","downloads":25,"tags":["language:en","size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-08-28T19:35:47.000Z","key":""},{"_id":"66d19315b4396d43c32e72a6","id":"OALL/details_mlabonne__FineLlama-3.1-8B","author":"OALL","disabled":false,"gated":false,"lastModified":"2024-08-30T09:38:41.000Z","likes":0,"trendingScore":0,"private":false,"sha":"ee9c8088e755e2d5a9a14598af22077287dfe1c8","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/FineLlama-3.1-8B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/FineLlama-3.1-8B.\nThe dataset is composed of 136 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest results.\nAn… See the full description on the dataset page: https://huggingface.co/datasets/OALL/details_mlabonne__FineLlama-3.1-8B.","downloads":7,"tags":["size_categories:100K<n<1M","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-08-30T09:38:29.000Z","key":""},{"_id":"66d5b42a35c36f266f7de95a","id":"open-llm-leaderboard/mlabonne__ChimeraLlama-3-8B-v3-details","author":"open-llm-leaderboard","disabled":false,"gated":"auto","lastModified":"2025-02-13T16:51:38.000Z","likes":0,"trendingScore":0,"private":false,"sha":"d4f1d576be53fa26fb92f080c3787486bc100d74","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/ChimeraLlama-3-8B-v3\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/ChimeraLlama-3-8B-v3\nThe dataset is composed of 38 configuration(s), each one corresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest results.\nAn… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard/mlabonne__ChimeraLlama-3-8B-v3-details.","downloads":28,"tags":["size_categories:10K<n<100K","format:json","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-09-02T12:48:42.000Z","key":""},{"_id":"66d5c5f141a7e81c201e337f","id":"open-llm-leaderboard/mlabonne__ChimeraLlama-3-8B-v2-details","author":"open-llm-leaderboard","disabled":false,"gated":"auto","lastModified":"2025-02-13T16:42:49.000Z","likes":0,"trendingScore":0,"private":false,"sha":"081aee37044ded78f752445a2b09e3055e624e14","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/ChimeraLlama-3-8B-v2\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/ChimeraLlama-3-8B-v2\nThe dataset is composed of 38 configuration(s), each one corresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest results.\nAn… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard/mlabonne__ChimeraLlama-3-8B-v2-details.","downloads":17,"tags":["size_categories:10K<n<100K","format:json","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-09-02T14:04:33.000Z","key":""},{"_id":"66e11303919f283fbd698165","id":"OALL/details_mlabonne__Meta-Llama-3.1-8B-Instruct-abliterated","author":"OALL","disabled":false,"gated":false,"lastModified":"2024-09-11T04:22:51.000Z","likes":0,"trendingScore":0,"private":false,"sha":"647e213b7db7a049148704b44b5ce8923be9761a","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Meta-Llama-3.1-8B-Instruct-abliterated\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Meta-Llama-3.1-8B-Instruct-abliterated.\nThe dataset is composed of 136 configuration, each one coresponding to one of the evaluated task.\nThe dataset has been created from 2 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always… See the full description on the dataset page: https://huggingface.co/datasets/OALL/details_mlabonne__Meta-Llama-3.1-8B-Instruct-abliterated.","downloads":9,"tags":["size_categories:100K<n<1M","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-09-11T03:48:19.000Z","key":""},{"_id":"66e4d1965c06b7719c2eba9d","id":"open-llm-leaderboard/mlabonne__Meta-Llama-3.1-8B-Instruct-abliterated-details","author":"open-llm-leaderboard","disabled":false,"gated":"auto","lastModified":"2025-02-13T11:30:36.000Z","likes":0,"trendingScore":0,"private":false,"sha":"a8a5890328921b6d16a40aaa6e55d1f6d79bcdd9","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Meta-Llama-3.1-8B-Instruct-abliterated\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Meta-Llama-3.1-8B-Instruct-abliterated\nThe dataset is composed of 38 configuration(s), each one corresponding to one of the evaluated task.\nThe dataset has been created from 3 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard/mlabonne__Meta-Llama-3.1-8B-Instruct-abliterated-details.","downloads":17,"tags":["size_categories:10K<n<100K","format:json","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-09-13T23:58:14.000Z","key":""},{"_id":"66eef6025a65f26be7693456","id":"mlabonne/orca-math-word-problems-80k","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-09-23T13:40:30.000Z","likes":4,"trendingScore":0,"private":false,"sha":"889dfec76c8c072c195fa3bdf0c8f492600c5146","description":"I removed samples where \"question\" character length was over 1,000 and \"answer\" character length was over 2,000, then randomly subsampled 80k rows.\n","downloads":34,"tags":["language:en","size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-09-21T16:36:18.000Z","key":""},{"_id":"66f3b4f20c84594c18c5c3c9","id":"open-llm-leaderboard/mlabonne__BigQwen2.5-Echo-47B-Instruct-details","author":"open-llm-leaderboard","disabled":false,"gated":"auto","lastModified":"2025-02-13T16:47:10.000Z","likes":0,"trendingScore":0,"private":false,"sha":"4060343b647721eaf2cf16a4409b4f344d08ce49","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/BigQwen2.5-Echo-47B-Instruct\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/BigQwen2.5-Echo-47B-Instruct\nThe dataset is composed of 38 configuration(s), each one corresponding to one of the evaluated task.\nThe dataset has been created from 1 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard/mlabonne__BigQwen2.5-Echo-47B-Instruct-details.","downloads":14,"tags":["size_categories:10K<n<100K","format:json","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-09-25T07:00:02.000Z","key":""},{"_id":"66f8bf9f80596a774a5e2e0b","id":"open-llm-leaderboard/mlabonne__BigQwen2.5-52B-Instruct-details","author":"open-llm-leaderboard","disabled":false,"gated":"auto","lastModified":"2025-02-13T11:33:21.000Z","likes":0,"trendingScore":0,"private":false,"sha":"eb7f2f5d014a44fd463361cab2efe2d26e85621d","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/BigQwen2.5-52B-Instruct\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/BigQwen2.5-52B-Instruct\nThe dataset is composed of 38 configuration(s), each one corresponding to one of the evaluated task.\nThe dataset has been created from 2 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard/mlabonne__BigQwen2.5-52B-Instruct-details.","downloads":18,"tags":["size_categories:10K<n<100K","format:json","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-09-29T02:46:55.000Z","key":""},{"_id":"66fc321082899af8ebd1dea8","id":"MLamateur/controlnet-dataset","author":"MLamateur","disabled":false,"gated":false,"lastModified":"2024-10-01T20:35:32.000Z","likes":0,"trendingScore":0,"private":false,"sha":"46c36cd6f2d7bb1c87f327c3d917f5baded5049c","downloads":17,"tags":["size_categories:1K<n<10K","format:parquet","modality:image","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-10-01T17:32:00.000Z","key":""},{"_id":"670be648bb2498474fc40ac0","id":"mlabonne/lmsys-arena-human-preference-55k-sharegpt","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-10-18T14:52:09.000Z","likes":11,"trendingScore":0,"private":false,"sha":"a5c298838e4bd8a8c9165d80b725b92b9e2ef285","downloads":50,"tags":["license:apache-2.0","size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-10-13T15:24:56.000Z","key":""},{"_id":"670bfa6bbb2498474fc82e8c","id":"mlabonne/ultrachat_200k_sft","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-10-13T16:51:18.000Z","likes":3,"trendingScore":0,"private":false,"sha":"6f8327f46a33d5ccf8144e6d3fc87339587680a2","downloads":496,"tags":["size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-10-13T16:50:51.000Z","key":""},{"_id":"670f5228021859516e8d6ba4","id":"open-llm-leaderboard/mlabonne__Hermes-3-Llama-3.1-70B-lorablated-details","author":"open-llm-leaderboard","disabled":false,"gated":"auto","lastModified":"2025-02-13T16:45:39.000Z","likes":0,"trendingScore":0,"private":false,"sha":"d1552e2404eb95f699bb1a44502b4ddb614acfd9","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Hermes-3-Llama-3.1-70B-lorablated\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Hermes-3-Llama-3.1-70B-lorablated\nThe dataset is composed of 38 configuration(s), each one corresponding to one of the evaluated task.\nThe dataset has been created from 2 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing… See the full description on the dataset page: https://huggingface.co/datasets/open-llm-leaderboard/mlabonne__Hermes-3-Llama-3.1-70B-lorablated-details.","downloads":18,"tags":["size_categories:10K<n<100K","format:json","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-10-16T05:42:00.000Z","key":""},{"_id":"672c30cab2f2dc21e1007f88","id":"rootflo/ml-asr-data-v1","author":"rootflo","disabled":false,"gated":false,"lastModified":"2024-11-07T03:20:05.000Z","likes":0,"trendingScore":0,"private":false,"sha":"93f1ed5bdb1d8555c44933567fa1380f95d49152","downloads":24,"tags":["size_categories:1K<n<10K","format:parquet","modality:audio","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-11-07T03:15:22.000Z","key":""},{"_id":"672c3ab67c4f7b937eb06784","id":"JianhongTu/MLAN","author":"JianhongTu","disabled":false,"gated":false,"lastModified":"2024-11-12T23:47:26.000Z","likes":0,"trendingScore":0,"private":false,"sha":"cde65a5c1e19cb6fa355167d2f82b6b841527d64","downloads":9,"tags":["license:cc0-1.0","modality:image","region:us"],"createdAt":"2024-11-07T03:57:42.000Z","key":""},{"_id":"672c3e998cfb611881314fdd","id":"rootflo/ml-asr-data-v2","author":"rootflo","disabled":false,"gated":false,"lastModified":"2024-11-07T04:19:38.000Z","likes":0,"trendingScore":0,"private":false,"sha":"539b61b9df0d95755b51e2497348bda886c421fd","downloads":15,"tags":["size_categories:1K<n<10K","format:parquet","modality:audio","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-11-07T04:14:17.000Z","key":""},{"_id":"673a3173bc83dde1402958a7","id":"mlabonne/orca-agentinstruct-1M-v1-cleaned","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-01-25T16:01:27.000Z","likes":69,"trendingScore":0,"private":false,"sha":"9643fc2a6ad2a5c6e2d1bddc9cbdeecd7cadc5a5","description":"\n\t\n\t\t\n\t\t🐋 Orca-AgentInstruct-1M-v1-cleaned\n\t\n\nThis is a cleaned version of the microsoft/orca-agentinstruct-1M-v1 dataset released by Microsoft.\n\norca-agentinstruct-1M-v1 is a fully synthetic dataset using only raw text publicly available on the web as seed data. It is a subset of the full AgentInstruct dataset (~25M samples) that created Orca-3-Mistral. Compared to Mistral 7B Instruct, the authors claim 40% improvement on AGIEval, 19% improvement on MMLU, 54% improvement on GSM8K, 38%… See the full description on the dataset page: https://huggingface.co/datasets/mlabonne/orca-agentinstruct-1M-v1-cleaned.","downloads":343,"tags":["task_categories:question-answering","language:en","license:cdla-permissive-2.0","size_categories:1M<n<10M","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-11-17T18:09:55.000Z","key":""},{"_id":"673be6cc0d8a679806dd1d86","id":"WangResearchLab/MLAN","author":"WangResearchLab","disabled":false,"gated":false,"lastModified":"2024-11-19T01:26:11.000Z","likes":2,"trendingScore":0,"private":false,"sha":"0bbb00828420dd8fc142f737c03b04d767f798d3","downloads":15,"tags":["license:cc0-1.0","modality:image","region:us"],"createdAt":"2024-11-19T01:15:56.000Z","key":""},{"_id":"673f8672bed7c843233b513a","id":"mlabonne/smoltalk-flat","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2024-11-21T19:17:18.000Z","likes":4,"trendingScore":0,"private":false,"sha":"5f64793b5b4c1964f3a5db59c63f0973cff98c63","description":"\n\t\n\t\t\n\t\tsmoltalk-flat\n\t\n\nA lazy flattened version of the excellent HuggingFaceTB/smoltalk dataset so the \"all\" split becomes \"default\" split (due to compatibility issues with most fine-tuning frameworks).\nFor MaziyarPanahi!\n","downloads":298,"tags":["size_categories:1M<n<10M","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2024-11-21T19:13:54.000Z","key":""},{"_id":"67478ca1261fd5af602b394b","id":"yaninsanity/Mla629FLHNBkFWXX86","author":"yaninsanity","disabled":false,"gated":"manual","lastModified":"2024-11-27T21:19:35.000Z","likes":0,"trendingScore":0,"private":false,"sha":"975a19fdf73def3fb46c1483ee24e943245ca8b2","downloads":2,"tags":["region:us"],"createdAt":"2024-11-27T21:18:25.000Z","key":""},{"_id":"674e7f6f7c967a07636e7c90","id":"nyu-dice-lab/lm-eval-results-mlabonne-Zebrafish-7B-private","author":"nyu-dice-lab","disabled":false,"gated":false,"lastModified":"2024-12-03T03:56:39.000Z","likes":0,"trendingScore":0,"private":false,"sha":"58bcf68a0a45e3e3c2083272f6c29e3a634ed2b6","description":"\n\t\n\t\t\n\t\tDataset Card for Evaluation run of mlabonne/Zebrafish-7B\n\t\n\n\n\nDataset automatically created during the evaluation run of model mlabonne/Zebrafish-7B\nThe dataset is composed of 62 configuration(s), each one corresponding to one of the evaluated task.\nThe dataset has been created from 2 run(s). Each run can be found as a specific split in each configuration, the split being named using the timestamp of the run.The \"train\" split is always pointing to the latest results.\nAn additional… See the full description on the dataset page: https://huggingface.co/datasets/nyu-dice-lab/lm-eval-results-mlabonne-Zebrafish-7B-private.","downloads":25,"tags":["size_categories:100K<n<1M","format:json","modality:tabular","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2024-12-03T03:47:59.000Z","key":""},{"_id":"677792ef4ec29e1bb5fb4462","id":"HgHgFace/math_mlalgo_dataset","author":"HgHgFace","disabled":false,"gated":false,"lastModified":"2025-01-03T08:36:21.000Z","likes":0,"trendingScore":0,"private":false,"sha":"c558501ff60feb2ebcb64d767938febd9abc991b","downloads":7,"tags":["license:llama3.3","size_categories:10K<n<100K","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-03T07:34:07.000Z","key":""},{"_id":"67790b9321dc30c084bb98d5","id":"lamm-mit/mlabonne-orca-math-word-problems-80k","author":"lamm-mit","disabled":false,"gated":false,"lastModified":"2025-01-04T10:21:26.000Z","likes":1,"trendingScore":0,"private":false,"sha":"0efe650252fc6b88d22a0d0d42769a3a490c8a7f","downloads":34,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-04T10:21:07.000Z","key":""},{"_id":"677a976578ac1cec96438fc4","id":"QuixiAI/mlabonne_orca-agentinstruct-1M-v1-cleaned-DolphinLabeled","author":"QuixiAI","disabled":false,"gated":false,"lastModified":"2025-01-05T14:58:05.000Z","likes":6,"trendingScore":0,"private":false,"sha":"ec8af8f7d56ca0eb27b0d1c648bb6061cbb9c191","description":"\n\t\n\t\t\n\t\torca-agentinstruct-1M-v1-cleaned DolphinLabeled\n\t\n\n\n\t\n\t\t\n\t\tPart of the DolphinLabeled series of datasets\n\t\n\n\n\t\n\t\t\n\t\tPresented by Eric Hartford and Cognitive Computations\n\t\n\nThe purpose of this dataset is to enable filtering of orca-agentinstruct-1M-v1-cleaned dataset.\nThe original dataset is mlabonne/orca-agentinstruct-1M-v1-cleaned\n(thank you to microsoft and mlabonne)\nI have modified the dataset using two scripts.\n\ndedupe.py - removes rows with identical final response.\nlabel.py -… See the full description on the dataset page: https://huggingface.co/datasets/QuixiAI/mlabonne_orca-agentinstruct-1M-v1-cleaned-DolphinLabeled.","downloads":27,"tags":["task_categories:question-answering","language:en","license:cdla-permissive-2.0","size_categories:1M<n<10M","format:json","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-05T14:29:57.000Z","key":""},{"_id":"677eb02c49f8714ffe4c3f8b","id":"vineet10/mlabonne_simplified","author":"vineet10","disabled":false,"gated":false,"lastModified":"2025-01-08T17:04:56.000Z","likes":0,"trendingScore":0,"private":false,"sha":"6e79ab4664bbfa253871e5e5e90ba82c671bedb3","downloads":9,"tags":["size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-08T17:04:44.000Z","key":""},{"_id":"677eb8a32ee4442e3de60fe2","id":"vineet10/HF_Style_mlabonne","author":"vineet10","disabled":false,"gated":false,"lastModified":"2025-01-08T17:41:04.000Z","likes":0,"trendingScore":0,"private":false,"sha":"cde1ae1d1ada6c78acfa93a4b710bb91d3002f98","downloads":6,"tags":["size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-08T17:40:51.000Z","key":""},{"_id":"677ebcca8a997cb9bfe40d4a","id":"vineet10/HF_Style_mlabonne_FineTome-100","author":"vineet10","disabled":false,"gated":false,"lastModified":"2025-01-08T17:58:46.000Z","likes":0,"trendingScore":0,"private":false,"sha":"788d5ef4c4df158dcd863f5813c9e57674800e44","downloads":4,"tags":["size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-08T17:58:34.000Z","key":""},{"_id":"6782662a815071eef29cafc5","id":"math-extraction-comp/mlabonne__AlphaMonarch-7B","author":"math-extraction-comp","disabled":false,"gated":false,"lastModified":"2025-01-12T21:24:52.000Z","likes":0,"trendingScore":0,"private":false,"sha":"14c7e7672d46e81737f208e9552d5ac6aa988581","downloads":7,"tags":["size_categories:1K<n<10K","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-11T12:38:02.000Z","key":""},{"_id":"6782666181e69ba91a5eee86","id":"math-extraction-comp/mlabonne__NeuralBeagle14-7B","author":"math-extraction-comp","disabled":false,"gated":false,"lastModified":"2025-01-12T21:25:33.000Z","likes":0,"trendingScore":0,"private":false,"sha":"28c5365385d1a28d04a573a3419dca5a9f751b9a","downloads":6,"tags":["size_categories:1K<n<10K","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-11T12:38:57.000Z","key":""},{"_id":"678266b3228db3b4c238e634","id":"math-extraction-comp/mlabonne__OrpoLlama-3-8B","author":"math-extraction-comp","disabled":false,"gated":false,"lastModified":"2025-01-12T21:26:22.000Z","likes":0,"trendingScore":0,"private":false,"sha":"982683dc7604ca9e34ce9f7b2380180ec73a6864","downloads":8,"tags":["size_categories:1K<n<10K","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-11T12:40:19.000Z","key":""},{"_id":"6782e0b1add0de55cc36f495","id":"MLamateur/tradmas","author":"MLamateur","disabled":false,"gated":false,"lastModified":"2025-01-11T21:26:08.000Z","likes":0,"trendingScore":0,"private":false,"sha":"26ab14e526be00bb61aa046bd0bac5a19b4c8871","downloads":35,"tags":["size_categories:n<1K","format:text","modality:image","modality:text","library:datasets","library:mlcroissant","region:us"],"createdAt":"2025-01-11T21:20:49.000Z","key":""},{"_id":"678591b07ecdfd2fdb1a3f9f","id":"mlabonne/smoltalk-semhashed","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-01-13T22:29:16.000Z","likes":8,"trendingScore":0,"private":false,"sha":"c4bd9c0dfc79986b9a5a29b976eeeb3b50804e23","description":"\n\t\n\t\t\n\t\tSmolTalk SemHashed\n\t\n\n\nThis is a near-deduplicated version of smoltalk created with the semhash library.\nInstead of MinHash deduplication, it uses embeddings generated with minishlab/potion-base-8M, a distilled version of BAAI/bge-base-en-v1.5, and a threshold of 0.95 (see the vicinity library).❤️ Kudos to minishlab for this super cool stuff!\n","downloads":43,"tags":["size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-13T22:20:32.000Z","key":""},{"_id":"6786e8ff8fc5dcb0357115f1","id":"danivpv/ml-arxiv-instruct","author":"danivpv","disabled":false,"gated":false,"lastModified":"2026-08-14T16:47:50.000Z","likes":0,"trendingScore":0,"private":false,"sha":"adc185bcda202fa9e94a3a4b3280e74e7ed18880","description":"\n\t\n\t\t\n\t\n\t\n\t\tML ArXiv Instruct Dataset\n\t\n\nA synthetically generated instruction-following dataset for training domain-expert\nlanguage models in Machine Learning. Contains instruction/answer pairs derived\ndirectly from ArXiv ML papers.\nUsed to train danivpv/Llama-ML-Expert-Instruct-1b.\nPart of the LLM-ArXiv-Domain-Expert pipeline.\n\n\t\n\t\t\n\t\n\t\n\t\tData Generation Pipeline\n\t\n\n\nArXiv ML papers parsed with IBM's docling\nParsed content chunked into coherent extracts\nAn LLM (ChatOpenAI) generates one… See the full description on the dataset page: https://huggingface.co/datasets/danivpv/ml-arxiv-instruct.","downloads":31,"tags":["task_categories:text-generation","language:en","license:mit","size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us","arxiv","machine-learning","instruction-tuning","synthetic"],"createdAt":"2025-01-14T22:45:19.000Z","key":""},{"_id":"6789d0f752c3093b1151b460","id":"MLap/Sentiment2Emoji","author":"MLap","disabled":false,"gated":false,"lastModified":"2025-10-13T15:30:08.000Z","likes":2,"trendingScore":0,"private":false,"sha":"a0dbe937f5bce5ddbdeaa650e8b239c1d704cb93","description":"This dataset contains text mapped with emoji, where each emoji description captures the underlying sentiment of the text crisply.\nIt is designed for sentiment analysis and emotion classification tasks, making it ideal for natural language processing (NLP) enthusiasts.\nDataset usecase: Fine-tune an LLM on sentiment-to-emoji use cases.\n\n\t\n\t\t\n\t\tPlease cite it as:\n\t\n\n@dataset{Aman_Sentiment2Emoji_2025,\n  title        = {Sentiment2Emoji},\n  author       = {Aman Prakash},\n  year         = {2025}… See the full description on the dataset page: https://huggingface.co/datasets/MLap/Sentiment2Emoji.","downloads":17,"tags":["language:en","license:mit","size_categories:n<1K","format:csv","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-17T03:39:35.000Z","key":""},{"_id":"678a7d2d2d604a498f52262e","id":"danivpv/ml-arxiv-dpo","author":"danivpv","disabled":false,"gated":false,"lastModified":"2026-08-14T16:46:40.000Z","likes":0,"trendingScore":0,"private":false,"sha":"4a179e7402d861aa93025ef1eff14d2d00ca39ef","description":"\n\t\n\t\t\n\t\n\t\n\t\tML ArXiv Preference Dataset (DPO)\n\t\n\nA preference alignment dataset of instruction/chosen/rejected triples, built for\nDirect Preference Optimization (DPO) and RLHF-style alignment toward an academic,\nauthoritative Machine Learning tone.\nPart of the LLM-ArXiv-Domain-Expert pipeline.\n\n\t\n\t\t\n\t\n\t\n\t\tData Generation Pipeline\n\t\n\nBuilt on the same extraction pipeline as ml-arxiv-instruct —\nArXiv papers parsed with docling, chunked into extracts. For each extract:\n\ninstruction: a… See the full description on the dataset page: https://huggingface.co/datasets/danivpv/ml-arxiv-dpo.","downloads":28,"tags":["task_categories:text-generation","language:en","size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us","arxiv","machine-learning","dpo","preference-alignment","synthetic"],"createdAt":"2025-01-17T15:54:21.000Z","key":""},{"_id":"679410c4f154af562e1cbce6","id":"mlabonne/EightAya","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-01-24T22:14:30.000Z","likes":0,"trendingScore":0,"private":false,"sha":"609a30292ff89e4768e684df3574354da976354a","downloads":10,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-24T22:14:28.000Z","key":""},{"_id":"679411c7b08f4b7f3a777e9c","id":"mlabonne/WildGPT-4-HF","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-01-24T22:19:09.000Z","likes":0,"trendingScore":0,"private":false,"sha":"faa04bdebd092b0910e09dde8088d405bddcd97d","downloads":19,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-24T22:18:47.000Z","key":""},{"_id":"679548d1a29a934c085b438e","id":"mlao01/km-news-qwen2.5","author":"mlao01","disabled":false,"gated":false,"lastModified":"2025-02-15T21:25:12.000Z","likes":0,"trendingScore":0,"private":false,"sha":"917a43346dd6927ca9ddc366710b86409223ae83","downloads":10,"tags":["size_categories:1M<n<10M","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-25T20:25:53.000Z","key":""},{"_id":"679ba146815f472f661482a5","id":"mlabonne/dolphin-r1-deepseek","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-01-30T17:09:47.000Z","likes":19,"trendingScore":0,"private":false,"sha":"22b55c9e89a9f5e428eac65c1968a9ca7e8b880b","description":"\n\t\n\t\t\n\t\tDolphin R1 DeepSeek 🐬\n\t\n\nAn Apache-2.0 dataset curated by Eric Hartford and Cognitive Computations. The purpose of this dataset is to train R1-style reasoning models.\nThis is a reformatted version of the DeepSeek subset for ease of use. It adds the model's response to the conversation with the following special tokens: <|begin_of_thought|>, <|end_of_thought|>, <|begin_of_solution|>, <|end_of_solution|>.\nPlease like the original dataset if you enjoy this reformatted version. Thanks to… See the full description on the dataset page: https://huggingface.co/datasets/mlabonne/dolphin-r1-deepseek.","downloads":127,"tags":["license:apache-2.0","size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-30T15:56:54.000Z","key":""},{"_id":"679ba1fe256f46e1a27f5ba4","id":"mlabonne/dolphin-r1-flash","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-01-30T17:09:44.000Z","likes":6,"trendingScore":0,"private":false,"sha":"1ae1b0ea57a085c6b2d8a9cb232b7d188efc453d","description":"\n\t\n\t\t\n\t\tDolphin R1 Flash 🐬\n\t\n\nAn Apache-2.0 dataset curated by Eric Hartford and Cognitive Computations. The purpose of this dataset is to train R1-style reasoning models.\nThis is a reformatted version of the Flash subset for ease of use. It adds the model's response to the conversation with the following special tokens: <|begin_of_thought|>, <|end_of_thought|>, <|begin_of_solution|>, <|end_of_solution|>.\nPlease like the original dataset if you enjoy this reformatted version. Thanks to Eric… See the full description on the dataset page: https://huggingface.co/datasets/mlabonne/dolphin-r1-flash.","downloads":35,"tags":["license:apache-2.0","size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-30T15:59:58.000Z","key":""},{"_id":"679bd14457edf64913308040","id":"mlabonne/OpenThoughts-79k-filtered","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-01-30T21:22:15.000Z","likes":6,"trendingScore":0,"private":false,"sha":"6a5ff50639ca6a75e9a5ad06a6ab9a9791d0514d","description":"This is a fixed version of open-thoughts/OpenThoughts-114k that removes the 32,390 wrong math answers observed in open-r1/OpenThoughts-114k-math.\nThanks to the open-r1 org for this work. I'm curious to know more about the origin of this issue.\n","downloads":59,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-01-30T19:21:40.000Z","key":""},{"_id":"67a60b9ab769071cd0f14b9b","id":"mlabonne/s1K-formatted","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-02-07T13:41:14.000Z","likes":3,"trendingScore":0,"private":false,"sha":"3945b4c9d43dfa1404e9a01eb8e11ea63774c950","description":"This is a reformatted version of simplescaling/s1K with an HF/OAI format.\nI created the \"messages\" column and added special tokens for CoT: <|begin_of_thought|>, <|end_of_thought|>, <|begin_of_solution|>, <|end_of_solution|>.\n","downloads":27,"tags":["size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-02-07T13:33:14.000Z","key":""},{"_id":"67aa1e57963961b88475eab5","id":"Mohamed-DLM/eld7e7_CFeW61i_MLA_mp3_updated","author":"Mohamed-DLM","disabled":false,"gated":false,"lastModified":"2025-03-13T11:47:04.000Z","likes":0,"trendingScore":0,"private":false,"sha":"b81956b1be5a8e108b3de2844abfcf83d170689f","downloads":5,"tags":["size_categories:n<1K","format:parquet","modality:audio","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-02-10T15:42:15.000Z","key":""},{"_id":"67b2dca3052b802b4a2cc398","id":"fzliu/ml-arxiv-papers-voyage-3-lite","author":"fzliu","disabled":false,"gated":false,"lastModified":"2025-02-17T06:53:11.000Z","likes":0,"trendingScore":0,"private":false,"sha":"759ee1877c59c524bface0ae2f4de18b21257dff","downloads":25,"tags":["license:afl-3.0","size_categories:100K<n<1M","format:arrow","modality:tabular","modality:text","library:datasets","library:mlcroissant","region:us"],"createdAt":"2025-02-17T06:52:19.000Z","key":""},{"_id":"67b70c926b0e4fb2c2b13df3","id":"mlabonne/natural_reasoning-formatted","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-02-21T09:50:09.000Z","likes":17,"trendingScore":0,"private":false,"sha":"d18d42924f7c14430554c634c7cac0435250fc04","description":"\n\t\n\t\t\n\t\tNatural reasoning\n\t\n\nThis is a reformatted version of facebook/natural_reasoning to match the HF/OAI format.\n","downloads":60,"tags":["task_categories:text-generation","language:en","license:cc-by-nc-4.0","size_categories:1M<n<10M","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-02-20T11:05:54.000Z","key":""},{"_id":"67bc7356855a6b996b93b3cd","id":"mlabonne/smoltldr","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-02-26T14:16:52.000Z","likes":12,"trendingScore":0,"private":false,"sha":"b6d383630fadd1c95f7a53bac2305ca67e77221d","description":"This dataset was designed for the fine-tune of HuggingFaceTB/SmolLM2-135M-Instruct using GRPO.\nIt is designed to summarize Reddit posts.\nYou can reproduce this training using this colab notebook. It takes about 40 minutes to train the model.\n","downloads":158,"tags":["language:en","size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-02-24T13:25:42.000Z","key":""},{"_id":"67cb3fbb425e78e3c281d3cb","id":"kohoutck/ml-ai-capstone","author":"kohoutck","disabled":false,"gated":false,"lastModified":"2025-03-07T19:11:47.000Z","likes":0,"trendingScore":0,"private":false,"sha":"73dc19a56f36268bf6bd625466f5a237d2958b76","downloads":6,"tags":["size_categories:n<1K","format:audiofolder","modality:audio","library:datasets","library:mlcroissant","region:us"],"createdAt":"2025-03-07T18:49:31.000Z","key":""},{"_id":"67d1aafe9b1ccef906a8abaf","id":"Mohamed-DLM/eld7e7_CFeW61i_MLA_mp3_updated_updated","author":"Mohamed-DLM","disabled":false,"gated":false,"lastModified":"2025-03-17T21:43:11.000Z","likes":0,"trendingScore":0,"private":false,"sha":"5d6dce061f269f0c4414565f634d97944ebd75bb","downloads":4,"tags":["size_categories:n<1K","format:parquet","modality:audio","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-03-12T15:40:46.000Z","key":""},{"_id":"67d1b64c9b18fe34ba02486f","id":"mlabonne/lmsys-arena-human-preference-55k","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-03-12T16:29:57.000Z","likes":4,"trendingScore":0,"private":false,"sha":"9b30e5e665f1a7b9af26457fe28e0f5b4925429e","downloads":112,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-03-12T16:29:00.000Z","key":""},{"_id":"67d1b70be10c7e6bb9be91bc","id":"mlabonne/lmsys-arena-human-sft-55k","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-03-12T16:33:03.000Z","likes":6,"trendingScore":0,"private":false,"sha":"b11644d6cc73d2917cee436e8ddcbe947db5aac5","downloads":27,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-03-12T16:32:11.000Z","key":""},{"_id":"67d1c0b1a0c6548ae99b0024","id":"mlabonne/ultrafeedback-sft","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-03-12T17:13:27.000Z","likes":3,"trendingScore":0,"private":false,"sha":"9e5964479e6c5c9db8ff0b1057889e3056f33a3e","downloads":27,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-03-12T17:13:21.000Z","key":""},{"_id":"67d1c4ca7d61c9b8c4eb029e","id":"mlabonne/MetaMathQA-chat","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-03-12T17:31:07.000Z","likes":1,"trendingScore":0,"private":false,"sha":"74cd9145e1b1448e477486331d2d66d6a0ba0d4b","downloads":33,"tags":["size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-03-12T17:30:50.000Z","key":""},{"_id":"67d1e4835cfbd07d54f1ec3e","id":"mlabonne/opc-sft-stage2-chat","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-03-12T19:46:35.000Z","likes":1,"trendingScore":0,"private":false,"sha":"1d11752ecef9156c40e5444523e4fda6fb7b4861","downloads":30,"tags":["size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-03-12T19:46:11.000Z","key":""},{"_id":"67d1ea858417cb4a601356f9","id":"mlabonne/ToolACE","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-03-12T20:12:46.000Z","likes":6,"trendingScore":0,"private":false,"sha":"60fed4475c09729dd11db59a2092d8503df074c1","downloads":121,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-03-12T20:11:49.000Z","key":""},{"_id":"67dcd1f3825b014322744a14","id":"MLap/Book-Scan-OCR","author":"MLap","disabled":false,"gated":false,"lastModified":"2025-10-13T15:32:49.000Z","likes":3,"trendingScore":0,"private":false,"sha":"1c866fda3b5c2845197fb43c2f6847a92197a2e1","description":"\n\t\n\t\t\n\t\tBest Usage\n\t\n\n\nSuitable for fine-tuning Vision-Language Models (e.g., PaliGemma).\n\n\n\t\n\t\t\n\t\tDataset Creation\n\t\n\nThis dataset was generated using Mistral OCR and Google Lens, followed by manual cleaning for improved accuracy.  \n\n\t\n\t\t\n\t\tImage Source\n\t\n\nImages are sourced from Sarvam.ai.  \n\n\t\n\t\t\n\t\tPlease cite it as:\n\t\n\n@dataset{Aman_MLap_BookScanOCR_2025,\n  title        = {Book-Scan-OCR},\n  author       = {Aman Prakash},\n  year         = {2025},\n  publisher    = {Hugging Face},\n  url… See the full description on the dataset page: https://huggingface.co/datasets/MLap/Book-Scan-OCR.","downloads":30,"tags":["task_categories:image-to-text","language:en","license:cc-by-4.0","size_categories:n<1K","format:parquet","modality:image","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us","finetuning","VLM","OCR","Text","TextExtraction"],"createdAt":"2025-03-21T02:41:55.000Z","key":""},{"_id":"67fb76fb019efee226b29832","id":"MLap/GeoDE","author":"MLap","disabled":false,"gated":false,"lastModified":"2025-04-13T14:58:16.000Z","likes":1,"trendingScore":0,"private":false,"sha":"acbae31823b22f84c3bbb11585d37d50aabaac44","description":"Official Paper\nNumber of country classes: 40Total number of images: 61925  \n\n\t\n\t\t\n\t\tImage count per country_ip class\n\t\n\n\n\t\n\t\t\nCountry\nNumber of Images\n\n\n\t\t\nAngola\n10\n\n\nArgentina\n3193\n\n\nBotswana\n3\n\n\nBrazil\n16\n\n\nBulgaria\n1\n\n\nCameroon\n1\n\n\nChina\n1565\n\n\nColombia\n3703\n\n\nEgypt\n2449\n\n\nFrance\n59\n\n\nGhana\n1\n\n\nGreece\n45\n\n\nIndonesia\n5311\n\n\nIreland\n2\n\n\nItaly\n3933\n\n\nJapan\n6500\n\n\nJordan\n43\n\n\nMalaysia\n55\n\n\nMexico\n2723\n\n\nMoldova\n2\n\n\nNetherlands\n18\n\n\nNigeria\n5729\n\n\nPhilippines\n2906\n\n\nPoland\n68\n\n\nPortugal\n139… See the full description on the dataset page: https://huggingface.co/datasets/MLap/GeoDE.","downloads":67,"tags":["task_categories:image-classification","language:en","license:cc-by-4.0","size_categories:10K<n<100K","format:parquet","modality:image","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us","Vision","VLM","GeoDE"],"createdAt":"2025-04-13T08:34:03.000Z","key":""},{"_id":"680158d64c2af8d3913d942c","id":"french-datasets/mlabonne_medical-mqca-fr","author":"french-datasets","disabled":false,"gated":false,"lastModified":"2025-04-17T19:40:10.000Z","likes":0,"trendingScore":0,"private":false,"sha":"dcf3b256c176693d538fe43eb7c7d87064012696","description":"Ce répertoire est vide, il a été créé pour améliorer le référencement du jeu de données mlabonne/medical-mqca-fr.\n","downloads":22,"tags":["language:fra","region:us"],"createdAt":"2025-04-17T19:39:02.000Z","key":""},{"_id":"68015940c56bc1e73166219f","id":"french-datasets/mlabonne_medical-cases-fr","author":"french-datasets","disabled":false,"gated":false,"lastModified":"2025-04-17T19:41:25.000Z","likes":0,"trendingScore":0,"private":false,"sha":"98e84b9768cfc203d4e405bb7ee85183bbbb0b37","description":"Ce répertoire est vide, il a été créé pour améliorer le référencement du jeu de données mlabonne/medical-cases-fr.\n","downloads":21,"tags":["language:fra","region:us"],"createdAt":"2025-04-17T19:40:48.000Z","key":""},{"_id":"680159892f0a0d5e8ffbc144","id":"french-datasets/mlabonne_bactrian-fr","author":"french-datasets","disabled":false,"gated":false,"lastModified":"2025-04-17T19:42:44.000Z","likes":0,"trendingScore":0,"private":false,"sha":"e0c076107da11bcf84327dee93fc0492e1c1eb25","description":"Ce répertoire est vide, il a été créé pour améliorer le référencement du jeu de données mlabonne/bactrian-fr.\n","downloads":4,"tags":["language:fra","region:us"],"createdAt":"2025-04-17T19:42:01.000Z","key":""},{"_id":"680159d40e72d7bd239d7a90","id":"french-datasets/mlabonne_french_alpaca","author":"french-datasets","disabled":false,"gated":false,"lastModified":"2025-04-17T19:43:54.000Z","likes":0,"trendingScore":0,"private":false,"sha":"0290d1a5543cbb0fdcf8f8d6f4585478f7092fc4","description":"Ce répertoire est vide, il a été créé pour améliorer le référencement du jeu de données mlabonne/french_alpaca.\n","downloads":3,"tags":["language:fra","region:us"],"createdAt":"2025-04-17T19:43:16.000Z","key":""},{"_id":"680cce65f1a0be37efe0c3b5","id":"iamshreeji-copy1/MLADDC_T2","author":"iamshreeji-copy1","disabled":false,"gated":false,"lastModified":"2025-04-26T12:15:38.000Z","likes":0,"trendingScore":0,"private":false,"sha":"d1d1b31af97963f943bc404e382ca6092b6857a9","downloads":1,"tags":["region:us"],"createdAt":"2025-04-26T12:15:33.000Z","key":""},{"_id":"6816d5235b5eb901bc575dc6","id":"mlaurindo/DFM_Models","author":"mlaurindo","disabled":false,"gated":false,"lastModified":"2025-05-04T03:25:26.000Z","likes":0,"trendingScore":0,"private":false,"sha":"8d97e22c88dafa44202fc8f49be0d9a048516e90","downloads":10,"tags":["region:us"],"createdAt":"2025-05-04T02:46:59.000Z","key":""},{"_id":"681848a77f12850cb3ac36d4","id":"MLawrence/LegalMicheal","author":"MLawrence","disabled":false,"gated":false,"lastModified":"2025-05-05T05:13:24.000Z","likes":0,"trendingScore":0,"private":false,"sha":"49d238ebd4126a4cd42a160dda56e085ee58f134","downloads":11,"tags":["license:mit","size_categories:n<1K","format:json","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-05-05T05:12:07.000Z","key":""},{"_id":"6818492ad193541d2f592d69","id":"MLawrence/MichealLegal","author":"MLawrence","disabled":false,"gated":false,"lastModified":"2025-05-05T05:14:39.000Z","likes":0,"trendingScore":0,"private":false,"sha":"7476aff4e7003cff47a348c2e24a3bae5a9cd210","downloads":15,"tags":["license:apache-2.0","size_categories:n<1K","format:json","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-05-05T05:14:18.000Z","key":""},{"_id":"681b7079bead91148f580b6b","id":"inayarhmns/MLAMA-dod","author":"inayarhmns","disabled":false,"gated":false,"lastModified":"2025-05-07T14:39:05.000Z","likes":1,"trendingScore":0,"private":false,"sha":"afaba72e7406e847ae66fbba673f840f02261307","downloads":4,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-05-07T14:38:49.000Z","key":""},{"_id":"68294290935b0f2fcdf6446a","id":"inayarhmns/MLAMA-dod-185","author":"inayarhmns","disabled":false,"gated":false,"lastModified":"2025-05-18T02:15:02.000Z","likes":0,"trendingScore":0,"private":false,"sha":"c67dbfd31de4065f7fbb507c84bae5de1c42c50b","downloads":12,"tags":["size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-05-18T02:14:40.000Z","key":""},{"_id":"6839a5b5f0458c8bcb807ff8","id":"mlazniewski/initial_dataset","author":"mlazniewski","disabled":false,"gated":false,"lastModified":"2025-05-30T14:30:18.000Z","likes":0,"trendingScore":0,"private":false,"sha":"cff2f75fcf3d2354f27982e959e9c59bdef5bea5","description":"\n\t\n\t\t\n\t\tMy Cool Dataset\n\t\n\nThis dataset is an example of how to create and upload a dataset card using Python. I use only to practice how to manipulate\ndataset iteslf, add new data remove them. Fix typos.\n\n\t\n\t\t\n\t\tDataset Details\n\t\n\n\nLanguage: English\nLicense: MIT\nTags: text-classification, example\n\n","downloads":14,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-05-30T12:33:57.000Z","key":""},{"_id":"683a2548794afac0687eae09","id":"ValierJuri/mlaad-2000-audio","author":"ValierJuri","disabled":false,"gated":false,"lastModified":"2025-05-30T22:17:44.000Z","likes":0,"trendingScore":0,"private":false,"sha":"7d5db142880b5d7b335e118ffd3ceb03b779b94e","downloads":16,"tags":["size_categories:1K<n<10K","format:arrow","modality:tabular","modality:text","library:datasets","library:mlcroissant","region:us"],"createdAt":"2025-05-30T21:38:16.000Z","key":""},{"_id":"683c8ee021ca86ce52b4079b","id":"mlabonne/FineTome-100k-dedup","author":"mlabonne","disabled":false,"gated":false,"lastModified":"2025-06-01T19:05:21.000Z","likes":8,"trendingScore":0,"private":false,"sha":"a8a30a0feff9707a72d373ef42a7ba12fd1bda21","downloads":110,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-06-01T17:33:20.000Z","key":""},{"_id":"68456d28935830fd28f1529e","id":"MohMounir/MLAD-chunked","author":"MohMounir","disabled":false,"gated":false,"lastModified":"2025-06-08T10:59:52.000Z","likes":0,"trendingScore":0,"private":false,"sha":"e50b818b9d122e6ba94a5fb995e570c5ab4b85cd","downloads":2,"tags":["region:us"],"createdAt":"2025-06-08T10:59:52.000Z","key":""},{"_id":"6845779d894b008990914576","id":"MohMounir/MLAAD-chunked","author":"MohMounir","disabled":false,"gated":false,"lastModified":"2025-06-08T11:44:34.000Z","likes":0,"trendingScore":0,"private":false,"sha":"ebb771701030cd90b2caa8cf2d1380cdb80d5376","downloads":3,"tags":["region:us"],"createdAt":"2025-06-08T11:44:29.000Z","key":""},{"_id":"684cc5484e4e79538c9d2ae6","id":"dyadd/mlai_lerobot","author":"dyadd","disabled":false,"gated":false,"lastModified":"2025-06-14T00:41:44.000Z","likes":0,"trendingScore":0,"private":false,"sha":"2201e84459659a935273ba0a03135a587b0e7cd3","downloads":6,"tags":["region:us"],"createdAt":"2025-06-14T00:41:44.000Z","key":""},{"_id":"684e08f86bc89554fbf44839","id":"mlazniewski/trialbench-combined","author":"mlazniewski","disabled":false,"gated":false,"lastModified":"2025-06-15T00:03:12.000Z","likes":0,"trendingScore":0,"private":false,"sha":"157f8800bb9bce5ca07ca48f260b97dc0171622b","description":"    # TrialBench: Clinical Trial Outcome Prediction Dataset\n\nTrialBench is a curated dataset collection designed to support machine learning research on clinical trial outcome prediction. It includes multiple tasks relevant to the analysis of trial success, safety, and patient behavior, extracted and preprocessed from publicly available clinical trial data.\n\n\t\n\t\t\n\t\tDataset Structure\n\t\n\nThis repository contains multiple configurations, each corresponding to a specific prediction task and… See the full description on the dataset page: https://huggingface.co/datasets/mlazniewski/trialbench-combined.","downloads":16,"tags":["size_categories:10K<n<100K","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-06-14T23:42:48.000Z","key":""},{"_id":"6851bb5fc672ff2bf8dacd04","id":"MLap/English-French-Portuguese-Lexicon","author":"MLap","disabled":false,"gated":false,"lastModified":"2025-10-13T15:28:13.000Z","likes":0,"trendingScore":0,"private":false,"sha":"e6cfd9c7b174b1bac13416c1a48be6d73579d44f","description":"\n\t\n\t\t\n\t\tDemo Notebook for this dataset\n\t\n\n\nDataset created with Gemini-Flash-2.5 API with the prompt given below:\n\nGenerate a list of 500 simple and commonly used English words, each translated into French and Portuguese. Format the output as CSV with the columns: English, French, Portuguese. Only include single words (no phrases or verbs starting with ‘to’, like ‘to eat’ or ‘to go’). Avoid grammatical verbs and ensure no repetitions.\n\nExtensive manual cleaning was done with the help of Google… See the full description on the dataset page: https://huggingface.co/datasets/MLap/English-French-Portuguese-Lexicon.","downloads":11,"tags":["language:en","language:fr","language:pt","license:mit","size_categories:n<1K","format:csv","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us","NLP","Embedding Similarity","High Resource Language","Low Resource Language"],"createdAt":"2025-06-17T19:00:47.000Z","key":""},{"_id":"68594aa4557b9577a017bc02","id":"mlazniewski/ct-assistant-guardrails","author":"mlazniewski","disabled":false,"gated":false,"lastModified":"2025-06-23T13:20:27.000Z","likes":0,"trendingScore":0,"private":false,"sha":"c09b43570152611b67f1aebb2a41bcacf3123f2d","description":"\n\t\n\t\t\n\t\tCT Assistant Guardrails\n\t\n\nThis dataset compiles toxic, medically inappropriate, and out-of-scope questions to train or evaluate language models specializing in clinical trial assistance.\n\n\t\n\t\t\n\t\tStructure\n\t\n\nEach entry contains:\n\nquestion: A user-style prompt\nanswers: Refusal-safe response template\ncircle: Categorization of the undesirability level\n\n\n\t\n\t\t\n\t\tCircles\n\t\n\n\nCircle_7: Unsafe, toxic, or clearly unethical requests\nCircle_6: General questions, irrelevant to clinical trials… See the full description on the dataset page: https://huggingface.co/datasets/mlazniewski/ct-assistant-guardrails.","downloads":15,"tags":["size_categories:1K<n<10K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-06-23T12:37:56.000Z","key":""},{"_id":"685c89bd0638565e8a47c9c3","id":"ronniross/ml-algorithm-dataset","author":"ronniross","disabled":false,"gated":false,"lastModified":"2026-03-14T11:32:10.000Z","likes":1,"trendingScore":0,"private":false,"sha":"df0ad524db04e650188220e5a14c589a05968000","description":"\n\t\n\t\t\n\t\tml-algorithm-dataset\n\t\n\n A conjecture of datasets specifically designed for Machine Learning training and tuning pipelines, mostly novel algorithms and their representations as RAW ASCII and LaTeX, connected to the asi-ecosystem framework.\n","downloads":14,"tags":["region:us"],"createdAt":"2025-06-25T23:43:57.000Z","key":""},{"_id":"686d738601563dc3f27e3055","id":"cmu-mlsp/DFADD_MLAAD_DiffSSD_VoxCeleb2","author":"cmu-mlsp","disabled":false,"gated":false,"lastModified":"2025-07-09T08:48:28.000Z","likes":1,"trendingScore":0,"private":false,"sha":"a195f48a8325c9cef2e8025fb4b4cbc8da7eec11","downloads":572,"tags":["size_categories:1M<n<10M","format:parquet","modality:audio","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-07-08T19:37:42.000Z","key":""},{"_id":"68714a44613b55871b7d8a0c","id":"MLap/SentiHin-2500","author":"MLap","disabled":false,"gated":false,"lastModified":"2025-10-13T15:35:16.000Z","likes":0,"trendingScore":0,"private":false,"sha":"e9689e24b7f0e9fa15792fc35598e4464f887219","description":"\n\n\t\n\t\t\n\t\tSentiHin-2500 original Hindi sentiment analysis CSV dataset, consists of 2500 rows.\n\t\n\n\nThe CSV dataset is generated using OpenAI's GPT4o and Claude 4 Sonnet with a specific prompt. For more details, visit project page on GitHub.\n\n\n\t\n\t\t\n\t\n\t\n\t\tAll the 2500 rows of this CSV dataset are manually verified for correctness.\n\t\n\nIf you use SentiHin-2500 in your research or projects, please cite it as:\n@dataset{sentiHin2025,\n  author       = {Aman Prakash},\n  title        = {SentiHin-2500}… See the full description on the dataset page: https://huggingface.co/datasets/MLap/SentiHin-2500.","downloads":21,"tags":["task_categories:text-classification","language:hi","license:cc-by-4.0","size_categories:1K<n<10K","format:csv","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us","sentiment","analysis","sentiment-analysis"],"createdAt":"2025-07-11T17:30:44.000Z","key":""},{"_id":"68761c39e775c2dbf1858aec","id":"laylarsssss/swe_v0.1_jsonl_wo_mlang_large100_wo_v0.0","author":"laylarsssss","disabled":false,"gated":false,"lastModified":"2025-07-15T09:54:39.000Z","likes":0,"trendingScore":0,"private":false,"sha":"3d87d2a92efbd70bcecc88545001009b9cab1a75","downloads":611,"tags":["region:us"],"createdAt":"2025-07-15T09:15:37.000Z","key":""},{"_id":"6881903b47f3d7284e153096","id":"hinmer/MLAsim","author":"hinmer","disabled":false,"gated":false,"lastModified":"2025-07-24T01:50:16.000Z","likes":0,"trendingScore":0,"private":false,"sha":"855913142af31bc47f0b875fe4fb7f3e2b5b7c7f","downloads":4,"tags":["region:us"],"createdAt":"2025-07-24T01:45:31.000Z","key":""},{"_id":"68873ecf0dbfb24aadb47246","id":"laylarsssss/swe_v0.1_jsonl_wo_mlang_large100_wo_v0.0_deltag","author":"laylarsssss","disabled":false,"gated":false,"lastModified":"2025-07-28T09:20:21.000Z","likes":0,"trendingScore":0,"private":false,"sha":"3b7074e8d9893c8568693d711c2cc8775f80f472","downloads":188,"tags":["size_categories:n<1K","format:json","modality:tabular","modality:text","library:datasets","library:dask","library:mlcroissant","region:us"],"createdAt":"2025-07-28T09:11:43.000Z","key":""},{"_id":"6887ea47a404b1e20e331ebc","id":"mlazniewski/mlazniewski_skin_graft","author":"mlazniewski","disabled":false,"gated":false,"lastModified":"2025-07-28T21:23:22.000Z","likes":0,"trendingScore":0,"private":false,"sha":"044d153aa90f555fa9f93c07177eef03cb536da3","downloads":8,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-07-28T21:23:19.000Z","key":""},{"_id":"688e856f2204bcdb79c9a650","id":"voxmenthe/mlabonneperfectblendConverted","author":"voxmenthe","disabled":false,"gated":false,"lastModified":"2025-08-02T21:40:05.000Z","likes":0,"trendingScore":0,"private":false,"sha":"741d9ee0769e7ef87000e492d9426315a1802c97","downloads":45,"tags":["size_categories:1M<n<10M","format:parquet","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-08-02T21:38:55.000Z","key":""},{"_id":"68952aa57df9424a33c09794","id":"Amtwakel/mlazim_pages_dataset","author":"Amtwakel","disabled":false,"gated":false,"lastModified":"2025-08-07T22:37:35.000Z","likes":0,"trendingScore":0,"private":false,"sha":"5df861997ecae7087f1df4665617d556105a5075","downloads":4,"tags":["size_categories:1K<n<10K","format:parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-08-07T22:37:25.000Z","key":""},{"_id":"689bba0f5e9d3bb7a6dbc547","id":"Amtwakel/mlazim_summaries","author":"Amtwakel","disabled":false,"gated":false,"lastModified":"2025-08-12T22:02:57.000Z","likes":0,"trendingScore":0,"private":false,"sha":"299d07aab9793750cf605145576aa41b1b859928","downloads":6,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-08-12T22:02:55.000Z","key":""},{"_id":"689bba293a96b4adbc18bcb4","id":"Amtwakel/mlazim_qa","author":"Amtwakel","disabled":false,"gated":false,"lastModified":"2025-08-12T22:03:23.000Z","likes":0,"trendingScore":0,"private":false,"sha":"cd21f75f891339a50d37a056bcfa1e45cc55e142","downloads":4,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-08-12T22:03:21.000Z","key":""},{"_id":"689bba6dbd7d96ba3c96e2b8","id":"Amtwakel/mlazim_key_points","author":"Amtwakel","disabled":false,"gated":false,"lastModified":"2025-08-12T22:04:31.000Z","likes":0,"trendingScore":0,"private":false,"sha":"b91076b64a102471d751402784a6bf0b8ac36bd8","downloads":4,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-08-12T22:04:29.000Z","key":""},{"_id":"689bc50b09cb26ac33794359","id":"Amtwakel/mlazim_biography","author":"Amtwakel","disabled":false,"gated":false,"lastModified":"2025-08-12T22:49:50.000Z","likes":0,"trendingScore":0,"private":false,"sha":"dae7bac98a7b3b5478d54dc41c8e285492a20370","downloads":4,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-08-12T22:49:47.000Z","key":""},{"_id":"68b4c25378ca52408fc609cd","id":"mlap1n/mmlu_formatted","author":"mlap1n","disabled":false,"gated":false,"lastModified":"2025-08-31T21:45:02.000Z","likes":0,"trendingScore":0,"private":false,"sha":"121d92df8dd9b1ed5c494d409d0a45463923fe0b","downloads":8,"tags":["size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2025-08-31T21:44:51.000Z","key":""},{"_id":"68cd847f82e1bf61a6532805","id":"Mlacc2339/Hello","author":"Mlacc2339","disabled":false,"gated":false,"lastModified":"2025-09-19T16:27:44.000Z","likes":0,"trendingScore":0,"private":false,"sha":"1b6af8896c05064820dfcdc8647da62f196b595b","downloads":4,"tags":["license:apache-2.0","region:us"],"createdAt":"2025-09-19T16:27:43.000Z","key":""},{"_id":"68e2a7066d58fbfa2abea9e6","id":"xxuan-speech/MCL-MLAAD","author":"xxuan-speech","disabled":false,"gated":false,"lastModified":"2025-10-13T12:25:42.000Z","likes":0,"trendingScore":0,"private":false,"sha":"68833bfd308bb0663132ca3cd234ccf3f40c529a","description":"\n\t\n\t\t\n\t\tIntroduction\n\t\n\nMCL-MLAAD is the first multilingual benchmark for speech deepfake source tracing. It spans mono- and cross-lingual protocols, includes DSP and SSL baselines, studies language-specific fine-tuning for cross-lingual generalization, and tests robustness to unseen languages/speakers. See arXiv:2508.04143.\n\n\t\n\t\t\n\t\tDownload the Dataset\n\t\n\nInstall the datasets package:  \npip install datasets\n\nLog in with your Hugging Face account:\nhuggingface-cli login\n\nLoad the dataset in… See the full description on the dataset page: https://huggingface.co/datasets/xxuan-speech/MCL-MLAAD.","downloads":524,"tags":["license:mit","arxiv:2508.04143","region:us"],"createdAt":"2025-10-05T17:12:38.000Z","key":""},{"_id":"68f9570a3c304b704796dca7","id":"AdilRumy/ML-A1","author":"AdilRumy","disabled":false,"gated":false,"lastModified":"2025-10-22T22:25:14.000Z","likes":0,"trendingScore":0,"private":false,"sha":"f9c85a0d0aee6c5057002a251477548f947fa990","downloads":2,"tags":["size_categories:1K<n<10K","format:text","modality:image","modality:text","library:datasets","library:mlcroissant","region:us"],"createdAt":"2025-10-22T22:13:30.000Z","key":""},{"_id":"68fa87a4d67001dcf0d27717","id":"itsmuriuki/ml-arxiv-instruct","author":"itsmuriuki","disabled":false,"gated":false,"lastModified":"2025-10-24T18:38:14.000Z","likes":0,"trendingScore":0,"private":false,"sha":"a498ca2dba0c14cc46744d7e31845db6b7797343","downloads":4,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-10-23T19:53:08.000Z","key":""},{"_id":"68fa8f9001f015ddf3e19b7b","id":"itsmuriuki/ml-arxiv-dpo","author":"itsmuriuki","disabled":false,"gated":false,"lastModified":"2025-10-24T18:49:32.000Z","likes":0,"trendingScore":0,"private":false,"sha":"d1463b15a01e0a274550ea2eb4ac138c33480117","downloads":3,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-10-23T20:26:56.000Z","key":""},{"_id":"691ac9d804e393a9d93f2fa1","id":"DiaoYiya/MLAE_NUSWIDE","author":"DiaoYiya","disabled":false,"gated":false,"lastModified":"2025-11-17T07:08:08.000Z","likes":0,"trendingScore":0,"private":false,"sha":"cddd681a860f0aa6b242f2d963c30c834670c3dc","downloads":7,"tags":["license:apache-2.0","region:us"],"createdAt":"2025-11-17T07:08:08.000Z","key":""},{"_id":"69404a177f753d0d628671f6","id":"agentlans/mlabonne-open-perfectblend","author":"agentlans","disabled":false,"gated":false,"lastModified":"2025-12-15T17:51:24.000Z","likes":0,"trendingScore":0,"private":false,"sha":"90041a84e20ffd6a4f0e1617ad6337190113b847","downloads":62,"tags":["license:apache-2.0","size_categories:1M<n<10M","format:json","modality:text","library:datasets","library:pandas","library:mlcroissant","library:polars","region:us"],"createdAt":"2025-12-15T17:49:11.000Z","key":""},{"_id":"6956fe38607c867c5c00b1fb","id":"mueller91/MLAAD-tiny","author":"mueller91","disabled":false,"gated":false,"lastModified":"2026-05-27T21:55:40.000Z","likes":3,"trendingScore":0,"private":false,"sha":"9143e5ea709575ebab6bec52840a1043aada7bb1","description":"\n\t\n\t\t\n\t\tWelcome to MLAAD-tiny\n\t\n\nMLAAD-tiny is a very small subset of the full MLAAD dataset, designed for education, prototyping, and debugging.\nMany teaching environments (e.g. Colab, Kaggle, university notebooks -- se this notebook for example) impose strict storage limits, which makes large-scale audio deepfake datasets impractical to use. To address this, we provide MLAAD-tiny, a compact yet representative version of MLAAD.\n\n\t\n\t\t\n\t\n\t\n\t\tDownload\n\t\n\ngit lfs install\ngit clone… See the full description on the dataset page: https://huggingface.co/datasets/mueller91/MLAAD-tiny.","downloads":2898,"tags":["task_categories:audio-classification","language:en","license:cc-by-nc-4.0","size_categories:10K<n<100K","format:audiofolder","modality:audio","library:datasets","library:mlcroissant","region:us","deepfake","deepfake-detection","audio-deepfake","audio-deepfake-detection","mlaad"],"createdAt":"2026-01-01T23:07:36.000Z","key":""},{"_id":"6961213842c82235dec064ae","id":"mlaszlo/mrz-dataset","author":"mlaszlo","disabled":false,"gated":false,"lastModified":"2026-01-09T15:45:10.000Z","likes":0,"trendingScore":0,"private":false,"sha":"bfe68f6886537f8c63880b96a8edd69e3624ba5e","downloads":16,"tags":["size_categories:1K<n<10K","format:parquet","modality:image","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-01-09T15:39:36.000Z","key":""},{"_id":"6961c5757e8350085b19f21b","id":"watate/ml-act-data","author":"watate","disabled":false,"gated":false,"lastModified":"2026-01-10T03:20:25.000Z","likes":0,"trendingScore":0,"private":false,"sha":"9af522a965aea4e218a78c2b68cdbd25d4a0d801","downloads":3,"tags":["region:us"],"createdAt":"2026-01-10T03:20:21.000Z","key":""},{"_id":"697e92e1440ea4e00c954cda","id":"MLARG/NBLS-1K","author":"MLARG","disabled":false,"gated":false,"lastModified":"2026-02-01T18:01:25.000Z","likes":0,"trendingScore":0,"private":false,"sha":"27e83127d25ccbc978bf74a3db829f606e6a1618","downloads":8,"tags":["region:us"],"createdAt":"2026-01-31T23:40:17.000Z","key":""},{"_id":"697f8b412f09c7aacaf1ef66","id":"MLARG/NBLS-2K","author":"MLARG","disabled":false,"gated":false,"lastModified":"2026-02-02T19:24:00.000Z","likes":0,"trendingScore":0,"private":false,"sha":"43b1b3a7de2583b816d2ca2dd49d0ea510c8ae49","downloads":1071,"tags":["region:us"],"createdAt":"2026-02-01T17:20:01.000Z","key":""},{"_id":"6980cc2d69577c8304a41558","id":"Dcroix/MLAssets","author":"Dcroix","disabled":false,"gated":false,"lastModified":"2026-02-18T08:34:01.000Z","likes":0,"trendingScore":0,"private":false,"sha":"883add29c06a4282fd939efd5955db56b4610105","downloads":4,"tags":["license:agpl-3.0","size_categories:n<1K","format:imagefolder","modality:image","library:datasets","library:mlcroissant","region:us"],"createdAt":"2026-02-02T16:09:17.000Z","key":""},{"_id":"698df67b8e755fb1620c2cc7","id":"mlahmy/seal-rag-evaluation-data","author":"mlahmy","disabled":false,"gated":false,"lastModified":"2026-02-12T15:54:34.000Z","likes":0,"trendingScore":0,"private":false,"sha":"0ab5b8bf24ccabb8674c2d72f2df3bb831291a67","description":"\n\t\n\t\t\n\t\tSEAL-RAG Evaluation Data\n\t\n\nThis repository contains the evaluation slices used in the paper Replace, Don't Expand: Mitigating Context Dilution in Multi-Hop RAG.\n\n\t\n\t\t\n\t\tFiles\n\t\n\n\nhotpotqa_1000_v1.csv: The 1,000-sample slice from HotpotQA used for the primary evaluation.\n2WikiMultihopQA_200_v1.csv: The 200-sample slice from 2WikiMultihopQA used for additional testing.\n\n\n\t\n\t\t\n\t\tCitation\n\t\n\nIf you use this data, please cite the original paper.\n","downloads":42,"tags":["task_categories:question-answering","language:en","license:mit","arxiv:2512.10787","region:us","rag","evaluation","hotpotqa"],"createdAt":"2026-02-12T15:49:15.000Z","key":""},{"_id":"699450b7c80384e845229c0f","id":"mlaszlo/mrz-simpo-dataset","author":"mlaszlo","disabled":false,"gated":false,"lastModified":"2026-02-17T11:27:57.000Z","likes":0,"trendingScore":0,"private":false,"sha":"eb2f62b26932680b92c6acbc0f5365276db47800","downloads":10,"tags":["size_categories:1K<n<10K","format:parquet","format:optimized-parquet","modality:image","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-02-17T11:27:51.000Z","key":""},{"_id":"69ba6f915fb0538dfe423cde","id":"xxuan-speech/MLAAD_Protocols_for_Source-Speaker_Disentanglement_Research","author":"xxuan-speech","disabled":false,"gated":false,"lastModified":"2026-03-18T09:28:49.000Z","likes":0,"trendingScore":0,"private":false,"sha":"a854fb06a76ceb3da6a8a6e32142747ead5f6e2c","downloads":14,"tags":["size_categories:100K<n<1M","format:text","modality:text","library:datasets","library:mlcroissant","region:us"],"createdAt":"2026-03-18T09:25:37.000Z","key":""},{"_id":"69de8fcb9c09f4521371cc7b","id":"34data/v14-MLAAD-Fake-part_01","author":"34data","disabled":false,"gated":false,"lastModified":"2026-04-14T19:05:02.000Z","likes":0,"trendingScore":0,"private":false,"sha":"a1b2fda06597adb3907050aac1d5edb866a9a26a","downloads":28,"tags":["size_categories:1K<n<10K","format:audiofolder","modality:audio","library:datasets","library:mlcroissant","region:us"],"createdAt":"2026-04-14T19:04:43.000Z","key":""},{"_id":"69de8fe0096a251ce8b13fcb","id":"34data/v14-MLAAD-Fake-part_02","author":"34data","disabled":false,"gated":false,"lastModified":"2026-04-14T19:05:16.000Z","likes":0,"trendingScore":0,"private":false,"sha":"aaf6d4751a1558a36b04191a05f24c2b8efdc3ad","downloads":15,"tags":["size_categories:1K<n<10K","format:audiofolder","modality:audio","library:datasets","library:mlcroissant","region:us"],"createdAt":"2026-04-14T19:05:04.000Z","key":""},{"_id":"69de8ff5b5f4b2e0167a94fe","id":"34data/v14-MLAAD-Fake-part_03","author":"34data","disabled":false,"gated":false,"lastModified":"2026-04-14T19:05:43.000Z","likes":0,"trendingScore":0,"private":false,"sha":"cf9c54512479e84e4247424b272e871bd47f463e","downloads":18,"tags":["size_categories:1K<n<10K","format:audiofolder","modality:audio","library:datasets","library:mlcroissant","region:us"],"createdAt":"2026-04-14T19:05:25.000Z","key":""},{"_id":"69de9009f8fa32e682e99d37","id":"34data/v14-MLAAD-Fake-part_04","author":"34data","disabled":false,"gated":false,"lastModified":"2026-04-14T19:06:00.000Z","likes":0,"trendingScore":0,"private":false,"sha":"e8a80bb7a203a28a4cdd45de835ba7a5514e7f3c","downloads":17,"tags":["size_categories:1K<n<10K","format:audiofolder","modality:audio","library:datasets","library:mlcroissant","region:us"],"createdAt":"2026-04-14T19:05:45.000Z","key":""},{"_id":"69de901c552376f75d10031f","id":"34data/v14-MLAAD-Fake-part_05","author":"34data","disabled":false,"gated":false,"lastModified":"2026-04-14T19:06:14.000Z","likes":0,"trendingScore":0,"private":false,"sha":"05d0bffea1f8af81d04ca7f452b4c7bbf3c08fc7","downloads":23,"tags":["region:us"],"createdAt":"2026-04-14T19:06:04.000Z","key":""},{"_id":"69de90341bdd25b2747f09c4","id":"34data/v14-MLAAD-Fake-part_06","author":"34data","disabled":false,"gated":false,"lastModified":"2026-04-14T19:06:41.000Z","likes":0,"trendingScore":0,"private":false,"sha":"9c6f1e694727463a7265bceeee33619d2ab1fdf5","downloads":16,"tags":["size_categories:1K<n<10K","format:audiofolder","modality:audio","library:datasets","library:mlcroissant","region:us"],"createdAt":"2026-04-14T19:06:28.000Z","key":""},{"_id":"69de904a07f4e6a31f729800","id":"34data/v14-MLAAD-Fake-part_07","author":"34data","disabled":false,"gated":false,"lastModified":"2026-04-14T19:07:06.000Z","likes":0,"trendingScore":0,"private":false,"sha":"6ee391650155124ed35dd1b977f547c43b96e891","downloads":18,"tags":["size_categories:1K<n<10K","format:audiofolder","modality:audio","library:datasets","library:mlcroissant","region:us"],"createdAt":"2026-04-14T19:06:50.000Z","key":""},{"_id":"69de905cf5f5a426fc1d0c34","id":"34data/v14-MLAAD-Fake-part_08","author":"34data","disabled":false,"gated":false,"lastModified":"2026-04-14T19:07:19.000Z","likes":0,"trendingScore":0,"private":false,"sha":"5872ecfe82434acde036bfe9aabcd09935b2afe1","downloads":17,"tags":["size_categories:1K<n<10K","format:audiofolder","modality:audio","library:datasets","library:mlcroissant","region:us"],"createdAt":"2026-04-14T19:07:08.000Z","key":""},{"_id":"69de907303a903ee698aa37b","id":"34data/v14-MLAAD-Fake-part_09","author":"34data","disabled":false,"gated":false,"lastModified":"2026-04-14T19:07:41.000Z","likes":0,"trendingScore":0,"private":false,"sha":"1376c752b480ce5f3e1206da3cf2b97177e6c282","downloads":15,"tags":["size_categories:1K<n<10K","format:audiofolder","modality:audio","library:datasets","library:mlcroissant","region:us"],"createdAt":"2026-04-14T19:07:31.000Z","key":""},{"_id":"69de9086f5f5a426fc1d0fba","id":"34data/v14-MLAAD-Fake-part_10","author":"34data","disabled":false,"gated":false,"lastModified":"2026-04-14T19:08:08.000Z","likes":0,"trendingScore":0,"private":false,"sha":"2dd32fcc5e11a3cf6ff5dbb710038f03a78ff4fa","downloads":17,"tags":["size_categories:1K<n<10K","format:audiofolder","modality:audio","library:datasets","library:mlcroissant","region:us"],"createdAt":"2026-04-14T19:07:50.000Z","key":""},{"_id":"69e46724bf20d3a18feb72a6","id":"abhid1234/ml-advisor-benchmark","author":"abhid1234","disabled":false,"gated":false,"lastModified":"2026-04-19T05:24:55.000Z","likes":0,"trendingScore":0,"private":false,"sha":"c523513a26babca9b554f335dd4603f62820e06c","description":"\n\t\n\t\t\n\t\tML Experiment Advisor Benchmark\n\t\n\nA 30-task benchmark for evaluating how well a language-model agent can advise on ML hyperparameter tuning, given experiment history and source code. Derived from 16 real training runs of Karpathy's autoresearch on an A40 GPU, plus 5 synthetic extensions covering edge cases.\nBuilt for the meta-agent-improver project.\n\n\t\n\t\t\n\t\n\t\n\t\tWhat's in each task\n\t\n\nEvery task is a workspace containing:\n\nresults.tsv — experiment history up to that point (commit… See the full description on the dataset page: https://huggingface.co/datasets/abhid1234/ml-advisor-benchmark.","downloads":83,"tags":["task_categories:text-generation","license:mit","size_categories:n<1K","region:us","ml","hyperparameter-tuning","agent-benchmark","prompt-engineering","meta-agent"],"createdAt":"2026-04-19T05:24:52.000Z","key":""},{"_id":"69ebcc86b4bb807c0406e7b5","id":"Saisaket25/MLAAD-tiny","author":"Saisaket25","disabled":false,"gated":false,"lastModified":"2026-04-24T20:03:19.000Z","likes":0,"trendingScore":0,"private":false,"sha":"709195a6673d5a0de27217818f0befa8ce4f68cc","description":"\n\t\n\t\t\n\t\tWelcome to MLAAD-tiny\n\t\n\nMLAAD-tiny is a very small subset of the full MLAAD dataset, designed for education, prototyping, and debugging.\nMany teaching environments (e.g. Colab, Kaggle, university notebooks) impose strict storage limits, which makes large-scale audio deepfake datasets impractical to use. To address this, we provide MLAAD-tiny, a compact yet representative version of MLAAD.\n\n\t\n\t\t\n\t\n\t\n\t\tDownload\n\t\n\ngit lfs install\ngit clone… See the full description on the dataset page: https://huggingface.co/datasets/Saisaket25/MLAAD-tiny.","downloads":10,"tags":["task_categories:audio-classification","language:en","license:cc-by-nc-4.0","size_categories:10K<n<100K","region:us","deepfake","deepfake-detection","audio-deepfake","audio-deepfake-detection","mlaad"],"createdAt":"2026-04-24T20:03:18.000Z","key":""},{"_id":"69ecd2c2d0ea0cd027d98c7a","id":"ligaments-dev/ml-agent-dataset-638347b0","author":"ligaments-dev","disabled":false,"gated":false,"lastModified":"2026-04-25T14:42:13.000Z","likes":0,"trendingScore":0,"private":false,"sha":"538ccbeb5c28689b60906b3c6d639f0537d6abc9","downloads":3,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-04-25T14:42:10.000Z","key":""},{"_id":"69ecd2c7e2a3a8fd83e9a843","id":"ligaments-dev/ml-agent-dataset-ebc73898","author":"ligaments-dev","disabled":false,"gated":false,"lastModified":"2026-04-25T14:42:17.000Z","likes":0,"trendingScore":0,"private":false,"sha":"470cde4d38ace770876fdc361ac4cb1e22b262e4","downloads":2,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-04-25T14:42:15.000Z","key":""},{"_id":"69ecd2ca13df8335e84600f7","id":"ligaments-dev/ml-agent-dataset-30a02bc8","author":"ligaments-dev","disabled":false,"gated":false,"lastModified":"2026-04-25T14:42:23.000Z","likes":0,"trendingScore":0,"private":false,"sha":"9180fa19c7cf4bf86f41808164da20dc2c7e1260","downloads":2,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-04-25T14:42:18.000Z","key":""},{"_id":"69ecd2d1736ed8e3f90b9eef","id":"ligaments-dev/ml-agent-dataset-2dff28b2","author":"ligaments-dev","disabled":false,"gated":false,"lastModified":"2026-04-25T14:42:27.000Z","likes":0,"trendingScore":0,"private":false,"sha":"fd2064cef5652c6d2ba29f6387efe038890d6544","downloads":4,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-04-25T14:42:25.000Z","key":""},{"_id":"69ecd2d5d0ea0cd027d98e8b","id":"ligaments-dev/ml-agent-dataset-d69b2c9f","author":"ligaments-dev","disabled":false,"gated":false,"lastModified":"2026-04-25T14:42:32.000Z","likes":0,"trendingScore":0,"private":false,"sha":"9a0ba05b99281ddf60b09fa795b36dc500a4c68a","downloads":4,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-04-25T14:42:29.000Z","key":""},{"_id":"69ecd2d988524a7ee687c1f0","id":"ligaments-dev/ml-agent-dataset-9bbf0af3","author":"ligaments-dev","disabled":false,"gated":false,"lastModified":"2026-04-25T14:42:36.000Z","likes":0,"trendingScore":0,"private":false,"sha":"b8daa6b16347c4708ee8d87384dc553582973cc8","downloads":2,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-04-25T14:42:33.000Z","key":""},{"_id":"69ecd2dfd37fc63c770f6b24","id":"ligaments-dev/ml-agent-dataset-7451917a","author":"ligaments-dev","disabled":false,"gated":false,"lastModified":"2026-04-25T14:42:41.000Z","likes":0,"trendingScore":0,"private":false,"sha":"d397a0a19363024b00267331ea7be094c0385a73","downloads":4,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-04-25T14:42:39.000Z","key":""},{"_id":"69ecd314debd7e7a90b31d3a","id":"ligaments-dev/ml-agent-dataset-cd5bbb77","author":"ligaments-dev","disabled":false,"gated":false,"lastModified":"2026-04-25T14:43:34.000Z","likes":0,"trendingScore":0,"private":false,"sha":"277af58ed3cd3f60657aa19ce1f432b618e8fa5c","downloads":2,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-04-25T14:43:32.000Z","key":""},{"_id":"69ecd31aeb28bdece217a673","id":"ligaments-dev/ml-agent-dataset-cf9d5f74","author":"ligaments-dev","disabled":false,"gated":false,"lastModified":"2026-04-25T14:43:40.000Z","likes":0,"trendingScore":0,"private":false,"sha":"2755335f96ee8140941ac2cc8622a855079efb6e","downloads":3,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-04-25T14:43:38.000Z","key":""},{"_id":"69ecd31e32455be9c11dcd7c","id":"ligaments-dev/ml-agent-dataset-3eff78c0","author":"ligaments-dev","disabled":false,"gated":false,"lastModified":"2026-04-25T14:43:44.000Z","likes":0,"trendingScore":0,"private":false,"sha":"84bab660499012770d19b00206845282cf68daae","downloads":2,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-04-25T14:43:42.000Z","key":""},{"_id":"69ecd32256f14d943d14c1d6","id":"ligaments-dev/ml-agent-dataset-a5a88f5a","author":"ligaments-dev","disabled":false,"gated":false,"lastModified":"2026-04-25T14:43:48.000Z","likes":0,"trendingScore":0,"private":false,"sha":"2fcc12fae6f5f3bc5203d2da6dbafd11604efad8","downloads":6,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-04-25T14:43:46.000Z","key":""},{"_id":"69ecd32574f91d5807f09573","id":"ligaments-dev/ml-agent-dataset-1b7ad3ef","author":"ligaments-dev","disabled":false,"gated":false,"lastModified":"2026-04-25T14:43:54.000Z","likes":0,"trendingScore":0,"private":false,"sha":"5f6881fbbc1fabd3d99d08ce388d45015f363ed6","downloads":6,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-04-25T14:43:49.000Z","key":""},{"_id":"69ecd32c88524a7ee687cec0","id":"ligaments-dev/ml-agent-dataset-8ab33ac5","author":"ligaments-dev","disabled":false,"gated":false,"lastModified":"2026-04-25T14:43:58.000Z","likes":0,"trendingScore":0,"private":false,"sha":"4dd3c0f1f28d2e4bfbefbc64cb789fa5ef39a45d","downloads":2,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-04-25T14:43:56.000Z","key":""},{"_id":"69ecd330605cd817dc71b2f5","id":"ligaments-dev/ml-agent-dataset-3078691b","author":"ligaments-dev","disabled":false,"gated":false,"lastModified":"2026-04-25T14:44:03.000Z","likes":0,"trendingScore":0,"private":false,"sha":"da1d63fd82a0e21391a6d81a5eb80def2bbb4070","downloads":0,"tags":["size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-04-25T14:44:00.000Z","key":""},{"_id":"69f030655bcf72bbd90f8df0","id":"hoangminh1110/ML_Assignment252_Feature","author":"hoangminh1110","disabled":false,"gated":false,"lastModified":"2026-04-28T11:12:49.000Z","likes":0,"trendingScore":0,"private":false,"sha":"563c65358523ff466b18c72b22bb26f82e3a30c4","downloads":2,"tags":["region:us"],"createdAt":"2026-04-28T03:58:29.000Z","key":""},{"_id":"69f0ebeb436258f57e3b8db0","id":"mlai-dante/painter-creation","author":"mlai-dante","disabled":false,"gated":false,"lastModified":"2026-05-09T20:58:43.000Z","likes":0,"trendingScore":0,"private":false,"sha":"5f52d031bb776f4f07a81c9e9ef490f4c67081c5","downloads":3,"tags":["size_categories:n<1K","format:imagefolder","modality:image","library:datasets","library:mlcroissant","region:us"],"createdAt":"2026-04-28T17:18:35.000Z","key":""},{"_id":"69f5f35e1aff7e6eedb48579","id":"anonymous-submission-nips/mlaire-xquad","author":"anonymous-submission-nips","disabled":false,"gated":false,"lastModified":"2026-05-06T17:32:08.000Z","likes":0,"trendingScore":0,"private":false,"sha":"75852d77d79e10ec59c04a2007cfd74647f41e9d","description":"\n\t\n\t\t\n\t\tMLAIRE-XQUAD\n\t\n\nXQuAD reformatted for language-aware retrieval evaluation. Each passage appears once per language; relevance is encoded by group_id matching.\nThis repository is part of the MLAIRE benchmark, submitted anonymously\nto the NeurIPS 2026 Evaluations & Datasets Track. Authors and affiliations\nare withheld for double-blind review.\n\n\t\n\t\t\n\t\tDefault top-k\n\t\n\nReported metrics in the paper use top-20.\n\n\t\n\t\t\n\t\tLayout\n\t\n\ncorpus/test-*.parquet      _id, text, title, language, group_id… See the full description on the dataset page: https://huggingface.co/datasets/anonymous-submission-nips/mlaire-xquad.","downloads":22,"tags":["task_categories:text-retrieval","multilinguality:multilingual","language:en","language:es","language:de","language:el","language:ru","language:tr","language:ar","language:vi","language:th","language:zh","language:hi","language:ro","license:cc-by-sa-4.0","size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us","retrieval","mlaire","multilingual"],"createdAt":"2026-05-02T12:51:42.000Z","key":""},{"_id":"69f5f36852f75c3da495ac11","id":"anonymous-submission-nips/mlaire-mlqa","author":"anonymous-submission-nips","disabled":false,"gated":false,"lastModified":"2026-05-06T17:31:27.000Z","likes":0,"trendingScore":0,"private":false,"sha":"d2720e10f3244a88b63bfca65c81d225da1082b7","description":"\n\t\n\t\t\n\t\tMLAIRE-MLQA\n\t\n\nMLQA reformatted for language-aware retrieval evaluation. Passages are deduplicated at the context level via union-find on the original MLQA ids. Relevance is encoded by group_id matching.\nThis repository is part of the MLAIRE benchmark, submitted anonymously\nto the NeurIPS 2026 Evaluations & Datasets Track. Authors and affiliations\nare withheld for double-blind review.\n\n\t\n\t\t\n\t\tDefault top-k\n\t\n\nReported metrics in the paper use top-20.\n\n\t\n\t\t\n\t\tLayout… See the full description on the dataset page: https://huggingface.co/datasets/anonymous-submission-nips/mlaire-mlqa.","downloads":14,"tags":["task_categories:text-retrieval","multilinguality:multilingual","language:en","language:ar","language:de","language:es","language:hi","language:vi","language:zh","license:cc-by-sa-3.0","size_categories:100K<n<1M","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us","retrieval","multilingual","mlaire"],"createdAt":"2026-05-02T12:51:52.000Z","key":""},{"_id":"69f5f4bab73e584f37b6858b","id":"anonymous-submission-nips/mlaire-belebele","author":"anonymous-submission-nips","disabled":false,"gated":false,"lastModified":"2026-05-06T17:30:51.000Z","likes":0,"trendingScore":0,"private":false,"sha":"9ce048dd8f283912b1da94295a14a9d63c4dc5a5","description":"\n\t\n\t\t\n\t\tMLAIRE-BELEBELE\n\t\n\nBelebele reformatted for language-aware retrieval evaluation. 488 underlying passages, each available in 122 languages (joined globally by the original link field). Relevance is encoded by group_id matching.\nThis repository is part of the MLAIRE benchmark, submitted anonymously\nto the NeurIPS 2026 Evaluations & Datasets Track. Authors and affiliations\nare withheld for double-blind review.\n\n\t\n\t\t\n\t\n\t\n\t\tDefault top-k\n\t\n\nReported metrics in the paper use top-200.… See the full description on the dataset page: https://huggingface.co/datasets/anonymous-submission-nips/mlaire-belebele.","downloads":88,"tags":["task_categories:text-retrieval","multilinguality:multilingual","language:acm","language:af","language:am","language:apc","language:ar","language:ars","language:ary","language:arz","language:as","language:az","language:bg","language:bm","language:bn","language:bo","language:ca","language:ceb","language:ckb","language:cs","language:da","language:de","language:el","language:en","language:es","language:et","language:eu","language:fa","language:fi","language:fr","language:fuv","language:gaz","language:gn","language:gu","language:ha","language:he","language:hi","language:hr","language:ht","language:hu","language:hy","language:id","language:ig","language:ilo","language:is","language:it","language:ja","language:jv","language:ka","language:kac","language:kea","language:kk","language:km","language:kn","language:ko","language:ky","language:lg","language:ln","language:lo","language:lt","language:luo","language:lv","language:mg","language:mi","language:mk","language:ml","language:mn","language:mr","language:ms","language:mt","language:my","language:nb","language:ne","language:nl","language:nso","language:ny","language:or","language:pa","language:pl","language:ps","language:pt","language:ro","language:ru","language:rw","language:sd","language:shn","language:si","language:sk","language:sl","language:sn","language:so","language:sq","language:sr","language:ss","language:st","language:su","language:sv","language:sw","language:ta","language:te","language:tg","language:th","language:ti","language:tl","language:tn","language:tr","language:ts","language:uk","language:ur","language:uz","language:vi","language:war","language:wo","language:xh","language:yo","language:zh","language:zu","license:cc-by-sa-4.0","size_categories:10M<n<100M","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us","retrieval","multilingual","language-aware-ir","mlaire"],"createdAt":"2026-05-02T12:57:30.000Z","key":""},{"_id":"6a059ff82a40e5696f66984f","id":"ligaments-dev/ml-agent-data-869cb60a","author":"ligaments-dev","disabled":false,"gated":false,"lastModified":"2026-05-14T10:12:09.000Z","likes":1,"trendingScore":0,"private":false,"sha":"91437f2b7b367e33de43d43eb39b9a31de805cbe","downloads":4,"tags":["size_categories:n<1K","format:json","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-05-14T10:12:08.000Z","key":""},{"_id":"6a106563fc176e49e039c93a","id":"moonscape-software/MLAAD_Audit","author":"moonscape-software","disabled":false,"gated":"manual","lastModified":"2026-05-22T15:25:04.000Z","likes":0,"trendingScore":0,"private":false,"sha":"431bfe61cab7bdaa39be4f3fa9d5d25dd531c8dc","description":"\n\t\n\t\t\n\t\tMLAAD — SSA Acoustic Feature Audit\n\t\n\nMoonscape Software | Synthetic Speech Atlas\nResearch audit contribution to the MLAAD dataset team\n\n\n\t\n\t\t\n\t\tOverview\n\t\n\nThis repository contains acoustic feature measurements extracted from the\nMLAAD (Multilingual Audio Anti-Spoofing Dataset) corpus by the Moonscape\nSynthetic Speech Atlas (SSA) pipeline.\n298,000 rows. 152 columns. One row per MLAAD clip.\nNo audio files are included. Each row contains classical signal processing\nand biomechanical… See the full description on the dataset page: https://huggingface.co/datasets/moonscape-software/MLAAD_Audit.","downloads":7,"tags":["task_categories:audio-classification","language:multilingual","size_categories:100K<n<1M","modality:audio","region:us","audio","deepfake-detection","speech","synthetic-speech","anti-spoofing","acoustic-features","biomechanics","tts","mlaad"],"createdAt":"2026-05-22T14:17:07.000Z","key":""},{"_id":"6a38ba33c07930e399b9fa4a","id":"DuoNeural/ml-ai-engineer-sft","author":"DuoNeural","disabled":false,"gated":false,"lastModified":"2026-06-22T04:29:40.000Z","likes":1,"trendingScore":0,"private":false,"sha":"fcb0c542a957e8f440eeb1b9a27b4d7baee16d9f","description":"\n\t\n\t\t\n\t\n\t\n\t\tDuoNeural ML/AI Engineer SFT Dataset\n\t\n\nA synthetic instruction-tuning dataset for training an LLM to be a useful pairing partner on ML/AI engineering work — debugging training runs, reasoning about architecture and infra choices, reviewing experiment design, and explaining core ML concepts with the specificity of someone who's actually run the experiments.\n\n\t\n\t\t\n\t\n\t\n\t\tWhy this dataset exists\n\t\n\nMost general instruction-tuning data treats ML engineering questions the same as any… See the full description on the dataset page: https://huggingface.co/datasets/DuoNeural/ml-ai-engineer-sft.","downloads":60,"tags":["task_categories:text-generation","task_categories:question-answering","language:en","license:cc-by-4.0","size_categories:1K<n<10K","format:json","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us","synthetic","machine-learning","ml-engineering","instruction-tuning","sft"],"createdAt":"2026-06-22T04:29:39.000Z","key":""},{"_id":"6a4291e179b302953527cf6b","id":"mlai-dante/gasr-ds","author":"mlai-dante","disabled":false,"gated":false,"lastModified":"2026-06-29T16:12:20.000Z","likes":0,"trendingScore":0,"private":false,"sha":"c394cf9d873c15facd85035ec90dd1330db0877d","downloads":3,"tags":["region:us"],"createdAt":"2026-06-29T15:40:17.000Z","key":""},{"_id":"6a4731824986c60fa8f1356d","id":"tepirale/open-perfectblend-deduplicate-mlabonne","author":"tepirale","disabled":false,"gated":false,"lastModified":"2026-07-03T05:52:51.000Z","likes":0,"trendingScore":0,"private":false,"sha":"30c4dfacf24bb69657b5b04986a87cc0cc69fbb0","description":"\n\t\n\t\t\n\t\n\t\n\t\tdataset\n\t\n\n\nThis dataset is from: https://huggingface.co/datasets/mlabonne/open-perfectblend\n\nThe dataset was deduplicated and empty conversations were removed.\n\nStarting rows of the dataset: '1,420,909' - ending rows of the dataset: '569,314'\n\n\n\n\t\n\t\t\nsource\noriginal\ndedup\nret.%\n\n\n\t\t\nmeta-math/MetaMathQA\n386,043\n26,600\n6.9%\n\n\nopenbmb/UltraInteract_sft\n272,266\n57,984\n21.3%\n\n\nmlabonne/ultrachat_200k_sft\n207,865\n185,667\n89.3%\n\n\nHuggingFaceH4/orca-math-word-problems-200k\n199,707\n71,905… See the full description on the dataset page: https://huggingface.co/datasets/tepirale/open-perfectblend-deduplicate-mlabonne.","downloads":19,"tags":["task_categories:text-generation","language:en","license:apache-2.0","size_categories:100K<n<1M","format:parquet","format:optimized-parquet","modality:text","library:datasets","library:dask","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-07-03T03:50:26.000Z","key":""},{"_id":"6a519f5c32bfea14be5f4305","id":"mlai-dante/gasr-waxal-omni-data","author":"mlai-dante","disabled":false,"gated":false,"lastModified":"2026-07-11T01:59:34.000Z","likes":0,"trendingScore":0,"private":false,"sha":"36b3e52d1056f67c87f098af5358e17d3db7f88d","downloads":14,"tags":["size_categories:10K<n<100K","format:parquet","modality:text","library:datasets","library:dask","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-07-11T01:41:48.000Z","key":""},{"_id":"6a58381ed765baef752fd59c","id":"binzhango/ml-agent-repro-17897-results","author":"binzhango","disabled":false,"gated":false,"lastModified":"2026-07-16T01:48:33.000Z","likes":0,"trendingScore":0,"private":false,"sha":"0682889e1191be1303a5bc07f1aca6997c033868","description":"\n\t\n\t\t\n\t\n\t\n\t\tML-Agent ICML 2026 reproduction (#17897)\n\t\n\nThis bundle audits and partially reproduces the five requested claims for\n\"ML-Agent: Reinforcing LLM Agents for Autonomous Machine Learning Engineering\"\n(OpenReview kcPPWaoegr, arXiv 2505.23723).\nThe headline benchmark cannot be independently rerun from the public release:\nthe official repository still withholds the trained checkpoint, evaluation\ncode, RL code, and training trajectories. The bundle therefore separates:\n\nsource… See the full description on the dataset page: https://huggingface.co/datasets/binzhango/ml-agent-repro-17897-results.","downloads":47,"tags":["region:us","icml2026-repro","paper-kcPPWaoegr","ml-agent","reproducibility"],"createdAt":"2026-07-16T01:47:10.000Z","key":""},{"_id":"6a58633fd9c9280d246ae169","id":"alon-albalak/ml-ai-questions-llm-judge","author":"alon-albalak","disabled":false,"gated":false,"lastModified":"2026-07-16T04:51:19.000Z","likes":0,"trendingScore":0,"private":false,"sha":"db05b46e8347c7487f6047c15b05bc099fa0afa2","downloads":7,"tags":["size_categories:n<1K","format:parquet","format:optimized-parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-07-16T04:51:11.000Z","key":""},{"_id":"6a59eedd005c81efd938b67e","id":"danil-e/harbor-datasets-mlab","author":"danil-e","disabled":false,"gated":false,"lastModified":"2026-08-27T13:38:49.000Z","likes":0,"trendingScore":0,"private":false,"sha":"c057bfe8f95cc6d1342136554928afb14383ec7b","description":"\n\t\n\t\t\n\t\n\t\n\t\tharbor-datasets-mlab\n\t\n\nHarbor-format MLAgentBench (arXiv:2310.03302), posed under the benchmark's own\nprotocol: every workspace already contains a train.py that runs end to end but\nscores poorly, and the task is to improve it — not to solve the problem from an\nempty directory. Task descriptions are MLAgentBench's research_problem.txt verbatim.\n10 tasks across tabular / text / vision / graph / segmentation / algorithmic. Agent-neutral\nand self-contained: registry.json at the root… See the full description on the dataset page: https://huggingface.co/datasets/danil-e/harbor-datasets-mlab.","downloads":68,"tags":["arxiv:2310.03302","region:us"],"createdAt":"2026-07-17T08:59:09.000Z","key":""},{"_id":"6a631f5f809b38bd07addad2","id":"MLamateur/trackingset","author":"MLamateur","disabled":false,"gated":false,"lastModified":"2026-07-24T08:46:10.000Z","likes":0,"trendingScore":0,"private":false,"sha":"e330f26092a99d9c2fb8dcd0ac77e9f7150ce164","downloads":94,"tags":["size_categories:10K<n<100K","format:imagefolder","modality:image","library:datasets","library:mlcroissant","region:us"],"createdAt":"2026-07-24T08:16:31.000Z","key":""},{"_id":"6a6d25642a4cb555f50ff599","id":"lucas1026/mlagentgym-train-tasks","author":"lucas1026","disabled":false,"gated":false,"lastModified":"2026-07-31T23:58:53.000Z","likes":0,"trendingScore":0,"private":false,"sha":"a50cf535660083b8d04e8cae935245b167782464","downloads":1321,"tags":["region:us"],"createdAt":"2026-07-31T22:44:52.000Z","key":""},{"_id":"6a6d3050fac73d4697580e28","id":"lucas1026/mlagentgym-eval-tasks","author":"lucas1026","disabled":false,"gated":false,"lastModified":"2026-08-01T00:31:56.000Z","likes":0,"trendingScore":0,"private":false,"sha":"92904cfd4566ef75239c5d771ce0921daa4ba786","downloads":491,"tags":["region:us"],"createdAt":"2026-07-31T23:31:28.000Z","key":""},{"_id":"6a70928bf9dde61e1897c12e","id":"mkurman/synthlabs-mlabonne-open-perfectblend","author":"mkurman","disabled":false,"gated":false,"lastModified":"2026-08-03T13:07:57.000Z","likes":0,"trendingScore":0,"private":false,"sha":"d9f9c07fc175f51d49245da1d2a2967eb06c3069","description":"\n\t\n\t\t\n\t\n\t\n\t\tPerfectBlend Synth Reasoning\n\t\n\nSynthetic reasoning traces generated for mlabonne/open-perfectblend. Each record contains conversations converted from ShareGPT format (from/value) to standard message format (role/content) with synthetically generated reasoning_content attached to each assistant turn.\n\n\t\n\t\t\n\t\n\t\n\t\tDataset Summary\n\t\n\n\n27,265 records across 8 source datasets\n37,158 reasoning turns (99.7% format compliance)\nAverage 1,376 chars per reasoning trace\nReasoning generated… See the full description on the dataset page: https://huggingface.co/datasets/mkurman/synthlabs-mlabonne-open-perfectblend.","downloads":65,"tags":["task_categories:text-generation","language:en","license:apache-2.0","size_categories:10K<n<100K","format:parquet","format:optimized-parquet","modality:tabular","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us","synthetic","reasoning","sft","chain-of-thought","multi-turn"],"createdAt":"2026-08-03T13:07:23.000Z","key":""},{"_id":"6a744a61519e17b64f5c9074","id":"xxuan-speech/MLAAD_v8","author":"xxuan-speech","disabled":false,"gated":false,"lastModified":"2026-08-06T08:48:33.000Z","likes":0,"trendingScore":0,"private":false,"sha":"7aa01791eea45bd52530450b5b5c459b55686596","downloads":16,"tags":["region:us"],"createdAt":"2026-08-06T08:48:33.000Z","key":""},{"_id":"6a795570500ada5b47ee4f33","id":"thanhnx12/mla-moe-dataset-public","author":"thanhnx12","disabled":false,"gated":false,"lastModified":"2026-08-12T14:38:01.000Z","likes":0,"trendingScore":0,"private":false,"sha":"f43aa32d7547d24f5f2e0790a41225c0db37a742","description":"\n\t\n\t\t\n\t\n\t\n\t\tmla-moe public evaluation dataset\n\t\n\nPublic (participant-facing) input set for the MLA-MoE inference throughput exercise.\nTwo target models, 512 prompts each, with the model's own greedy continuation as the\ncorrectness reference:\n\n\t\n\t\t\nsubset\ntarget model\n\n\n\t\t\nglm47/\nzai-org/GLM-4.7-Flash\n\n\ndsv2lite/\ndeepseek-ai/DeepSeek-V2-Lite\n\n\n\t\n\nGrading is done on a held-out private set with the same size and length\ndistribution, generated the same way but from disjoint source text. Tune… See the full description on the dataset page: https://huggingface.co/datasets/thanhnx12/mla-moe-dataset-public.","downloads":124,"tags":["task_categories:text-generation","language:en","size_categories:1K<n<10K","format:text","modality:text","library:datasets","library:mlcroissant","region:us","inference-benchmark","throughput","evaluation"],"createdAt":"2026-08-10T04:37:04.000Z","key":""},{"_id":"6a84d13cb195b18f4365bf64","id":"agu18dec/iolens-onpolicy-pt-mlayer-pairs","author":"agu18dec","disabled":false,"gated":false,"lastModified":"2026-08-18T23:06:55.000Z","likes":0,"trendingScore":0,"private":false,"sha":"ea89611de3c36d85318f0dd02a4ebb302eb14d8d","downloads":455,"tags":["region:us"],"createdAt":"2026-08-18T21:40:12.000Z","key":""},{"_id":"6a8f10c53d53f4874878f2c2","id":"MLArtiste/Chesset","author":"MLArtiste","disabled":false,"gated":false,"lastModified":"2026-08-26T18:39:31.000Z","likes":0,"trendingScore":0,"private":false,"sha":"05c1b54c385a77e45c0dab875b8e32bce0e6ffbd","downloads":3345,"tags":["license:mit","region:us"],"createdAt":"2026-08-26T16:13:57.000Z","key":""},{"_id":"6a90b346fb0b87f119221cf7","id":"KShivendu/miriad-mlateon-colbert-smoke","author":"KShivendu","disabled":false,"gated":false,"lastModified":"2026-08-27T21:59:44.000Z","likes":0,"trendingScore":0,"private":false,"sha":"c2a1cef990821c7f1e842d5aae491eef5ef41ea1","description":"\n\t\n\t\t\n\t\n\t\n\t\tMIRIAD 200, encoded with mLateOn-medical\n\t\n\nMulti-vector (ColBERT-style) embeddings for\ntomaarsen/miriad-benchmark-200k,\nproduced with multi-vector-encoder/mLateOn-medical.\n\n\t\n\t\t\n\n\n\n\n\t\t\npassages\n200\n\n\ntoken vectors\n176,014\n\n\nmean vectors / passage\n880.07\n\n\ndim\n128\n\n\nstored dtype\nfloat16\n\n\nembeddings size\n0.05 GB\n\n\nraw text encoded\n1 MB\n\n\n\t\n\nThe embeddings are 49x larger than the text\nthey came from, which is why late-interaction retrieval needs quantization or pooling.… See the full description on the dataset page: https://huggingface.co/datasets/KShivendu/miriad-mlateon-colbert-smoke.","downloads":40,"tags":["task_categories:text-retrieval","license:apache-2.0","size_categories:n<1K","format:parquet","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us","colbert","late-interaction","multi-vector","embeddings","medical"],"createdAt":"2026-08-27T21:59:34.000Z","key":""},{"_id":"6a9b51846555e7a799f1033b","id":"araujocaio/ml-agriculture","author":"araujocaio","disabled":false,"gated":false,"lastModified":"2026-09-04T23:17:27.000Z","likes":0,"trendingScore":0,"private":false,"sha":"6cbc3ac123cf231b185204ca71ff3df58515f7de","description":"\n\t\n\t\t\n\t\n\t\n\t\tAgriculture Multimodal3 Data Notes\n\t\n\n\n\t\n\t\t\n\t\n\t\n\t\tDataset summary\n\t\n\nA documented Agriculture data-preparation workflow for Multimodal3 records. The bundled rows demonstrate the schema and validation path rather than pretending to be a full training corpus.\n\n\t\n\t\t\n\t\n\t\n\t\tIncluded material\n\t\n\n\nloader.py — loading, cleaning, and split preparation code.\ndataset_infos.json — schema and split metadata.\nmetadata_sample.jsonl — small, human-readable records for checking the schema.… See the full description on the dataset page: https://huggingface.co/datasets/araujocaio/ml-agriculture.","downloads":23,"tags":["license:apache-2.0","region:us","dataset","agriculture","multimodal3"],"createdAt":"2026-09-04T23:17:24.000Z","key":""},{"_id":"6a9c7bdc189479d6b35dde66","id":"bharatindie/ap_mla_mp_minister_2024","author":"bharatindie","disabled":false,"gated":false,"lastModified":"2026-09-05T20:31:16.000Z","likes":0,"trendingScore":0,"private":false,"sha":"e530e4b6dccf3a48bcffa8f0575fb057887a2ad2","downloads":9,"tags":["license:apache-2.0","size_categories:n<1K","format:json","modality:text","library:datasets","library:pandas","library:polars","library:mlcroissant","region:us"],"createdAt":"2026-09-05T20:30:20.000Z","key":""},{"_id":"6a9deea6eae84bed0c4d9cde","id":"MLArtiste/mate_in_one","author":"MLArtiste","disabled":false,"gated":false,"lastModified":"2026-09-07T06:15:13.000Z","likes":0,"trendingScore":0,"private":false,"sha":"7b3b8d6529e97658dcdd7ec4ee4ea7bef84e0030","downloads":0,"tags":["license:mit","region:us"],"createdAt":"2026-09-06T22:52:22.000Z","key":""}]