{"_id":"64e3488a315c08ea11d90c3e","id":"pkufool/libriheavy","author":"pkufool","sha":"df5cdf9392c1f0fb56eac1c38fc1256426b24afe","lastModified":"2023-09-19T11:35:45.000Z","private":false,"gated":false,"disabled":false,"tags":["license:apache-2.0","size_categories:10K<n<100K","format:json","modality:tabular","modality:text","library:datasets","library:dask","library:mlcroissant","library:polars","arxiv:2309.08105","region:us"],"description":"\n\t\n\t\t\n\t\n\t\n\t\tLibriheavy: a 50,000 hours ASR corpus with punctuation casing and context\n\t\n\nLibriheavy is a labeled version of Librilight, read our paper for more details.\nSee https://github.com/k2-fsa/libriheavy for more details.\n\n\t\n\t\t\n\t\n\t\n\t\tCitation\n\t\n\n@misc{kang2023libriheavy,\n      title={Libriheavy: a 50,000 hours ASR corpus with punctuation casing and context}, \n      author={Wei Kang and Xiaoyu Yang and Zengwei Yao and Fangjun Kuang and Yifan Yang and Liyong Guo and Long Lin and Daniel… See the full description on the dataset page: https://huggingface.co/datasets/pkufool/libriheavy.","downloads":509,"likes":20,"cardData":{"license":"apache-2.0"},"siblings":[{"rfilename":".gitattributes"},{"rfilename":"README.md"},{"rfilename":"libriheavy_cuts_dev.jsonl.gz"},{"rfilename":"libriheavy_cuts_large.jsonl.gz"},{"rfilename":"libriheavy_cuts_medium.jsonl.gz"},{"rfilename":"libriheavy_cuts_small.jsonl.gz"},{"rfilename":"libriheavy_cuts_test_clean.jsonl.gz"},{"rfilename":"libriheavy_cuts_test_clean_large.jsonl.gz"},{"rfilename":"libriheavy_cuts_test_other.jsonl.gz"},{"rfilename":"libriheavy_cuts_test_other_large.jsonl.gz"},{"rfilename":"raw/libriheavy_cuts_large.jsonl.gz"},{"rfilename":"raw/libriheavy_cuts_medium.jsonl.gz"},{"rfilename":"raw/libriheavy_cuts_small.jsonl.gz"}],"createdAt":"2023-08-21T11:20:42.000Z","usedStorage":48961107309}