{"url":"/dataset/l3cube-mahacorpus","name":"L3Cube-MahaCorpus","full_name":null,"description_markdown":"**L3Cube-MahaCorpus** is a Marathi monolingual data set scraped from different internet sources. We expand the existing Marathi monolingual corpus with 24.8M sentences and 289M tokens. We also present, MahaBERT, MahaAlBERT, and MahaRoBerta all BERT-based masked language models, and MahaFT, the fast text word embeddings both trained on full Marathi corpus with 752M tokens.","description_withheld":null,"homepage":"https://github.com/l3cube-pune/MarathiNLP","introduced_date":"2023-06-24","introduced_date_note":null,"introduced_by":{"paper":"/paper/my-boli-code-mixed-marathi-english-corpora","title":"My Boli: Code-mixed Marathi-English Corpora, Pretrained Language Models and Evaluation Benchmarks","first_author":"Tanmay Chavan","url":null},"license":{"name":"Creative Commons Attribution-NonCommercial-ShareAlike 4.0 International License","url":"http://creativecommons.org/licenses/by-nc-sa/4.0/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Sentiment Analysis","url":"/task/sentiment-analysis","datasets_with_task":"/datasets/task/sentiment-analysis"},{"name":"Hate Speech Detection","url":"/task/hate-speech-detection","datasets_with_task":"/datasets/task/hate-speech-detection"},{"name":"Language Identification","url":"/task/language-identification","datasets_with_task":"/datasets/task/language-identification"}],"languages":[{"name":"Marathi","url":"/datasets/language/marathi"}],"variants":["L3Cube-MahaCorpus"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}