{"url":"/dataset/nepali-text-corpus","name":"Nepali Text Corpus","full_name":null,"description_markdown":"Overview\r\nNepali-Text-Corpus is a comprehensive collection of approximately 6.4 million articles in the Nepali language. This dataset is the largest text dataset on Nepali Language. It encompasses a diverse range of text types, including news articles, blogs, and more, making it an invaluable resource for researchers, developers, and enthusiasts in the fields of Natural Language Processing (NLP) and computational linguistics.\r\n\r\nDataset Details\r\nTotal Articles: ~6.4 million\r\nLanguage: Nepali\r\nSize: 27.5 GB (in csv)\r\nSource: Collected from various Nepali news websites, blogs, and other online platforms.","description_withheld":null,"homepage":"https://huggingface.co/datasets/IRIISNEPAL/Nepali-Text-Corpus","introduced_date":"2024-09-14","introduced_date_note":null,"introduced_by":{"paper":"/paper/development-of-pre-trained-transformer-based","title":"Development of Pre-Trained Transformer-based Models for the Nepali Language","first_author":"Prajwal Thapa","url":null},"license":null,"modalities":[],"tasks":[],"languages":[{"name":"Nepali (macrolanguage)","url":"/datasets/language/nepali-macrolanguage"}],"variants":["Nepali Text Corpus"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}