{"url":"/dataset/vaksancayah","name":"Vāksañcayaḥ","full_name":"Sanskrit Speech Corpus by IIT Bombay","description_markdown":"This Sanskrit speech corpus has more than 78 hours of audio data and contains recordings of 45,953 sentences with a sampling rate of 22KHz. The content is mainly readings of texts spanning over various Śāstras of Saṃskṛtam literature and also includes contemporary stories, radio program, extempore discourse, etc.","description_withheld":null,"homepage":"https://www.cse.iitb.ac.in/~asr/","introduced_date":"2021-06-02","introduced_date_note":null,"introduced_by":{"paper":"/paper/automatic-speech-recognition-in-sanskrit-a","title":"Automatic Speech Recognition in Sanskrit: A New Speech Corpus and Modelling Insights","first_author":"Devaraja Adiga","url":null},"license":{"name":"Creative Commons Attribution-NonCommercial 4.0 International License","url":"https://creativecommons.org/licenses/by-nc/4.0/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"}],"languages":[{"name":"Sanskrit","url":"/datasets/language/sanskrit"}],"variants":["Vāksañcayaḥ"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}