{"url":"/dataset/vedantany-10m","name":"VedantaNY-10M","full_name":null,"description_markdown":"VedantaNY-10M is a curated dataset of over 750 hours of transcripts from public discourses on the Indian philosophy of Advaita Vedanta. Sourced from 612 YouTube lectures by Swami Sarvapriyananda of the Vedanta Society of New York (VSNY), the dataset contains ~10 million tokens. These lectures offer a comprehensive exposition of Advaita Vedanta, making the dataset an invaluable resource for philosophy and linguistics research.","description_withheld":null,"homepage":"https://sites.google.com/view/vedantany-10m/dataset","introduced_date":"2024-08-15","introduced_date_note":null,"introduced_by":{"paper":"/paper/ancient-wisdom-modern-tools-exploring","title":"Ancient Wisdom, Modern Tools: Exploring Retrieval-Augmented LLMs for Ancient Indian Philosophy","first_author":"Priyanka Mandikal","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"Sanskrit","url":"/datasets/language/sanskrit"}],"variants":["VedantaNY-10M"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}