{"url":"/dataset/medmcqa","name":"MedMCQA","full_name":null,"description_markdown":"**MedMCQA** is a large-scale, Multiple-Choice Question Answering (MCQA) dataset designed to address real-world medical entrance exam questions.\r\n\r\nMedMCQA has more than 194k high-quality AIIMS & NEET PG entrance exam MCQs covering 2.4k healthcare topics and 21 medical subjects are collected with an average token length of 12.77 and high topical diversity.","description_withheld":null,"homepage":"https://medmcqa.github.io/","introduced_date":"2022-03-27","introduced_date_note":null,"introduced_by":{"paper":"/paper/medmcqa-a-large-scale-multi-subject-multi","title":"MedMCQA : A Large-scale Multi-Subject Multi-Choice Dataset for Medical domain Question Answering","first_author":"Ankit Pal","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Multiple Choice Question Answering (MCQA)","url":"/task/multiple-choice-qa","datasets_with_task":"/datasets/task/multiple-choice-qa"}],"languages":[],"variants":["MedMCQA","MedMCQA (with Context)"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/qompass/qmedmcqa","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/openlifescienceai/medmcqa","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/medmcqa","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/MedMCQA/MedMCQA","url":"https://github.com/MedMCQA/MedMCQA","frameworks":[]}],"num_papers_in_archive":144,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-21","task":"Multiple Choice Question Answering (MCQA)","dataset_variant":"MedMCQA","rows":22,"metrics":["Test Set (Acc-%)","Dev Set (Acc-%)"],"first_row_in_archive_order":{"model":"Med-PaLM 2 (ER)","paper":"/paper/towards-expert-level-medical-question","metrics":{"Test Set (Acc-%)":"0.723"},"code_links":[{"title":"m42-health/med42","url":"https://github.com/m42-health/med42"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/meditron-70b-scaling-medical-pretraining-for","title":"MEDITRON-70B: Scaling Medical Pretraining for Large Language Models","date":"2023-11-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":9,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/biomedgpt-open-multimodal-generative-pre","title":"BioMedGPT: Open Multimodal Generative Pre-trained Transformer for BioMedicine","date":"2023-08-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/towards-expert-level-medical-question","title":"Towards Expert-Level Medical Question Answering with Large Language Models","date":"2023-05-16","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/large-language-models-encode-clinical","title":"Large Language Models Encode Clinical Knowledge","date":"2022-12-26","rows_on_this_dataset":8,"code_links":1,"syntology":null},{"paper":"/paper/galactica-a-large-language-model-for-science-1","title":"Galactica: A Large Language Model for Science","date":"2022-11-16","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/variational-open-domain-question-answering","title":"Variational Open-Domain Question Answering","date":"2022-09-23","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/can-large-language-models-reason-about","title":"Can large language models reason about medical questions?","date":"2022-07-17","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":0,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/medmcqa-a-large-scale-multi-subject-multi","title":"MedMCQA : A Large-scale Multi-Subject Multi-Choice Dataset for Medical domain Question Answering","date":"2022-03-27","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":28,"samples_ran":9,"samples_unverified":19,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":4,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}