{"url":"/dataset/medqa-usmle","name":"MedQA","full_name":null,"description_markdown":"Multiple choice question answering based on the United States Medical License Exams (USMLE). The dataset is collected from the professional medical board exams. It covers three languages: English, simplified Chinese, and traditional Chinese, and contains 12,723, 34,251, and 14,123 questions for the three languages, respectively.","description_withheld":null,"homepage":"https://drive.google.com/file/d/1ImYUSLk9JbgHXOemfvyiDiirluZHPeQw/view","introduced_date":"2020-09-28","introduced_date_note":null,"introduced_by":{"paper":"/paper/what-disease-does-this-patient-have-a-large","title":"What Disease does this Patient Have? A Large-scale Open Domain Question Answering Dataset from Medical Exams","first_author":"Di Jin","url":null},"license":null,"modalities":[],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"}],"languages":[],"variants":["MedQA"],"data_loaders":[{"repo":"https://github.com/Goodnight77/datasets1","url":"https://github.com/Goodnight77/datasets1","frameworks":["tf","pytorch"]}],"num_papers_in_archive":328,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-answering-on-medqa-usmle","task":"Question Answering","dataset_variant":"MedQA","rows":27,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Med-Gemini","paper":"/paper/capabilities-of-gemini-models-in-medicine","metrics":{"Accuracy":"91.1"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/shakti-a-2-5-billion-parameter-small-language","title":"SHAKTI: A 2.5 Billion Parameter Small Language Model Optimized for Edge AI and Low-Resource Environments","date":"2024-10-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/medmobile-a-mobile-sized-language-model-with","title":"MedMobile: A mobile-sized language model with expert-level clinical capabilities","date":"2024-10-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/capabilities-of-gemini-models-in-medicine","title":"Capabilities of Gemini Models in Medicine","date":"2024-04-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/small-language-models-learn-enhanced","title":"Small Language Models Learn Enhanced Reasoning Skills from Medical Textbooks","date":"2024-03-30","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/can-generalist-foundation-models-outcompete","title":"Can Generalist Foundation Models Outcompete Special-Purpose Tuning? Case Study in Medicine","date":"2023-11-28","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":0,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/meditron-70b-scaling-medical-pretraining-for","title":"MEDITRON-70B: Scaling Medical Pretraining for Large Language Models","date":"2023-11-27","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":9,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/biomedgpt-open-multimodal-generative-pre","title":"BioMedGPT: Open Multimodal Generative Pre-trained Transformer for BioMedicine","date":"2023-08-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/towards-expert-level-medical-question","title":"Towards Expert-Level Medical Question Answering with Large Language Models","date":"2023-05-16","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/grapeqa-graph-augmentation-and-pruning-to","title":"GrapeQA: GRaph Augmentation and Pruning to Enhance Question-Answering","date":"2023-03-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/large-language-models-encode-clinical","title":"Large Language Models Encode Clinical Knowledge","date":"2022-12-26","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/galactica-a-large-language-model-for-science-1","title":"Galactica: A Large Language Model for Science","date":"2022-11-16","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deep-bidirectional-language-knowledge-graph","title":"Deep Bidirectional Language-Knowledge Graph Pretraining","date":"2022-10-17","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":2,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/variational-open-domain-question-answering","title":"Variational Open-Domain Question Answering","date":"2022-09-23","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/can-large-language-models-reason-about","title":"Can large language models reason about medical questions?","date":"2022-07-17","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":0,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/linkbert-pretraining-language-models-with","title":"LinkBERT: Pretraining Language Models with Document Links","date":"2022-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":0,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/biobert-a-pre-trained-biomedical-language","title":"BioBERT: a pre-trained biomedical language representation model for biomedical text mining","date":"2019-01-25","rows_on_this_dataset":2,"code_links":19,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":4,"samples_unverified":21,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":8,"samples_harvested":94,"samples_ran":15,"samples_unverified":79,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":5,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}