{"url":"/dataset/medconceptsqa","name":"MedConceptsQA","full_name":null,"description_markdown":"MedConceptsQA - Open Source Medical Concepts QA Benchmark\r\n\r\nThe benchmark can be found here: \r\nhttps://huggingface.co/datasets/ofir408/MedConceptsQA","description_withheld":null,"homepage":"https://huggingface.co/datasets/ofir408/MedConceptsQA","introduced_date":"2024-05-12","introduced_date_note":null,"introduced_by":{"paper":"/paper/medconceptsqa-open-source-medical-concepts-qa","title":"MedConceptsQA: Open Source Medical Concepts QA Benchmark","first_author":"Ofir Ben Shoham","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Zero-Shot Learning","url":"/task/zero-shot-learning","datasets_with_task":"/datasets/task/zero-shot-learning"},{"name":"Few-Shot Learning","url":"/task/few-shot-learning","datasets_with_task":"/datasets/task/few-shot-learning"},{"name":"Large Language Model","url":"/task/large-language-model","datasets_with_task":"/datasets/task/large-language-model"},{"name":"Medical Diagnosis","url":"/task/medical-diagnosis","datasets_with_task":"/datasets/task/medical-diagnosis"},{"name":"Medical Procedure","url":"/task/medical-procedure","datasets_with_task":"/datasets/task/medical-procedure"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["MedConceptsQA"],"data_loaders":[],"num_papers_in_archive":13,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/zero-shot-learning-on-medconceptsqa","task":"Zero-Shot Learning","dataset_variant":"MedConceptsQA","rows":13,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"gpt-4-0125-preview","paper":"/paper/gpt-4-technical-report-1","metrics":{"Accuracy":"52.489"},"code_links":[{"title":"openai/evals","url":"https://github.com/openai/evals"},{"title":"shmsw25/factscore","url":"https://github.com/shmsw25/factscore"},{"title":"unispac/visual-adversarial-examples-jailbreak-large-language-models","url":"https://github.com/unispac/visual-adversarial-examples-jailbreak-large-language-models"},{"title":"gpt4life/alpagasus","url":"https://github.com/gpt4life/alpagasus"},{"title":"emrgnt-cmplxty/zero-shot-replication","url":"https://github.com/emrgnt-cmplxty/zero-shot-replication"},{"title":"ethz-privsec/superhuman-ai-consistency","url":"https://github.com/ethz-privsec/superhuman-ai-consistency"},{"title":"ethz-spylab/superhuman-ai-consistency","url":"https://github.com/ethz-spylab/superhuman-ai-consistency"},{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"},{"title":"AUCOHL/RTL-Repo","url":"https://github.com/AUCOHL/RTL-Repo"},{"title":"zach-zhiling-zheng/reticular_chemist","url":"https://github.com/zach-zhiling-zheng/reticular_chemist"},{"title":"lflage/openfactscore","url":"https://github.com/lflage/openfactscore"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/few-shot-learning-on-medconceptsqa","task":"Few-Shot Learning","dataset_variant":"MedConceptsQA","rows":12,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"gpt-4-0125-preview","paper":"/paper/gpt-4-technical-report-1","metrics":{"Accuracy":"61.911"},"code_links":[{"title":"openai/evals","url":"https://github.com/openai/evals"},{"title":"shmsw25/factscore","url":"https://github.com/shmsw25/factscore"},{"title":"unispac/visual-adversarial-examples-jailbreak-large-language-models","url":"https://github.com/unispac/visual-adversarial-examples-jailbreak-large-language-models"},{"title":"gpt4life/alpagasus","url":"https://github.com/gpt4life/alpagasus"},{"title":"emrgnt-cmplxty/zero-shot-replication","url":"https://github.com/emrgnt-cmplxty/zero-shot-replication"},{"title":"ethz-privsec/superhuman-ai-consistency","url":"https://github.com/ethz-privsec/superhuman-ai-consistency"},{"title":"ethz-spylab/superhuman-ai-consistency","url":"https://github.com/ethz-spylab/superhuman-ai-consistency"},{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"},{"title":"AUCOHL/RTL-Repo","url":"https://github.com/AUCOHL/RTL-Repo"},{"title":"zach-zhiling-zheng/reticular_chemist","url":"https://github.com/zach-zhiling-zheng/reticular_chemist"},{"title":"lflage/openfactscore","url":"https://github.com/lflage/openfactscore"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/medconceptsqa-open-source-medical-concepts-qa","title":"MedConceptsQA: Open Source Medical Concepts QA Benchmark","date":"2024-05-12","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/small-language-models-learn-enhanced","title":"Small Language Models Learn Enhanced Reasoning Skills from Medical Textbooks","date":"2024-03-30","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/biomistral-a-collection-of-open-source","title":"BioMistral: A Collection of Open-Source Pretrained Large Language Models for Medical Domains","date":"2024-02-15","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/meditron-70b-scaling-medical-pretraining-for","title":"MEDITRON-70B: Scaling Medical Pretraining for Large Language Models","date":"2023-11-27","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":9,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zephyr-direct-distillation-of-lm-alignment","title":"Zephyr: Direct Distillation of LM Alignment","date":"2023-10-25","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/biomedgpt-open-multimodal-generative-pre","title":"BioMedGPT: Open Multimodal Generative Pre-trained Transformer for BioMedicine","date":"2023-08-18","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","rows_on_this_dataset":2,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llama-open-and-efficient-foundation-language-1","title":"LLaMA: Open and Efficient Foundation Language Models","date":"2023-02-27","rows_on_this_dataset":2,"code_links":57,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":58,"samples_ran":26,"samples_unverified":32,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gatortron-a-large-clinical-language-model-to","title":"GatorTron: A Large Clinical Language Model to Unlock Patient Information from Unstructured Electronic Health Records","date":"2022-02-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/clinical-longformer-and-clinical-bigbird","title":"Clinical-Longformer and Clinical-BigBird: Transformers for long clinical sequences","date":"2022-01-27","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/language-models-are-few-shot-learners","title":"Language Models are Few-Shot Learners","date":"2020-05-28","rows_on_this_dataset":2,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":65,"samples_ran":15,"samples_unverified":50,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/biobert-a-pre-trained-biomedical-language","title":"BioBERT: a pre-trained biomedical language representation model for biomedical text mining","date":"2019-01-25","rows_on_this_dataset":2,"code_links":19,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":4,"samples_unverified":21,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":8,"samples_harvested":171,"samples_ran":60,"samples_unverified":111,"pointer_only_for_licence":14,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}