{"url":"/task/multiple-choice-qa","name":"Multiple Choice Question Answering (MCQA)","slug":"multiple-choice-qa","description_markdown":"A multiple-choice question (MCQ) is composed of two parts: a stem that identifies the question or problem, and a set of alternatives or possible answers that contain a key that is the best answer to the question, and a number of distractors that are plausible but incorrect answers to the question.\r\n\r\nIn a k-way MCQA task, a model is provided with a question q, a set of candidate options O = {O1, . . . , Ok}, and a supporting context for each option C = {C1, . . . , Ck}. The model needs to predict the correct answer option that is best supported by the given contexts.","categories":[{"name":"Miscellaneous","url":"/area/miscellaneous"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"},{"name":"Reasoning","url":"/area/reasoning"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":65,"papers_with_code":37,"benchmarks":31,"benchmark_tables_in_archive":31,"benchmark_tables_shown":31,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":9,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-21","slug":"multiple-choice-question-answering-mcqa-on-21","dataset":"MedMCQA","dataset_url":"/dataset/medmcqa","rows_in_archive":22,"metrics":["Test Set (Acc-%)","Dev Set (Acc-%)"],"first_row_in_archive_order":{"model":"Med-PaLM 2 (ER)","paper_title":"Towards Expert-Level Medical Question Answering with Large Language Models","paper_url":"/paper/towards-expert-level-medical-question","paper_date":"2023-05-16","arxiv_id":"2305.09617","code_links":[{"title":"m42-health/med42","url":"https://github.com/m42-health/med42"}],"syntology":null}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-27","slug":"multiple-choice-question-answering-mcqa-on-27","dataset":"BIG-bench (Hyperbaton)","dataset_url":"/dataset/big-bench","rows_in_archive":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Bloomberg GPT (few-shot, k=3)","paper_title":"BloombergGPT: A Large Language Model for Finance","paper_url":"/paper/bloomberggpt-a-large-language-model-for","paper_date":"2023-03-30","arxiv_id":"2303.17564","code_links":[{"title":"yangletliu/finlora","url":"https://github.com/yangletliu/finlora"},{"title":"open-finance-lab/finlora","url":"https://github.com/open-finance-lab/finlora"}],"syntology":null}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-28","slug":"multiple-choice-question-answering-mcqa-on-28","dataset":"BIG-bench (Movie Recommendation)","dataset_url":"/dataset/big-bench","rows_in_archive":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM 2 (few-shot, k=3, CoT)","paper_title":"PaLM 2 Technical Report","paper_url":"/paper/palm-2-technical-report-1","paper_date":"2023-05-17","arxiv_id":"2305.10403","code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}],"syntology":null}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-29","slug":"multiple-choice-question-answering-mcqa-on-29","dataset":"BIG-bench (Navigate)","dataset_url":"/dataset/big-bench","rows_in_archive":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM 2 (few-shot, k=3, CoT)","paper_title":"PaLM 2 Technical Report","paper_url":"/paper/palm-2-technical-report-1","paper_date":"2023-05-17","arxiv_id":"2305.10403","code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}],"syntology":null}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-30","slug":"multiple-choice-question-answering-mcqa-on-30","dataset":"BIG-bench (Ruin Names)","dataset_url":"/dataset/big-bench","rows_in_archive":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM 2 (few-shot, k=3, Direct)","paper_title":"PaLM 2 Technical Report","paper_url":"/paper/palm-2-technical-report-1","paper_date":"2023-05-17","arxiv_id":"2305.10403","code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}],"syntology":null}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-11","slug":"multiple-choice-question-answering-mcqa-on-11","dataset":"MMLU (College Biology)","dataset_url":"/dataset/mmlu","rows_in_archive":8,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Med-PaLM 2 (ER)","paper_title":"Towards Expert-Level Medical Question Answering with Large Language Models","paper_url":"/paper/towards-expert-level-medical-question","paper_date":"2023-05-16","arxiv_id":"2305.09617","code_links":[{"title":"m42-health/med42","url":"https://github.com/m42-health/med42"}],"syntology":null}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-8","slug":"multiple-choice-question-answering-mcqa-on-8","dataset":"MMLU (Medical Genetics)","dataset_url":"/dataset/mmlu","rows_in_archive":8,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Med-PaLM 2 (ER)","paper_title":"Towards Expert-Level Medical Question Answering with Large Language Models","paper_url":"/paper/towards-expert-level-medical-question","paper_date":"2023-05-16","arxiv_id":"2305.09617","code_links":[{"title":"m42-health/med42","url":"https://github.com/m42-health/med42"}],"syntology":null}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-25","slug":"multiple-choice-question-answering-mcqa-on-25","dataset":"MMLU (Professional medicine)","dataset_url":"/dataset/mmlu","rows_in_archive":6,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Med-PaLM 2 (5-shot)","paper_title":"Towards Expert-Level Medical Question Answering with Large Language Models","paper_url":"/paper/towards-expert-level-medical-question","paper_date":"2023-05-16","arxiv_id":"2305.09617","code_links":[{"title":"m42-health/med42","url":"https://github.com/m42-health/med42"}],"syntology":null}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-10","slug":"multiple-choice-question-answering-mcqa-on-10","dataset":"MMLU (Elementary Mathematics)","dataset_url":"/dataset/mmlu","rows_in_archive":5,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla (few-shot, k=5)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-12","slug":"multiple-choice-question-answering-mcqa-on-12","dataset":"MMLU (High School Biology)","dataset_url":"/dataset/mmlu","rows_in_archive":5,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla (few-shot, k=5)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-13","slug":"multiple-choice-question-answering-mcqa-on-13","dataset":"MMLU (College Chemistry)","dataset_url":"/dataset/mmlu","rows_in_archive":5,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla (few-shot, k=5)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-16","slug":"multiple-choice-question-answering-mcqa-on-16","dataset":"MMLU (High School Mathematics)","dataset_url":"/dataset/mmlu","rows_in_archive":5,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GAL 120B (zero-shot)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-17","slug":"multiple-choice-question-answering-mcqa-on-17","dataset":"MMLU (Electrical Engineer)","dataset_url":"/dataset/mmlu","rows_in_archive":5,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GAL 120B (zero-shot)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-18","slug":"multiple-choice-question-answering-mcqa-on-18","dataset":"MMLU (College Physics)","dataset_url":"/dataset/mmlu","rows_in_archive":5,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla (few-shot, k=5)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-2","slug":"multiple-choice-question-answering-mcqa-on-2","dataset":"MMLU (Formal Logic)","dataset_url":"/dataset/mmlu","rows_in_archive":5,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher (few-shot, k=5)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-20","slug":"multiple-choice-question-answering-mcqa-on-20","dataset":"MMLU (High School Statistics)","dataset_url":"/dataset/mmlu","rows_in_archive":5,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla (few-shot, k=5)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-3","slug":"multiple-choice-question-answering-mcqa-on-3","dataset":"MMLU (Abstract Algebra)","dataset_url":"/dataset/mmlu","rows_in_archive":5,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GAL 30B (zero-shot)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-4","slug":"multiple-choice-question-answering-mcqa-on-4","dataset":"MMLU (Econometrics)","dataset_url":"/dataset/mmlu","rows_in_archive":5,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher (few-shot, k=5)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-5","slug":"multiple-choice-question-answering-mcqa-on-5","dataset":"MMLU (High School Computer Science)","dataset_url":"/dataset/mmlu","rows_in_archive":5,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GAL 120B (zero-shot)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-7","slug":"multiple-choice-question-answering-mcqa-on-7","dataset":"MMLU (College Mathematics)","dataset_url":"/dataset/mmlu","rows_in_archive":5,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GAL 120B (zero-shot)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-9","slug":"multiple-choice-question-answering-mcqa-on-9","dataset":"MMLU (Astronomy)","dataset_url":"/dataset/mmlu","rows_in_archive":5,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla (few-shot, k=5)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-14","slug":"multiple-choice-question-answering-mcqa-on-14","dataset":"MMLU (High School Chemistry)","dataset_url":"/dataset/mmlu","rows_in_archive":4,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla (few-shot, k=5)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-15","slug":"multiple-choice-question-answering-mcqa-on-15","dataset":"MMLU (College Computer Science)","dataset_url":"/dataset/mmlu","rows_in_archive":4,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla (few-shot, k=5)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-19","slug":"multiple-choice-question-answering-mcqa-on-19","dataset":"MMLU (High School Physics)","dataset_url":"/dataset/mmlu","rows_in_archive":4,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla (few-shot, k=5)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-31","slug":"multiple-choice-question-answering-mcqa-on-31","dataset":"BIG-bench (Novel Concepts)","dataset_url":"/dataset/big-bench","rows_in_archive":4,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM-540B (few-shot, k=5)","paper_title":"PaLM: Scaling Language Modeling with Pathways","paper_url":"/paper/palm-scaling-language-modeling-with-pathways-1","paper_date":"2022-04-05","arxiv_id":"2204.02311","code_links":[{"title":"lucidrains/CoCa-pytorch","url":"https://github.com/lucidrains/CoCa-pytorch"},{"title":"lucidrains/PaLM-pytorch","url":"https://github.com/lucidrains/PaLM-pytorch"},{"title":"google/paxml","url":"https://github.com/google/paxml"},{"title":"foundation-model-stack/fms-fsdp","url":"https://github.com/foundation-model-stack/fms-fsdp"},{"title":"lucidrains/PaLM-jax","url":"https://github.com/lucidrains/PaLM-jax"},{"title":"chrisociepa/allamo","url":"https://github.com/chrisociepa/allamo"},{"title":"conceptofmind/PaLM-flax","url":"https://github.com/conceptofmind/PaLM-flax"}],"syntology":{"n":37,"n_ran":30,"n_unverified":7,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-6","slug":"multiple-choice-question-answering-mcqa-on-6","dataset":"MMLU (Machine Learning)","dataset_url":"/dataset/mmlu","rows_in_archive":4,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla (few-shot, k=5)","paper_title":"Galactica: A Large Language Model for Science","paper_url":"/paper/galactica-a-large-language-model-for-science-1","paper_date":"2022-11-16","arxiv_id":"2211.09085","code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}],"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-qa-on-indicglue-wstp-pa","slug":"multiple-choice-qa-on-indicglue-wstp-pa","dataset":"IndicGLUE WSTP Pa","dataset_url":"/dataset/indicglue","rows_in_archive":3,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"xlmindic-base-uniscript","paper_title":"Does Transliteration Help Multilingual Language Modeling?","paper_url":"/paper/does-transliteration-help-multilingual","paper_date":"2022-01-29","arxiv_id":"2201.12501","code_links":[{"title":"ibraheem-moosa/xlm-indic","url":"https://github.com/ibraheem-moosa/xlm-indic"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-23","slug":"multiple-choice-question-answering-mcqa-on-23","dataset":"MMLU (Clinical Knowledge)","dataset_url":"/dataset/mmlu","rows_in_archive":3,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Med-PaLM 2 (ER)","paper_title":"Towards Expert-Level Medical Question Answering with Large Language Models","paper_url":"/paper/towards-expert-level-medical-question","paper_date":"2023-05-16","arxiv_id":"2305.09617","code_links":[{"title":"m42-health/med42","url":"https://github.com/m42-health/med42"}],"syntology":null}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-24","slug":"multiple-choice-question-answering-mcqa-on-24","dataset":"MMLU (Anatomy)","dataset_url":"/dataset/mmlu","rows_in_archive":3,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Med-PaLM 2 (ER)","paper_title":"Towards Expert-Level Medical Question Answering with Large Language Models","paper_url":"/paper/towards-expert-level-medical-question","paper_date":"2023-05-16","arxiv_id":"2305.09617","code_links":[{"title":"m42-health/med42","url":"https://github.com/m42-health/med42"}],"syntology":null}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-26","slug":"multiple-choice-question-answering-mcqa-on-26","dataset":"MMLU (College Medicine)","dataset_url":"/dataset/mmlu","rows_in_archive":3,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Med-PaLM (ER)","paper_title":"Towards Expert-Level Medical Question Answering with Large Language Models","paper_url":"/paper/towards-expert-level-medical-question","paper_date":"2023-05-16","arxiv_id":"2305.09617","code_links":[{"title":"m42-health/med42","url":"https://github.com/m42-health/med42"}],"syntology":null}},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-22","slug":"multiple-choice-question-answering-mcqa-on-22","dataset":"FrenchMedMCQA","dataset_url":"/dataset/frenchmedmcqa","rows_in_archive":2,"metrics":["Exact Match Accuracy","Hamming Score"],"first_row_in_archive_order":{"model":"CamemBERT","paper_title":"FrenchMedMCQA: A French Multiple-Choice Question Answering Dataset for Medical domain","paper_url":"/paper/frenchmedmcqa-a-french-multiple-choice-1","paper_date":"2023-04-09","arxiv_id":"2304.04280","code_links":[{"title":"qanastek/FrenchMedMCQA","url":"https://github.com/qanastek/FrenchMedMCQA"}],"syntology":null}}],"datasets":[{"url":"/dataset/mmlu","name":"MML","full_name":"Massive Multitask Language Understanding","num_papers_in_archive":1922},{"url":"/dataset/big-bench","name":"BIG-bench","full_name":"Beyond the Imitation Game Benchmark","num_papers_in_archive":349},{"url":"/dataset/medmcqa","name":"MedMCQA","full_name":"","num_papers_in_archive":144},{"url":"/dataset/indicglue","name":"IndicGLUE","full_name":"Indic General Language Understanding Evaluation Benchmark","num_papers_in_archive":16},{"url":"/dataset/m3ke","name":"M3KE","full_name":"Massive Multi-Level Multi-Subject Knowledge Evaluation Benchmark","num_papers_in_archive":14},{"url":"/dataset/frenchmedmcqa","name":"FrenchMedMCQA","full_name":"FrenchMedMCQA: A French Multiple-Choice Question Answering Dataset for Medical domain","num_papers_in_archive":6},{"url":"/dataset/secqa","name":"SecQA","full_name":"","num_papers_in_archive":4},{"url":"/dataset/sfd","name":"SF20K","full_name":"Short-Films 20K","num_papers_in_archive":1},{"url":"/dataset/lmcqa","name":"LMCQA","full_name":"Legal Multiple Choice Question Answering","num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[{"url":"/task/question-answering","name":"Question Answering"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":37,"tagged_in_all":65,"items":[{"url":"/paper/llama-2-open-foundation-and-fine-tuned-chat","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","date":"2023-07-18","arxiv_id":"2307.09288","repositories_listed":19,"syntology":{"n":52,"n_ran":31,"n_unverified":21,"n_pointer_only":16}},{"url":"/paper/palm-scaling-language-modeling-with-pathways-1","title":"PaLM: Scaling Language Modeling with Pathways","date":"2022-04-05","arxiv_id":"2204.02311","repositories_listed":7,"syntology":{"n":37,"n_ran":30,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/from-recognition-to-cognition-visual","title":"From Recognition to Cognition: Visual Commonsense Reasoning","date":"2018-11-27","arxiv_id":"1811.10830","repositories_listed":4,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/quality-question-answering-with-long-input","title":"QuALITY: Question Answering with Long Input Texts, Yes!","date":"2021-12-16","arxiv_id":"2112.08608","repositories_listed":3,"syntology":null},{"url":"/paper/scaling-language-models-methods-analysis-1","title":"Scaling Language Models: Methods, Analysis & Insights from Training Gopher","date":"2021-12-08","arxiv_id":"2112.11446","repositories_listed":3,"syntology":null},{"url":"/paper/learning-to-attend-on-essential-terms-an","title":"Learning to Attend On Essential Terms: An Enhanced Retriever-Reader Model for Open-domain Question Answering","date":"2018-08-28","arxiv_id":"1808.09492","repositories_listed":3,"syntology":null},{"url":"/paper/bloomberggpt-a-large-language-model-for","title":"BloombergGPT: A Large Language Model for Finance","date":"2023-03-30","arxiv_id":"2303.17564","repositories_listed":2,"syntology":null},{"url":"/paper/variational-open-domain-question-answering","title":"Variational Open-Domain Question Answering","date":"2022-09-23","arxiv_id":"2210.06345","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/training-compute-optimal-large-language","title":"Training Compute-Optimal Large Language Models","date":"2022-03-29","arxiv_id":"2203.15556","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_unverified":3,"n_pointer_only":4}},{"url":"/paper/mmm-multi-stage-multi-task-learning-for-multi","title":"MMM: Multi-stage Multi-task Learning for Multi-choice Reading Comprehension","date":"2019-10-01","arxiv_id":"1910.00458","repositories_listed":2,"syntology":null},{"url":"/paper/question-aware-knowledge-graph-prompting-for","title":"Question-Aware Knowledge Graph Prompting for Enhancing Large Language Models","date":"2025-03-30","arxiv_id":"2503.23523","repositories_listed":1,"syntology":null},{"url":"/paper/wrong-answers-can-also-be-useful-plausibleqa","title":"Wrong Answers Can Also Be Useful: PlausibleQA -- A Large-Scale QA Dataset with Answer Plausibility Scores","date":"2025-02-22","arxiv_id":"2502.16358","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-shortcomings-of-llms-in","title":"Investigating the Shortcomings of LLMs in Step-by-Step Legal Reasoning","date":"2025-02-08","arxiv_id":"2502.05675","repositories_listed":1,"syntology":null},{"url":"/paper/medg-krp-medical-graph-knowledge","title":"MedG-KRP: Medical Graph Knowledge Representation Probing","date":"2024-12-14","arxiv_id":"2412.10982","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-abilities-of-large-language","title":"KnowledgePrompts: Exploring the Abilities of Large Language Models to Solve Proportional Analogies via Knowledge-Enhanced Prompting","date":"2024-12-01","arxiv_id":"2412.00869","repositories_listed":1,"syntology":null},{"url":"/paper/differentiating-choices-via-commonality-for","title":"Differentiating Choices via Commonality for Multiple-Choice Question Answering","date":"2024-08-21","arxiv_id":"2408.11554","repositories_listed":1,"syntology":null},{"url":"/paper/econlogicqa-a-question-answering-benchmark","title":"EconLogicQA: A Question-Answering Benchmark for Evaluating Large Language Models in Economic Sequential Reasoning","date":"2024-05-13","arxiv_id":"2405.07938","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/adamole-fine-tuning-large-language-models","title":"AdaMoLE: Fine-Tuning Large Language Models with Adaptive Mixture of Low-Rank Adaptation Experts","date":"2024-05-01","arxiv_id":"2405.00361","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/can-a-multichoice-dataset-be-repurposed-for","title":"From Multiple-Choice to Extractive QA: A Case Study for English and Arabic","date":"2024-04-26","arxiv_id":"2404.17342","repositories_listed":1,"syntology":null},{"url":"/paper/artifacts-or-abduction-how-do-llms-answer","title":"Artifacts or Abduction: How Do LLMs Answer Multiple-Choice Questions Without the Question?","date":"2024-02-19","arxiv_id":"2402.12483","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_unverified":6,"n_pointer_only":10}},{"url":"/paper/meditron-70b-scaling-medical-pretraining-for","title":"MEDITRON-70B: Scaling Medical Pretraining for Large Language Models","date":"2023-11-27","arxiv_id":"2311.16079","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/fool-your-vision-and-language-model-with","title":"Fool Your (Vision and) Language Model With Embarrassingly Simple Permutations","date":"2023-10-02","arxiv_id":"2310.01651","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":5}},{"url":"/paper/biomedgpt-open-multimodal-generative-pre","title":"BioMedGPT: Open Multimodal Generative Pre-trained Transformer for BioMedicine","date":"2023-08-18","arxiv_id":"2308.09442","repositories_listed":1,"syntology":null},{"url":"/paper/m3ke-a-massive-multi-level-multi-subject","title":"M3KE: A Massive Multi-Level Multi-Subject Knowledge Evaluation Benchmark for Chinese Large Language Models","date":"2023-05-17","arxiv_id":"2305.10263","repositories_listed":1,"syntology":null},{"url":"/paper/towards-expert-level-medical-question","title":"Towards Expert-Level Medical Question Answering with Large Language Models","date":"2023-05-16","arxiv_id":"2305.09617","repositories_listed":1,"syntology":null},{"url":"/paper/frenchmedmcqa-a-french-multiple-choice-1","title":"FrenchMedMCQA: A French Multiple-Choice Question Answering Dataset for Medical domain","date":"2023-04-09","arxiv_id":"2304.04280","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-encode-clinical","title":"Large Language Models Encode Clinical Knowledge","date":"2022-12-26","arxiv_id":"2212.13138","repositories_listed":1,"syntology":null},{"url":"/paper/galactica-a-large-language-model-for-science-1","title":"Galactica: A Large Language Model for Science","date":"2022-11-16","arxiv_id":"2211.09085","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/leveraging-large-language-models-for-multiple","title":"Leveraging Large Language Models for Multiple Choice Question Answering","date":"2022-10-22","arxiv_id":"2210.12353","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/can-large-language-models-reason-about","title":"Can large language models reason about medical questions?","date":"2022-07-17","arxiv_id":"2207.08143","repositories_listed":1,"syntology":{"n":10,"n_ran":0,"n_unverified":10,"n_pointer_only":0}}],"syntology_records":13,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}