{"url":"/dataset/m3ke","name":"M3KE","full_name":"Massive Multi-Level Multi-Subject Knowledge Evaluation Benchmark","description_markdown":"**M3KE** is a Massive Multi-Level Multi-Subject Knowledge Evaluation benchmark, which is developed to measure knowledge acquired by Chinese large language models by testing their multitask accuracy in zero- and few-shot settings. We have collected 20,477 questions from 71 tasks. Our selection covers all major levels of Chinese education system, ranging from the primary school to college, as well as a wide variety of subjects, including humanities, history, politics, law, education, psychology, science, technology, art and religion. All questions are multiple-choice questions with four options, hence guaranteeing a standardized and unified assessment process.\r\n\r\nSource: [M3KE: A Massive Multi-Level Multi-Subject Knowledge Evaluation Benchmark for Chinese Large Language Models](https://arxiv.org/pdf/2305.10263v1.pdf)\r\n\r\nImage Source: [M3KE: A Massive Multi-Level Multi-Subject Knowledge Evaluation Benchmark for Chinese Large Language Models](https://arxiv.org/pdf/2305.10263v1.pdf)","description_withheld":null,"homepage":"https://github.com/tjunlp-lab/M3KE","introduced_date":"2023-05-17","introduced_date_note":null,"introduced_by":{"paper":"/paper/m3ke-a-massive-multi-level-multi-subject","title":"M3KE: A Massive Multi-Level Multi-Subject Knowledge Evaluation Benchmark for Chinese Large Language Models","first_author":"Chuang Liu","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Multiple Choice Question Answering (MCQA)","url":"/task/multiple-choice-qa","datasets_with_task":"/datasets/task/multiple-choice-qa"}],"languages":[],"variants":["M3KE"],"data_loaders":[],"num_papers_in_archive":14,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}