{"url":"/task/mmlu","name":"MMLU","slug":"mmlu","description_markdown":null,"categories":[{"name":"Knowledge Base","url":"/area/knowledge-base"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":340,"papers_with_code":156,"benchmarks":1,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/mmlu-on-mmlu-pro","slug":"mmlu-on-mmlu-pro","dataset":"MMLU-Pro","dataset_url":"/dataset/mmlu-pro","rows_in_archive":1,"metrics":["0-shot MRR"],"first_row_in_archive_order":{"model":"Orange-mini","paper_title":"MyGO Multiplex CoT: A Method for Self-Reflection in Large Language Models via Double Chain of Thought Thinking","paper_url":"/paper/mygo-multiplex-cot-a-method-for-self","paper_date":"2025-01-20","arxiv_id":"2501.13117","code_links":[{"title":"data-dream-gdsp/Multiplex-CoT","url":"https://github.com/data-dream-gdsp/Multiplex-CoT"}],"syntology":null}},{"leaderboard":null,"slug":"mmlu-on-mmlu-5-shots","dataset":"mmlu (5-shots)","dataset_url":null,"rows_in_archive":0,"metrics":["acc"],"first_row_in_archive_order":null},{"leaderboard":null,"slug":"mmlu-on-mmlu-chat-cot","dataset":"mmlu (chat CoT)","dataset_url":null,"rows_in_archive":0,"metrics":["exact_match"],"first_row_in_archive_order":null}],"datasets":[{"url":"/dataset/mmlu-pro","name":"MMLU-Pro","full_name":"","num_papers_in_archive":150}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":156,"tagged_in_all":340,"items":[{"url":"/paper/scaling-instruction-finetuned-language-models","title":"Scaling Instruction-Finetuned Language Models","date":"2022-10-20","arxiv_id":"2210.11416","repositories_listed":9,"syntology":{"n":17,"n_ran":8,"n_unverified":9,"n_pointer_only":2}},{"url":"/paper/chatglm-a-family-of-large-language-models","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","date":"2024-06-18","arxiv_id":"2406.12793","repositories_listed":7,"syntology":{"n":29,"n_ran":15,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/qwen2-technical-report","title":"Qwen2 Technical Report","date":"2024-07-15","arxiv_id":"2407.10671","repositories_listed":6,"syntology":null},{"url":"/paper/tinybenchmarks-evaluating-llms-with-fewer","title":"tinyBenchmarks: evaluating LLMs with fewer examples","date":"2024-02-22","arxiv_id":"2402.14992","repositories_listed":4,"syntology":{"n":39,"n_ran":10,"n_unverified":29,"n_pointer_only":17}},{"url":"/paper/datacomp-lm-in-search-of-the-next-generation","title":"DataComp-LM: In search of the next generation of training sets for language models","date":"2024-06-17","arxiv_id":"2406.11794","repositories_listed":3,"syntology":{"n":22,"n_ran":22,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/are-we-done-with-mmlu","title":"Are We Done with MMLU?","date":"2024-06-06","arxiv_id":"2406.04127","repositories_listed":3,"syntology":{"n":8,"n_ran":2,"n_unverified":6,"n_pointer_only":1}},{"url":"/paper/replug-retrieval-augmented-black-box-language","title":"REPLUG: Retrieval-Augmented Black-Box Language Models","date":"2023-01-30","arxiv_id":"2301.12652","repositories_listed":3,"syntology":{"n":13,"n_ran":0,"n_unverified":13,"n_pointer_only":13}},{"url":"/paper/lm2-large-memory-models","title":"LM2: Large Memory Models","date":"2025-02-09","arxiv_id":"2502.06049","repositories_listed":2,"syntology":null},{"url":"/paper/what-matters-in-transformers-not-all","title":"What Matters in Transformers? Not All Attention is Needed","date":"2024-06-22","arxiv_id":"2406.15786","repositories_listed":2,"syntology":{"n":10,"n_ran":5,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/mmlu-pro-a-more-robust-and-challenging-multi","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","date":"2024-06-03","arxiv_id":"2406.01574","repositories_listed":2,"syntology":{"n":12,"n_ran":9,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/owlore-outlier-weighed-layerwise-sampled-low","title":"OwLore: Outlier-weighed Layerwise Sampled Low-Rank Projection for Memory-Efficient LLM Fine-tuning","date":"2024-05-28","arxiv_id":"2405.18380","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/efficient-multi-prompt-evaluation-of-llms","title":"Efficient multi-prompt evaluation of LLMs","date":"2024-05-27","arxiv_id":"2405.17202","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/make-your-llm-fully-utilize-the-context","title":"Make Your LLM Fully Utilize the Context","date":"2024-04-25","arxiv_id":"2404.16811","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/a-strongreject-for-empty-jailbreaks","title":"A StrongREJECT for Empty Jailbreaks","date":"2024-02-15","arxiv_id":"2402.10260","repositories_listed":2,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/baichuan-2-open-large-scale-language-models","title":"Baichuan 2: Open Large-scale Language Models","date":"2023-09-19","arxiv_id":"2309.10305","repositories_listed":2,"syntology":null},{"url":"/paper/red-teaming-large-language-models-using-chain","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","date":"2023-08-18","arxiv_id":"2308.09662","repositories_listed":2,"syntology":null},{"url":"/paper/augmentation-adapted-retriever-improves","title":"Augmentation-Adapted Retriever Improves Generalization of Language Models as Generic Plug-In","date":"2023-05-27","arxiv_id":"2305.17331","repositories_listed":2,"syntology":{"n":11,"n_ran":5,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/art-automatic-multi-step-reasoning-and-tool","title":"ART: Automatic multi-step reasoning and tool-use for large language models","date":"2023-03-16","arxiv_id":"2303.09014","repositories_listed":2,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/few-shot-learning-with-retrieval-augmented","title":"Atlas: Few-shot Learning with Retrieval Augmented Language Models","date":"2022-08-05","arxiv_id":"2208.03299","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/unifying-language-learning-paradigms","title":"UL2: Unifying Language Learning Paradigms","date":"2022-05-10","arxiv_id":"2205.05131","repositories_listed":2,"syntology":{"n":16,"n_ran":0,"n_unverified":16,"n_pointer_only":0}},{"url":"/paper/training-compute-optimal-large-language","title":"Training Compute-Optimal Large Language Models","date":"2022-03-29","arxiv_id":"2203.15556","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_unverified":3,"n_pointer_only":4}},{"url":"/paper/step-wise-policy-for-rare-tool-knowledge","title":"Step-wise Policy for Rare-tool Knowledge (SPaRK): Offline RL that Drives Diverse Tool Use in LLMs","date":"2025-07-15","arxiv_id":"2507.11371","repositories_listed":1,"syntology":null},{"url":"/paper/the-delta-learning-hypothesis-preference","title":"The Delta Learning Hypothesis: Preference Tuning on Weak Data can Yield Strong Gains","date":"2025-07-08","arxiv_id":"2507.06187","repositories_listed":1,"syntology":null},{"url":"/paper/growing-transformers-modular-composition-and","title":"Growing Transformers: Modular Composition and Layer-wise Expansion on a Frozen Substrate","date":"2025-07-08","arxiv_id":"2507.07129","repositories_listed":1,"syntology":null},{"url":"/paper/emergent-semantics-beyond-token-embeddings","title":"Emergent Semantics Beyond Token Embeddings: Transformer LMs with Frozen Visual Unicode Representations","date":"2025-07-07","arxiv_id":"2507.04886","repositories_listed":1,"syntology":null},{"url":"/paper/any4-learned-4-bit-numeric-representation-for","title":"any4: Learned 4-bit Numeric Representation for LLMs","date":"2025-07-07","arxiv_id":"2507.04610","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/do-language-models-mirror-human-confidence","title":"Do Language Models Mirror Human Confidence? Exploring Psychological Insights to Address Overconfidence in LLMs","date":"2025-05-31","arxiv_id":"2506.00582","repositories_listed":1,"syntology":null},{"url":"/paper/helm-hyperbolic-large-language-models-via","title":"HELM: Hyperbolic Large Language Models via Mixture-of-Curvature Experts","date":"2025-05-30","arxiv_id":"2505.24722","repositories_listed":1,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/simulating-training-data-leakage-in-multiple","title":"Simulating Training Data Leakage in Multiple-Choice Benchmarks for LLM Evaluation","date":"2025-05-30","arxiv_id":"2505.24263","repositories_listed":1,"syntology":null},{"url":"/paper/silvr-a-simple-language-based-video-reasoning","title":"SiLVR: A Simple Language-based Video Reasoning Framework","date":"2025-05-30","arxiv_id":"2505.24869","repositories_listed":1,"syntology":null}],"syntology_records":19,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}