{"url":"/task/multi-task-language-understanding","name":"Multi-task Language Understanding","slug":"multi-task-language-understanding","description_markdown":"The test covers 57 tasks including elementary mathematics, US history, computer science, law, and more. https://arxiv.org/pdf/2009.03300.pdf","categories":[{"name":"Methodology","url":"/area/methodology"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":57,"papers_with_code":44,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":5,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/multi-task-language-understanding-on-mmlu","slug":"multi-task-language-understanding-on-mmlu","dataset":"MML","dataset_url":"/dataset/mmlu","rows_in_archive":44,"metrics":["Average (%)"],"first_row_in_archive_order":{"model":"GPT-4 o1(300b)","paper_title":"GPT-4o as the Gold Standard: A Scalable and General Purpose Approach to Filter Language Model Pretraining Data","paper_url":"/paper/sieve-general-purpose-data-filtering-system","paper_date":"2024-10-03","arxiv_id":"2410.02755","code_links":[],"syntology":null}},{"leaderboard":"/sota/multi-task-language-understanding-on-bbh-nlp","slug":"multi-task-language-understanding-on-bbh-nlp","dataset":"BBH-nlp","dataset_url":"/dataset/big-bench","rows_in_archive":15,"metrics":["Average (%)"],"first_row_in_archive_order":{"model":"Qwen2.5-72B","paper_title":null,"paper_url":null,"paper_date":"","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/multi-task-language-understanding-on-mgsm","slug":"multi-task-language-understanding-on-mgsm","dataset":"MGSM","dataset_url":"/dataset/mgsm","rows_in_archive":12,"metrics":["Average (%)"],"first_row_in_archive_order":{"model":"PaLM 2 (few-shot, k=8, SC)","paper_title":"PaLM 2 Technical Report","paper_url":"/paper/palm-2-technical-report-1","paper_date":"2023-05-17","arxiv_id":"2305.10403","code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}],"syntology":null}},{"leaderboard":"/sota/multi-task-language-understanding-on-bbh-alg","slug":"multi-task-language-understanding-on-bbh-alg","dataset":"BBH-alg","dataset_url":"/dataset/big-bench","rows_in_archive":7,"metrics":["Average (%)"],"first_row_in_archive_order":{"model":"code-davinci-002 175B (CoT)","paper_title":"Evaluating Large Language Models Trained on Code","paper_url":"/paper/evaluating-large-language-models-trained-on","paper_date":"2021-07-07","arxiv_id":"2107.03374","code_links":[{"title":"THUDM/CodeGeeX","url":"https://github.com/THUDM/CodeGeeX"},{"title":"ncoop57/gpt-code-clippy","url":"https://github.com/ncoop57/gpt-code-clippy"},{"title":"codedotal/gpt-code-clippy","url":"https://github.com/codedotal/gpt-code-clippy"},{"title":"openai/human-eval","url":"https://github.com/openai/human-eval"},{"title":"vhellendoorn/code-lms","url":"https://github.com/vhellendoorn/code-lms"},{"title":"glouppe/info8010-deep-learning","url":"https://github.com/glouppe/info8010-deep-learning"},{"title":"microsoft/PythonProgrammingPuzzles","url":"https://github.com/microsoft/PythonProgrammingPuzzles"},{"title":"fsoft-ai4code/codecapybara","url":"https://github.com/fsoft-ai4code/codecapybara"},{"title":"codefuse-ai/codefuse-evaluation","url":"https://github.com/codefuse-ai/codefuse-evaluation"},{"title":"my-other-github-account/llm-humaneval-benchmarks","url":"https://github.com/my-other-github-account/llm-humaneval-benchmarks"},{"title":"superli3/codenavi","url":"https://github.com/superli3/codenavi"},{"title":"superli3/CYRMPR","url":"https://github.com/superli3/CYRMPR"},{"title":"2796gaurav/human-eval","url":"https://github.com/2796gaurav/human-eval"}],"syntology":{"n":39,"n_ran":6,"n_unverified":33,"n_pointer_only":2}}},{"leaderboard":"/sota/multi-task-language-understanding-on-mmlu-5-1","slug":"multi-task-language-understanding-on-mmlu-5-1","dataset":"MMLU (5-Shot)","dataset_url":"/dataset/mmlu","rows_in_archive":1,"metrics":["MMLU (5-shot)"],"first_row_in_archive_order":{"model":"Sakalti/ultiima-78B","paper_title":"MERGE: Fast Private Text Generation","paper_url":"/paper/merge-fast-private-text-generation","paper_date":"2023-05-25","arxiv_id":"2305.15769","code_links":[{"title":"liangzid/MERGE","url":"https://github.com/liangzid/MERGE"}],"syntology":null}}],"datasets":[{"url":"/dataset/mmlu","name":"MML","full_name":"Massive Multitask Language Understanding","num_papers_in_archive":1922},{"url":"/dataset/bbh","name":"BBH","full_name":"BIG-Bench Hard","num_papers_in_archive":352},{"url":"/dataset/big-bench","name":"BIG-bench","full_name":"Beyond the Imitation Game Benchmark","num_papers_in_archive":349},{"url":"/dataset/mgsm","name":"MGSM","full_name":"Multilingual Grade School Math","num_papers_in_archive":107},{"url":"/dataset/tasksource","name":"Tasksource","full_name":"","num_papers_in_archive":3}],"subtasks":[],"parent_tasks":[{"url":"/task/multi-task-learning","name":"Multi-Task Learning"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":44,"tagged_in_all":57,"items":[{"url":"/paper/language-models-are-few-shot-learners","title":"Language Models are Few-Shot Learners","date":"2020-05-28","arxiv_id":"2005.14165","repositories_listed":67,"syntology":{"n":65,"n_ran":15,"n_unverified":50,"n_pointer_only":4}},{"url":"/paper/roberta-a-robustly-optimized-bert-pretraining","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","date":"2019-07-26","arxiv_id":"1907.11692","repositories_listed":67,"syntology":{"n":48,"n_ran":22,"n_unverified":26,"n_pointer_only":23}},{"url":"/paper/llama-open-and-efficient-foundation-language-1","title":"LLaMA: Open and Efficient Foundation Language Models","date":"2023-02-27","arxiv_id":"2302.13971","repositories_listed":57,"syntology":{"n":58,"n_ran":26,"n_unverified":32,"n_pointer_only":4}},{"url":"/paper/albert-a-lite-bert-for-self-supervised","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","date":"2019-09-26","arxiv_id":"1909.11942","repositories_listed":48,"syntology":{"n":126,"n_ran":46,"n_unverified":80,"n_pointer_only":22}},{"url":"/paper/language-models-are-unsupervised-multitask","title":"Language Models are Unsupervised Multitask Learners","date":"2019-02-14","arxiv_id":null,"repositories_listed":21,"syntology":null},{"url":"/paper/llama-2-open-foundation-and-fine-tuned-chat","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","date":"2023-07-18","arxiv_id":"2307.09288","repositories_listed":19,"syntology":{"n":52,"n_ran":31,"n_unverified":21,"n_pointer_only":16}},{"url":"/paper/measuring-massive-multitask-language","title":"Measuring Massive Multitask Language Understanding","date":"2020-09-07","arxiv_id":"2009.03300","repositories_listed":18,"syntology":{"n":26,"n_ran":5,"n_unverified":21,"n_pointer_only":1}},{"url":"/paper/evaluating-large-language-models-trained-on","title":"Evaluating Large Language Models Trained on Code","date":"2021-07-07","arxiv_id":"2107.03374","repositories_listed":13,"syntology":{"n":39,"n_ran":6,"n_unverified":33,"n_pointer_only":2}},{"url":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","arxiv_id":"2303.08774","repositories_listed":11,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/gpt-neox-20b-an-open-source-autoregressive-1","title":"GPT-NeoX-20B: An Open-Source Autoregressive Language Model","date":"2022-04-14","arxiv_id":"2204.06745","repositories_listed":11,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/scaling-instruction-finetuned-language-models","title":"Scaling Instruction-Finetuned Language Models","date":"2022-10-20","arxiv_id":"2210.11416","repositories_listed":9,"syntology":{"n":17,"n_ran":8,"n_unverified":9,"n_pointer_only":2}},{"url":"/paper/glm-130b-an-open-bilingual-pre-trained-model","title":"GLM-130B: An Open Bilingual Pre-trained Model","date":"2022-10-05","arxiv_id":"2210.02414","repositories_listed":9,"syntology":{"n":21,"n_ran":5,"n_unverified":16,"n_pointer_only":0}},{"url":"/paper/palm-scaling-language-modeling-with-pathways-1","title":"PaLM: Scaling Language Modeling with Pathways","date":"2022-04-05","arxiv_id":"2204.02311","repositories_listed":7,"syntology":{"n":37,"n_ran":30,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/mixtral-of-experts","title":"Mixtral of Experts","date":"2024-01-08","arxiv_id":"2401.04088","repositories_listed":6,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/mistral-7b","title":"Mistral 7B","date":"2023-10-10","arxiv_id":"2310.06825","repositories_listed":6,"syntology":{"n":11,"n_ran":9,"n_unverified":2,"n_pointer_only":1}},{"url":"/paper/the-llama-3-herd-of-models","title":"The Llama 3 Herd of Models","date":"2024-07-31","arxiv_id":"2407.21783","repositories_listed":5,"syntology":{"n":9,"n_ran":2,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/deepseek-r1-incentivizing-reasoning","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","date":"2025-01-22","arxiv_id":"2501.12948","repositories_listed":4,"syntology":null},{"url":"/paper/replug-retrieval-augmented-black-box-language","title":"REPLUG: Retrieval-Augmented Black-Box Language Models","date":"2023-01-30","arxiv_id":"2301.12652","repositories_listed":3,"syntology":{"n":13,"n_ran":0,"n_unverified":13,"n_pointer_only":13}},{"url":"/paper/scaling-language-models-methods-analysis-1","title":"Scaling Language Models: Methods, Analysis & Insights from Training Gopher","date":"2021-12-08","arxiv_id":"2112.11446","repositories_listed":3,"syntology":null},{"url":"/paper/mmlu-pro-a-more-robust-and-challenging-multi","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","date":"2024-06-03","arxiv_id":"2406.01574","repositories_listed":2,"syntology":{"n":12,"n_ran":9,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/parameter-efficient-sparsity-crafting-from","title":"Parameter-Efficient Sparsity Crafting from Dense to Mixture-of-Experts for Instruction Tuning on General Tasks","date":"2024-01-05","arxiv_id":"2401.02731","repositories_listed":2,"syntology":null},{"url":"/paper/bloomberggpt-a-large-language-model-for","title":"BloombergGPT: A Large Language Model for Finance","date":"2023-03-30","arxiv_id":"2303.17564","repositories_listed":2,"syntology":null},{"url":"/paper/few-shot-learning-with-retrieval-augmented","title":"Atlas: Few-shot Learning with Retrieval Augmented Language Models","date":"2022-08-05","arxiv_id":"2208.03299","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/unifying-language-learning-paradigms","title":"UL2: Unifying Language Learning Paradigms","date":"2022-05-10","arxiv_id":"2205.05131","repositories_listed":2,"syntology":{"n":16,"n_ran":0,"n_unverified":16,"n_pointer_only":0}},{"url":"/paper/training-compute-optimal-large-language","title":"Training Compute-Optimal Large Language Models","date":"2022-03-29","arxiv_id":"2203.15556","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_unverified":3,"n_pointer_only":4}},{"url":"/paper/merging-models-with-fisher-weighted-averaging","title":"Merging Models with Fisher-Weighted Averaging","date":"2021-11-18","arxiv_id":"2111.09832","repositories_listed":2,"syntology":null},{"url":"/paper/unifiedqa-crossing-format-boundaries-with-a","title":"UnifiedQA: Crossing Format Boundaries With a Single QA System","date":"2020-05-02","arxiv_id":"2005.00700","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/tumlu-a-unified-and-native-language","title":"TUMLU: A Unified and Native Language Understanding Benchmark for Turkic Languages","date":"2025-02-16","arxiv_id":"2502.11020","repositories_listed":1,"syntology":null},{"url":"/paper/mmlu-cf-a-contamination-free-multi-task","title":"MMLU-CF: A Contamination-free Multi-task Language Understanding Benchmark","date":"2024-12-19","arxiv_id":"2412.15194","repositories_listed":1,"syntology":null},{"url":"/paper/llama-3-meets-moe-efficient-upcycling","title":"Llama 3 Meets MoE: Efficient Upcycling","date":"2024-12-13","arxiv_id":"2412.09952","repositories_listed":1,"syntology":null}],"syntology_records":21,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}