{"url":"/task/language-model-evaluation","name":"Language Model Evaluation","slug":"language-model-evaluation","description_markdown":"The task of using LLMs as evaluators of large language and vision language models.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"derived"},"counts":{"papers_tagged":69,"papers_with_code":31,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":0,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":31,"tagged_in_all":69,"items":[{"url":"/paper/challenge-llms-to-reason-about-reasoning-a","title":"MR-GSM8K: A Meta-Reasoning Benchmark for Large Language Model Evaluation","date":"2023-12-28","arxiv_id":"2312.17080","repositories_listed":2,"syntology":null},{"url":"/paper/scieval-a-multi-level-large-language-model","title":"SciEval: A Multi-Level Large Language Model Evaluation Benchmark for Scientific Research","date":"2023-08-25","arxiv_id":"2308.13149","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/bigbio-a-framework-for-data-centric","title":"BigBIO: A Framework for Data-Centric Biomedical Natural Language Processing","date":"2022-06-30","arxiv_id":"2206.15076","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/fable-a-novel-data-flow-analysis-benchmark-on","title":"FABLE: A Novel Data-Flow Analysis Benchmark on Procedural Text for Large Language Model Evaluation","date":"2025-05-30","arxiv_id":"2505.24258","repositories_listed":1,"syntology":null},{"url":"/paper/role-playing-evaluation-for-large-language","title":"Role-Playing Evaluation for Large Language Models","date":"2025-05-19","arxiv_id":"2505.13157","repositories_listed":1,"syntology":null},{"url":"/paper/m-absa-a-multilingual-dataset-for-aspect","title":"M-ABSA: A Multilingual Dataset for Aspect-Based Sentiment Analysis","date":"2025-02-17","arxiv_id":"2502.11824","repositories_listed":1,"syntology":null},{"url":"/paper/environmental-large-language-model-evaluation","title":"Environmental large language model Evaluation (ELLE) dataset: A Benchmark for Evaluating Generative AI applications in Eco-environment Domain","date":"2025-01-10","arxiv_id":"2501.06277","repositories_listed":1,"syntology":null},{"url":"/paper/automated-generation-of-challenging-multiple","title":"Automated Generation of Challenging Multiple-Choice Questions for Vision Language Model Evaluation","date":"2025-01-06","arxiv_id":"2501.03225","repositories_listed":1,"syntology":null},{"url":"/paper/template-matters-understanding-the-role-of","title":"Template Matters: Understanding the Role of Instruction Templates in Multimodal Language Model Evaluation and Training","date":"2024-12-11","arxiv_id":"2412.08307","repositories_listed":1,"syntology":null},{"url":"/paper/c-2-leva-toward-comprehensive-and","title":"C$^2$LEVA: Toward Comprehensive and Contamination-Free Language Model Evaluation","date":"2024-12-06","arxiv_id":"2412.04947","repositories_listed":1,"syntology":null},{"url":"/paper/dart-eval-a-comprehensive-dna-language-model","title":"DART-Eval: A Comprehensive DNA Language Model Evaluation Benchmark on Regulatory DNA","date":"2024-12-06","arxiv_id":"2412.05430","repositories_listed":1,"syntology":{"n":16,"n_ran":1,"n_unverified":15,"n_pointer_only":16}},{"url":"/paper/large-language-model-evaluation-via-matrix-1","title":"Large Language Model Evaluation via Matrix Nuclear-Norm","date":"2024-10-14","arxiv_id":"2410.10672","repositories_listed":1,"syntology":null},{"url":"/paper/enterprise-benchmarks-for-large-language","title":"Enterprise Benchmarks for Large Language Model Evaluation","date":"2024-10-11","arxiv_id":"2410.12857","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-the-bias-of-large-language-model","title":"Mitigating the Bias of Large Language Model Evaluation","date":"2024-09-25","arxiv_id":"2409.16788","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/a-suite-for-acoustic-language-model","title":"Salmon: A Suite for Acoustic Language Model Evaluation","date":"2024-09-11","arxiv_id":"2409.07437","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/inference-time-decontamination-reusing-leaked","title":"Inference-Time Decontamination: Reusing Leaked Benchmarks for Large Language Model Evaluation","date":"2024-06-20","arxiv_id":"2406.13990","repositories_listed":1,"syntology":null},{"url":"/paper/fennec-fine-grained-language-model-evaluation","title":"Fennec: Fine-grained Language Model Evaluation and Correction Extended through Branching and Bridging","date":"2024-05-20","arxiv_id":"2405.12163","repositories_listed":1,"syntology":null},{"url":"/paper/paraphrase-and-solve-exploring-and-exploiting","title":"Paraphrase and Solve: Exploring and Exploiting the Impact of Surface Form on Mathematical Reasoning in Large Language Models","date":"2024-04-17","arxiv_id":"2404.11500","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/evalverse-unified-and-accessible-library-for","title":"Evalverse: Unified and Accessible Library for Large Language Model Evaluation","date":"2024-04-01","arxiv_id":"2404.00943","repositories_listed":1,"syntology":null},{"url":"/paper/towards-personalized-evaluation-of-large","title":"Towards Personalized Evaluation of Large Language Models with An Anonymous Crowd-Sourcing Platform","date":"2024-03-13","arxiv_id":"2403.08305","repositories_listed":1,"syntology":null},{"url":"/paper/arabicmmlu-assessing-massive-multitask","title":"ArabicMMLU: Assessing Massive Multitask Language Understanding in Arabic","date":"2024-02-20","arxiv_id":"2402.12840","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_unverified":3,"n_pointer_only":6}},{"url":"/paper/avoiding-data-contamination-in-language-model","title":"LatestEval: Addressing Data Contamination in Language Model Evaluation through Dynamic and Time-Sensitive Test Construction","date":"2023-12-19","arxiv_id":"2312.12343","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":6}},{"url":"/paper/catwalk-a-unified-language-model-evaluation","title":"Catwalk: A Unified Language Model Evaluation Framework for Many Datasets","date":"2023-12-15","arxiv_id":"2312.10253","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-language-models-for-code","title":"Unifying the Perspectives of NLP and Software Engineering: A Survey on Language Models for Code","date":"2023-11-14","arxiv_id":"2311.07989","repositories_listed":1,"syntology":null},{"url":"/paper/estimating-contamination-via-perplexity","title":"Estimating Contamination via Perplexity: Quantifying Memorisation in Language Model Evaluation","date":"2023-09-19","arxiv_id":"2309.10677","repositories_listed":1,"syntology":null},{"url":"/paper/agentsims-an-open-source-sandbox-for-large","title":"AgentSims: An Open-Source Sandbox for Large Language Model Evaluation","date":"2023-08-08","arxiv_id":"2308.04026","repositories_listed":1,"syntology":null},{"url":"/paper/flask-fine-grained-language-model-evaluation","title":"FLASK: Fine-grained Language Model Evaluation based on Alignment Skill Sets","date":"2023-07-20","arxiv_id":"2307.10928","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":5}},{"url":"/paper/csts-conditional-semantic-textual-similarity","title":"C-STS: Conditional Semantic Textual Similarity","date":"2023-05-24","arxiv_id":"2305.15093","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":4}},{"url":"/paper/pronto-language-model-evaluations-for-859","title":"PrOnto: Language Model Evaluations for 859 Languages","date":"2023-05-22","arxiv_id":"2305.12612","repositories_listed":1,"syntology":null},{"url":"/paper/zjuklab-at-semeval-2021-task-4-negative","title":"ZJUKLAB at SemEval-2021 Task 4: Negative Augmentation with Language Model for Reading Comprehension of Abstract Meaning","date":"2021-02-25","arxiv_id":"2102.12828","repositories_listed":1,"syntology":null}],"syntology_records":10,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}