{"url":"/task/benchmarking","name":"Benchmarking","slug":"benchmarking","description_markdown":null,"categories":[{"name":"Miscellaneous","url":"/area/miscellaneous"},{"name":"Robots","url":"/area/robots"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":5548,"papers_with_code":2658,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":10,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/benchmarking-on-cloudeval-yaml","slug":"benchmarking-on-cloudeval-yaml","dataset":"CloudEval-YAML","dataset_url":null,"rows_in_archive":1,"metrics":["ACC"],"first_row_in_archive_order":{"model":"GPT-4 Turbo","paper_title":"CloudEval-YAML: A Practical Benchmark for Cloud Configuration Generation","paper_url":"/paper/cloudeval-yaml-a-practical-benchmark-for","paper_date":"2023-11-10","arxiv_id":"2401.06786","code_links":[{"title":"alibaba/cloudeval-yaml","url":"https://github.com/alibaba/cloudeval-yaml"}],"syntology":null}},{"leaderboard":"/sota/benchmarking-on-wiki-40b","slug":"benchmarking-on-wiki-40b","dataset":"Wiki-40B","dataset_url":"/dataset/wiki-40b","rows_in_archive":1,"metrics":["Perplexity"],"first_row_in_archive_order":{"model":"OutEffHop-Bert_base","paper_title":"Outlier-Efficient Hopfield Layers for Large Transformer-Based Models","paper_url":"/paper/outlier-efficient-hopfield-layers-for-large","paper_date":"2024-04-04","arxiv_id":"2404.03828","code_links":[{"title":"magics-lab/outeffhop","url":"https://github.com/magics-lab/outeffhop"}],"syntology":{"n":13,"n_ran":10,"n_unverified":3,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/wiki-40b","name":"Wiki-40B","full_name":"","num_papers_in_archive":30},{"url":"/dataset/cropandweed-dataset","name":"CropAndWeed","full_name":"","num_papers_in_archive":10},{"url":"/dataset/europarl-asr","name":"Europarl-ASR","full_name":"","num_papers_in_archive":8},{"url":"/dataset/fluidlab","name":"FluidLab","full_name":"","num_papers_in_archive":5},{"url":"/dataset/apron-dataset","name":"Apron Dataset","full_name":"","num_papers_in_archive":1},{"url":"/dataset/coco-n-medium","name":"COCO-N Medium","full_name":"","num_papers_in_archive":1},{"url":"/dataset/csts","name":"CSTS","full_name":"Correlation Structures in Time Series","num_papers_in_archive":1},{"url":"/dataset/genotex","name":"GenoTEX","full_name":"An LLM Agent Benchmark for Automated Gene Expression Data Analysis","num_papers_in_archive":1},{"url":"/dataset/psocr","name":"PsOCR","full_name":"Pashto OCR Dataset","num_papers_in_archive":1},{"url":"/dataset/semanticsugarbeets","name":"SemanticSugarBeets","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/data-free-knowledge-distillation","name":"Data-free Knowledge Distillation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":2658,"tagged_in_all":5548,"items":[{"url":"/paper/mmdetection-open-mmlab-detection-toolbox-and","title":"MMDetection: Open MMLab Detection Toolbox and Benchmark","date":"2019-06-17","arxiv_id":"1906.07155","repositories_listed":142,"syntology":{"n":82,"n_ran":14,"n_unverified":68,"n_pointer_only":0}},{"url":"/paper/learning-transferable-visual-models-from","title":"Learning Transferable Visual Models From Natural Language Supervision","date":"2021-02-26","arxiv_id":"2103.00020","repositories_listed":82,"syntology":{"n":20,"n_ran":16,"n_unverified":4,"n_pointer_only":16}},{"url":"/paper/fashion-mnist-a-novel-image-dataset-for","title":"Fashion-MNIST: a Novel Image Dataset for Benchmarking Machine Learning Algorithms","date":"2017-08-25","arxiv_id":"1708.07747","repositories_listed":37,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/cider-consensus-based-image-description","title":"CIDEr: Consensus-based Image Description Evaluation","date":"2014-11-20","arxiv_id":"1411.5726","repositories_listed":24,"syntology":{"n":32,"n_ran":12,"n_unverified":20,"n_pointer_only":27}},{"url":"/paper/the-starcraft-multi-agent-challenge","title":"The StarCraft Multi-Agent Challenge","date":"2019-02-11","arxiv_id":"1902.04043","repositories_listed":23,"syntology":{"n":15,"n_ran":6,"n_unverified":9,"n_pointer_only":13}},{"url":"/paper/benchmarking-graph-neural-networks","title":"Benchmarking Graph Neural Networks","date":"2020-03-02","arxiv_id":"2003.00982","repositories_listed":15,"syntology":{"n":23,"n_ran":1,"n_unverified":22,"n_pointer_only":0}},{"url":"/paper/benchmarking-deep-reinforcement-learning-for","title":"Benchmarking Deep Reinforcement Learning for Continuous Control","date":"2016-04-22","arxiv_id":"1604.06778","repositories_listed":15,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/benchmarking-neural-network-robustness-to-2","title":"Benchmarking Neural Network Robustness to Common Corruptions and Perturbations","date":"2019-03-28","arxiv_id":"1903.12261","repositories_listed":14,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/ms-marco-a-human-generated-machine-reading","title":"MS MARCO: A Human Generated MAchine Reading COmprehension Dataset","date":"2016-11-28","arxiv_id":"1611.09268","repositories_listed":14,"syntology":{"n":33,"n_ran":3,"n_unverified":30,"n_pointer_only":0}},{"url":"/paper/habitat-a-platform-for-embodied-ai-research","title":"Habitat: A Platform for Embodied AI Research","date":"2019-04-02","arxiv_id":"1904.01201","repositories_listed":13,"syntology":{"n":15,"n_ran":3,"n_unverified":12,"n_pointer_only":15}},{"url":"/paper/ai-fairness-360-an-extensible-toolkit-for","title":"AI Fairness 360: An Extensible Toolkit for Detecting, Understanding, and Mitigating Unwanted Algorithmic Bias","date":"2018-10-03","arxiv_id":"1810.01943","repositories_listed":13,"syntology":{"n":11,"n_ran":3,"n_unverified":8,"n_pointer_only":1}},{"url":"/paper/technical-report-on-the-cleverhans-v210","title":"Technical Report on the CleverHans v2.1.0 Adversarial Examples Library","date":"2016-10-03","arxiv_id":"1610.00768","repositories_listed":13,"syntology":{"n":26,"n_ran":3,"n_unverified":23,"n_pointer_only":26}},{"url":"/paper/a-large-annotated-medical-image-dataset-for","title":"A large annotated medical image dataset for the development and evaluation of segmentation algorithms","date":"2019-02-25","arxiv_id":"1902.09063","repositories_listed":12,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/coco-a-platform-for-comparing-continuous","title":"COCO: A Platform for Comparing Continuous Optimizers in a Black-Box Setting","date":"2016-03-29","arxiv_id":"1603.08785","repositories_listed":12,"syntology":null},{"url":"/paper/multitask-learning-and-benchmarking-with","title":"Multitask learning and benchmarking with clinical time series data","date":"2017-03-22","arxiv_id":"1703.07771","repositories_listed":11,"syntology":{"n":40,"n_ran":4,"n_unverified":36,"n_pointer_only":5}},{"url":"/paper/benchmarking-generalization-via-in-context","title":"Super-NaturalInstructions: Generalization via Declarative Instructions on 1600+ NLP Tasks","date":"2022-04-16","arxiv_id":"2204.07705","repositories_listed":10,"syntology":{"n":28,"n_ran":7,"n_unverified":21,"n_pointer_only":4}},{"url":"/paper/on-evaluation-of-embodied-navigation-agents","title":"On Evaluation of Embodied Navigation Agents","date":"2018-07-18","arxiv_id":"1807.06757","repositories_listed":10,"syntology":null},{"url":"/paper/comparative-evaluation-of-multi-agent-deep","title":"Benchmarking Multi-Agent Deep Reinforcement Learning Algorithms in Cooperative Tasks","date":"2020-06-14","arxiv_id":"2006.07869","repositories_listed":9,"syntology":{"n":12,"n_ran":7,"n_unverified":5,"n_pointer_only":7}},{"url":"/paper/torchreid-a-library-for-deep-learning-person","title":"Torchreid: A Library for Deep Learning Person Re-Identification in Pytorch","date":"2019-10-22","arxiv_id":"1910.10093","repositories_listed":9,"syntology":{"n":24,"n_ran":1,"n_unverified":23,"n_pointer_only":0}},{"url":"/paper/benchmarking-natural-language-understanding","title":"Benchmarking Natural Language Understanding Services for building Conversational Agents","date":"2019-03-13","arxiv_id":"1903.05566","repositories_listed":9,"syntology":null},{"url":"/paper/data-splits-and-metrics-for-method","title":"Data Splits and Metrics for Method Benchmarking on Surgical Action Triplet Datasets","date":"2022-04-11","arxiv_id":"2204.05235","repositories_listed":8,"syntology":null},{"url":"/paper/multitask-prompted-training-enables-zero-shot-1","title":"Multitask Prompted Training Enables Zero-Shot Task Generalization","date":"2021-10-15","arxiv_id":"2110.08207","repositories_listed":8,"syntology":{"n":15,"n_ran":8,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/capsule-vision-2024-challenge-multi-class","title":"Capsule Vision 2024 Challenge: Multi-Class Abnormality Classification for Video Capsule Endoscopy","date":"2024-08-09","arxiv_id":"2408.04940","repositories_listed":7,"syntology":null},{"url":"/paper/the-kits19-challenge-data-300-kidney-tumor","title":"The KiTS19 Challenge Data: 300 Kidney Tumor Cases with Clinical Context, CT Semantic Segmentations, and Surgical Outcomes","date":"2019-03-31","arxiv_id":"1904.00445","repositories_listed":7,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/leaf-a-benchmark-for-federated-settings","title":"LEAF: A Benchmark for Federated Settings","date":"2018-12-03","arxiv_id":"1812.01097","repositories_listed":7,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/benchmarking-robustness-of-3d-point-cloud","title":"Benchmarking Robustness of 3D Point Cloud Recognition Against Common Corruptions","date":"2022-01-28","arxiv_id":"2201.12296","repositories_listed":6,"syntology":{"n":29,"n_ran":14,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/fuxictr-an-open-benchmark-for-click-through","title":"BARS-CTR: Open Benchmarking for Click-Through Rate Prediction","date":"2020-09-12","arxiv_id":"2009.05794","repositories_listed":6,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/the-measure-of-intelligence","title":"On the Measure of Intelligence","date":"2019-11-05","arxiv_id":"1911.01547","repositories_listed":6,"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/the-liver-tumor-segmentation-benchmark-lits","title":"The Liver Tumor Segmentation Benchmark (LiTS)","date":"2019-01-13","arxiv_id":"1901.04056","repositories_listed":6,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/retrieve-merge-predict-augmenting-tables-with","title":"Retrieve, Merge, Predict: Augmenting Tables with Data Lakes","date":"2024-02-09","arxiv_id":"2402.06282","repositories_listed":5,"syntology":null}],"syntology_records":24,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}