{"url":"/dataset/elevater","name":"ELEVATER","full_name":"Evaluation of Language-augmented Visual Task-level Transfer","description_markdown":"The ELEVATER benchmark is a collection of resources for training, evaluating, and analyzing language-image models on image classification and object detection. ELEVATER consists of:\r\n\r\n- Benchmark: A benchmark suite that consists of 20 image classification datasets and 35 object detection datasets, augmented with external knowledge\r\n- Toolkit: An automatic hyper-parameter tuning toolkit; Strong language-augmented efficient model adaptation methods.\r\n- Baseline: Pre-trained language-free and language-augmented visual models.\r\n- Knowledge: A platform to study the benefit of external knowledge for vision problems.\r\n- Evaluation Metrics: Sample-efficiency (zero-, few-, and full-shot) and Parameter-efficiency.\r\n- Leaderboard: A public leaderboard to track performance on the benchmark\r\n\r\nThe ultimate goal of ELEVATER is to drive research in the development of language-image models to tackle core computer vision problems in the wild.","description_withheld":null,"homepage":"https://computer-vision-in-the-wild.github.io/ELEVATER/","introduced_date":"2022-04-19","introduced_date_note":null,"introduced_by":{"paper":"/paper/elevater-a-benchmark-and-toolkit-for","title":"ELEVATER: A Benchmark and Toolkit for Evaluating Language-Augmented Visual Models","first_author":"Chunyuan Li","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Image Classification","url":"/task/image-classification","datasets_with_task":"/datasets/task/image-classification"},{"name":"Object Detection","url":"/task/object-detection","datasets_with_task":"/datasets/task/object-detection"},{"name":"Few-Shot Object Detection","url":"/task/few-shot-object-detection","datasets_with_task":"/datasets/task/few-shot-object-detection"},{"name":"Zero-Shot Object Detection","url":"/task/zero-shot-object-detection","datasets_with_task":"/datasets/task/zero-shot-object-detection"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["ELEVATER","ODinW Full-shot 35 Tasks"],"data_loaders":[],"num_papers_in_archive":25,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/object-detection-on-odinw-full-shot-35-tasks","task":"Object Detection","dataset_variant":"ODinW Full-shot 35 Tasks","rows":2,"metrics":["AP"],"first_row_in_archive_order":{"model":"Grounding DINO 1.5 Pro","paper":"/paper/grounding-dino-1-5-advance-the-edge-of-open","metrics":{"AP":"72.4"},"code_links":[{"title":"mit-han-lab/efficientvit","url":"https://github.com/mit-han-lab/efficientvit"},{"title":"idea-research/grounded-sam-2","url":"https://github.com/idea-research/grounded-sam-2"},{"title":"idea-research/grounding-dino-1.5-api","url":"https://github.com/idea-research/grounding-dino-1.5-api"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-detection-on-elevater","task":"Object Detection","dataset_variant":"ELEVATER","rows":1,"metrics":["AP"],"first_row_in_archive_order":{"model":"GLIP-T","paper":"/paper/elevater-a-benchmark-and-toolkit-for","metrics":{"AP":"62.6"},"code_links":[{"title":"microsoft/GLIP","url":"https://github.com/microsoft/GLIP"},{"title":"computer-vision-in-the-wild/cvinw_readings","url":"https://github.com/computer-vision-in-the-wild/cvinw_readings"},{"title":"microsoft/esvit","url":"https://github.com/microsoft/esvit"},{"title":"microsoft/unicl","url":"https://github.com/microsoft/unicl"},{"title":"eric-ai-lab/pevit","url":"https://github.com/eric-ai-lab/pevit"},{"title":"Computer-Vision-in-the-Wild/Elevater_Toolkit_IC","url":"https://github.com/Computer-Vision-in-the-Wild/Elevater_Toolkit_IC"},{"title":"sincerass/mvlpt","url":"https://github.com/sincerass/mvlpt"},{"title":"microsoft/klite","url":"https://github.com/microsoft/klite"},{"title":"rsCPSyEu/ovd_cod","url":"https://github.com/rsCPSyEu/ovd_cod"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/grounding-dino-1-5-advance-the-edge-of-open","title":"Grounding DINO 1.5: Advance the \"Edge\" of Open-Set Object Detection","date":"2024-05-16","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/elevater-a-benchmark-and-toolkit-for","title":"ELEVATER: A Benchmark and Toolkit for Evaluating Language-Augmented Visual Models","date":"2022-04-19","rows_on_this_dataset":2,"code_links":9,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":20,"samples_ran":12,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":2,"samples_harvested":22,"samples_ran":13,"samples_unverified":9,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}