{"url":"/dataset/voc-mlt","name":"VOC-MLT","full_name":null,"description_markdown":"We construct the long-tailed version of VOC  from its 2012 train-val set. It contains 1,142 images from 20 classes, with a maximum of 775 images per class and a minimum of 4 images per class.  The ratio of head, medium, and tail classes after splitting is 6:6:8. We evaluate the performance on VOC2007 test set with 4952 images.","description_withheld":null,"homepage":"","introduced_date":"2020-07-19","introduced_date_note":null,"introduced_by":{"paper":"/paper/distribution-balanced-loss-for-multi-label","title":"Distribution-Balanced Loss for Multi-Label Classification in Long-Tailed Datasets","first_author":"Tong Wu","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Zero-Shot Learning","url":"/task/zero-shot-learning","datasets_with_task":"/datasets/task/zero-shot-learning"},{"name":"Long-tail Learning","url":"/task/long-tail-learning","datasets_with_task":"/datasets/task/long-tail-learning"},{"name":"Multi-Label Image Classification","url":"/task/multi-label-image-classification","datasets_with_task":"/datasets/task/multi-label-image-classification"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["VOC-MLT"],"data_loaders":[],"num_papers_in_archive":12,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/long-tail-learning-on-voc-mlt","task":"Long-tail Learning","dataset_variant":"VOC-MLT","rows":13,"metrics":["Average mAP"],"first_row_in_archive_order":{"model":"LMPT(ViT-B/16)","paper":"/paper/lmpt-prompt-tuning-with-class-specific","metrics":{"Average mAP":"87.88"},"code_links":[{"title":"richard-peng-xia/LMPT","url":"https://github.com/richard-peng-xia/LMPT"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-learning-on-voc-mlt","task":"Zero-Shot Learning","dataset_variant":"VOC-MLT","rows":2,"metrics":["Average mAP"],"first_row_in_archive_order":{"model":"CLIP(ResNet-50)","paper":"/paper/learning-transferable-visual-models-from","metrics":{"Average mAP":"84.30"},"code_links":[{"title":"openai/CLIP","url":"https://github.com/openai/CLIP"},{"title":"mlfoundations/open_clip","url":"https://github.com/mlfoundations/open_clip"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"facebookresearch/vissl","url":"https://github.com/facebookresearch/vissl"},{"title":"alibaba/EasyNLP","url":"https://github.com/alibaba/EasyNLP"},{"title":"apple/ml-mobileclip","url":"https://github.com/apple/ml-mobileclip"},{"title":"OML-Team/open-metric-learning","url":"https://github.com/OML-Team/open-metric-learning"},{"title":"FreddeFrallan/Multilingual-CLIP","url":"https://github.com/FreddeFrallan/Multilingual-CLIP"},{"title":"eps696/aphantasia","url":"https://github.com/eps696/aphantasia"},{"title":"muzairkhattak/multimodal-prompt-learning","url":"https://github.com/muzairkhattak/multimodal-prompt-learning"},{"title":"moein-shariatnia/OpenAI-CLIP","url":"https://github.com/moein-shariatnia/OpenAI-CLIP"},{"title":"facebookresearch/brainmagick","url":"https://github.com/facebookresearch/brainmagick"},{"title":"PaddlePaddle/PASSL","url":"https://github.com/PaddlePaddle/PASSL/blob/main/docs/Train_CLIP_model.md"},{"title":"taited/clip-score","url":"https://github.com/taited/clip-score"},{"title":"azshue/TPT","url":"https://github.com/azshue/TPT"},{"title":"clip-italian/clip-italian","url":"https://github.com/clip-italian/clip-italian"},{"title":"dhansmair/flamingo-mini","url":"https://github.com/dhansmair/flamingo-mini"},{"title":"ylqi/count-anything","url":"https://github.com/ylqi/count-anything"},{"title":"ml-jku/cloob","url":"https://github.com/ml-jku/cloob"},{"title":"sberbank-ai/ru-clip","url":"https://github.com/sberbank-ai/ru-clip"},{"title":"ai-forever/ru-clip","url":"https://github.com/ai-forever/ru-clip"},{"title":"Kaushalya/medclip","url":"https://github.com/Kaushalya/medclip"},{"title":"ajayjain/vectorascent","url":"https://github.com/ajayjain/vectorascent"},{"title":"linjieli222/hero_video_feature_extractor","url":"https://github.com/linjieli222/hero_video_feature_extractor"},{"title":"borisdayma/clip-jax","url":"https://github.com/borisdayma/clip-jax"},{"title":"sajjjadayobi/CLIPfa","url":"https://github.com/sajjjadayobi/CLIPfa"},{"title":"mertyg/post-hoc-cbm","url":"https://github.com/mertyg/post-hoc-cbm"},{"title":"mlbio-epfl/turtle","url":"https://github.com/mlbio-epfl/turtle"},{"title":"rinnakk/japanese-clip","url":"https://github.com/rinnakk/japanese-clip"},{"title":"salesforce/pb-ovd","url":"https://github.com/salesforce/pb-ovd"},{"title":"facebookresearch/clip-rocket","url":"https://github.com/facebookresearch/clip-rocket"},{"title":"sincerass/mvlpt","url":"https://github.com/sincerass/mvlpt"},{"title":"ericyinyzy/vlattack","url":"https://github.com/ericyinyzy/vlattack"},{"title":"redcaps-dataset/redcaps-downloader","url":"https://github.com/redcaps-dataset/redcaps-downloader"},{"title":"bespontaneous/proteus-pytorch","url":"https://github.com/bespontaneous/proteus-pytorch"},{"title":"shunk031/simple-aesthetics-predictor","url":"https://github.com/shunk031/simple-aesthetics-predictor"},{"title":"giantseaweed/decree","url":"https://github.com/giantseaweed/decree"},{"title":"sithu31296/simple-object-tracking","url":"https://github.com/sithu31296/simple-object-tracking"},{"title":"Gahyeonkim09/AAPL","url":"https://github.com/Gahyeonkim09/AAPL"},{"title":"filipbasara0/simple-clip","url":"https://github.com/filipbasara0/simple-clip"},{"title":"michi-3000/eyeclip","url":"https://github.com/michi-3000/eyeclip"},{"title":"baskargroup/Arboretum","url":"https://github.com/baskargroup/Arboretum"},{"title":"baskargroup/biotrove","url":"https://github.com/baskargroup/biotrove"},{"title":"kynkaat/role-of-imagenet-classes-in-fid","url":"https://github.com/kynkaat/role-of-imagenet-classes-in-fid"},{"title":"SforAiDl/CountCLIP","url":"https://github.com/SforAiDl/CountCLIP"},{"title":"mainaksingha01/applenet","url":"https://github.com/mainaksingha01/applenet"},{"title":"zhangxu0963/npc","url":"https://github.com/zhangxu0963/npc"},{"title":"mainaksingha01/odg-clip","url":"https://github.com/mainaksingha01/odg-clip"},{"title":"jhaprince/multibully","url":"https://github.com/jhaprince/multibully"},{"title":"klemens-floege/oneprot","url":"https://github.com/klemens-floege/oneprot"},{"title":"leolee99/CLIP_ITM","url":"https://github.com/leolee99/CLIP_ITM"},{"title":"madrylab/pretraining-distribution-shift-robustness","url":"https://github.com/madrylab/pretraining-distribution-shift-robustness"},{"title":"AndresPMD/Clip_CMR","url":"https://github.com/AndresPMD/Clip_CMR"},{"title":"fastscience-ai/medflamingo","url":"https://github.com/fastscience-ai/medflamingo"},{"title":"buyeah1109/KEN","url":"https://github.com/buyeah1109/KEN"},{"title":"pseulki/rococo","url":"https://github.com/pseulki/rococo"},{"title":"shkarupa-alex/tfclip","url":"https://github.com/shkarupa-alex/tfclip"},{"title":"IMvision12/keras-vision-models","url":"https://github.com/IMvision12/keras-vision-models"},{"title":"brown-palm/ObjectPrompt","url":"https://github.com/brown-palm/ObjectPrompt"},{"title":"YvanG/VQGAN-CLIP","url":"https://github.com/YvanG/VQGAN-CLIP"},{"title":"ramanakshay/clip","url":"https://github.com/ramanakshay/clip"},{"title":"NYU-DICE-Lab/open_clip","url":"https://github.com/NYU-DICE-Lab/open_clip"},{"title":"shivammehta25/clip","url":"https://github.com/shivammehta25/clip"},{"title":"minhanh151/respro","url":"https://github.com/minhanh151/respro"},{"title":"nopperl/clip_arxiv_pmc","url":"https://github.com/nopperl/clip_arxiv_pmc"},{"title":"s-a-malik/multi-few","url":"https://github.com/s-a-malik/multi-few"},{"title":"prabhupad26/100daysofML","url":"https://github.com/prabhupad26/100daysofML"},{"title":"armaank/archlectures","url":"https://github.com/armaank/archlectures"},{"title":"minhanh151/pre","url":"https://github.com/minhanh151/pre"},{"title":"yuuun/clip_pytorch","url":"https://github.com/yuuun/clip_pytorch"},{"title":"2024-MindSpore-1/Code2","url":"https://github.com/2024-MindSpore-1/Code2/tree/main/model-1/clip"},{"title":"lunaproject22/rpa","url":"https://github.com/lunaproject22/rpa"},{"title":"fiabdu/Commonly-Interesting-Images","url":"https://github.com/fiabdu/Commonly-Interesting-Images"},{"title":"iejMac/ScriptWriter","url":"https://github.com/iejMac/ScriptWriter"},{"title":"ZackPashkin/text2cartoon-pytorch-CLIP","url":"https://github.com/ZackPashkin/text2cartoon-pytorch-CLIP"},{"title":"bruthyu/bpt-vlm","url":"https://github.com/bruthyu/bpt-vlm"},{"title":"buyeah1109/finc","url":"https://github.com/buyeah1109/finc"},{"title":"pwc-1/Paper-8","url":"https://github.com/pwc-1/Paper-8/tree/main/clip"},{"title":"eify/open_clip","url":"https://github.com/eify/open_clip"},{"title":"2023-MindSpore-4/Code12","url":"https://github.com/2023-MindSpore-4/Code12/tree/main/MindFormers/clip"},{"title":"a736875071/clip-vit-large-patch14","url":"https://github.com/a736875071/clip-vit-large-patch14"},{"title":"nahidalam/open_clip","url":"https://github.com/nahidalam/open_clip"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/probability-guided-loss-for-long-tailed-multi","title":"Probability Guided Loss for Long-Tailed Multi-Label Image Classification","date":"2023-06-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/lmpt-prompt-tuning-with-class-specific","title":"LMPT: Prompt Tuning with Class-Specific Embedding Loss for Long-tailed Multi-Label Visual Recognition","date":"2023-05-08","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/long-tailed-multi-label-visual-recognition-by","title":"Long-Tailed Multi-Label Visual Recognition by Collaborative Training on Uniform and Re-Balanced Samplings","date":"2021-06-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-transferable-visual-models-from","title":"Learning Transferable Visual Models From Natural Language Supervision","date":"2021-02-26","rows_on_this_dataset":4,"code_links":82,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":20,"samples_ran":16,"samples_unverified":4,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/distribution-balanced-loss-for-multi-label","title":"Distribution-Balanced Loss for Multi-Label Classification in Long-Tailed Datasets","date":"2020-07-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-imbalanced-datasets-with-label","title":"Learning Imbalanced Datasets with Label-Distribution-Aware Margin Loss","date":"2019-06-18","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":2,"samples_unverified":9,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/large-scale-long-tailed-recognition-in-an","title":"Large-Scale Long-Tailed Recognition in an Open World","date":"2019-04-10","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/multi-label-image-recognition-with-graph","title":"Multi-Label Image Recognition with Graph Convolutional Networks","date":"2019-04-07","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/class-balanced-loss-based-on-effective-number","title":"Class-Balanced Loss Based on Effective Number of Samples","date":"2019-01-16","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":27,"samples_ran":7,"samples_unverified":20,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/focal-loss-for-dense-object-detection","title":"Focal Loss for Dense Object Detection","date":"2017-08-07","rows_on_this_dataset":1,"code_links":234,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":11,"samples_unverified":0,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/relay-backpropagation-for-effective-learning","title":"Relay Backpropagation for Effective Learning of Deep Convolutional Neural Networks","date":"2015-12-18","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":78,"samples_ran":36,"samples_unverified":42,"pointer_only_for_licence":25,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}