{"url":"/dataset/coco-mlt","name":"COCO-MLT","full_name":null,"description_markdown":"The COCO-MLT is created from MS COCO-2017, containing 1,909 images from 80 classes. The maximum of training number per class is 1,128 and the minimum is 6. We use the test set of COCO2017 with 5,000 for evaluation. The ratio of head, medium, and tail classes is 22:33:25 in COCO-MLT.","description_withheld":null,"homepage":"","introduced_date":"2020-07-19","introduced_date_note":null,"introduced_by":{"paper":"/paper/distribution-balanced-loss-for-multi-label","title":"Distribution-Balanced Loss for Multi-Label Classification in Long-Tailed Datasets","first_author":"Tong Wu","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Zero-Shot Learning","url":"/task/zero-shot-learning","datasets_with_task":"/datasets/task/zero-shot-learning"},{"name":"Long-tail Learning","url":"/task/long-tail-learning","datasets_with_task":"/datasets/task/long-tail-learning"},{"name":"Multi-Label Image Classification","url":"/task/multi-label-image-classification","datasets_with_task":"/datasets/task/multi-label-image-classification"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["COCO-MLT"],"data_loaders":[],"num_papers_in_archive":12,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/long-tail-learning-on-coco-mlt","task":"Long-tail Learning","dataset_variant":"COCO-MLT","rows":13,"metrics":["Average mAP"],"first_row_in_archive_order":{"model":"LMPT(ViT-B/16)","paper":"/paper/lmpt-prompt-tuning-with-class-specific","metrics":{"Average mAP":"66.19"},"code_links":[{"title":"richard-peng-xia/LMPT","url":"https://github.com/richard-peng-xia/LMPT"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-learning-on-coco-mlt","task":"Zero-Shot Learning","dataset_variant":"COCO-MLT","rows":2,"metrics":["Average mAP"],"first_row_in_archive_order":{"model":"ResNet-50","paper":"/paper/learning-transferable-visual-models-from","metrics":{"Average mAP":"56.19"},"code_links":[{"title":"openai/CLIP","url":"https://github.com/openai/CLIP"},{"title":"mlfoundations/open_clip","url":"https://github.com/mlfoundations/open_clip"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"facebookresearch/vissl","url":"https://github.com/facebookresearch/vissl"},{"title":"alibaba/EasyNLP","url":"https://github.com/alibaba/EasyNLP"},{"title":"apple/ml-mobileclip","url":"https://github.com/apple/ml-mobileclip"},{"title":"OML-Team/open-metric-learning","url":"https://github.com/OML-Team/open-metric-learning"},{"title":"FreddeFrallan/Multilingual-CLIP","url":"https://github.com/FreddeFrallan/Multilingual-CLIP"},{"title":"eps696/aphantasia","url":"https://github.com/eps696/aphantasia"},{"title":"muzairkhattak/multimodal-prompt-learning","url":"https://github.com/muzairkhattak/multimodal-prompt-learning"},{"title":"moein-shariatnia/OpenAI-CLIP","url":"https://github.com/moein-shariatnia/OpenAI-CLIP"},{"title":"facebookresearch/brainmagick","url":"https://github.com/facebookresearch/brainmagick"},{"title":"PaddlePaddle/PASSL","url":"https://github.com/PaddlePaddle/PASSL/blob/main/docs/Train_CLIP_model.md"},{"title":"taited/clip-score","url":"https://github.com/taited/clip-score"},{"title":"azshue/TPT","url":"https://github.com/azshue/TPT"},{"title":"clip-italian/clip-italian","url":"https://github.com/clip-italian/clip-italian"},{"title":"dhansmair/flamingo-mini","url":"https://github.com/dhansmair/flamingo-mini"},{"title":"ylqi/count-anything","url":"https://github.com/ylqi/count-anything"},{"title":"ml-jku/cloob","url":"https://github.com/ml-jku/cloob"},{"title":"sberbank-ai/ru-clip","url":"https://github.com/sberbank-ai/ru-clip"},{"title":"ai-forever/ru-clip","url":"https://github.com/ai-forever/ru-clip"},{"title":"Kaushalya/medclip","url":"https://github.com/Kaushalya/medclip"},{"title":"ajayjain/vectorascent","url":"https://github.com/ajayjain/vectorascent"},{"title":"linjieli222/hero_video_feature_extractor","url":"https://github.com/linjieli222/hero_video_feature_extractor"},{"title":"borisdayma/clip-jax","url":"https://github.com/borisdayma/clip-jax"},{"title":"sajjjadayobi/CLIPfa","url":"https://github.com/sajjjadayobi/CLIPfa"},{"title":"mertyg/post-hoc-cbm","url":"https://github.com/mertyg/post-hoc-cbm"},{"title":"mlbio-epfl/turtle","url":"https://github.com/mlbio-epfl/turtle"},{"title":"rinnakk/japanese-clip","url":"https://github.com/rinnakk/japanese-clip"},{"title":"salesforce/pb-ovd","url":"https://github.com/salesforce/pb-ovd"},{"title":"facebookresearch/clip-rocket","url":"https://github.com/facebookresearch/clip-rocket"},{"title":"sincerass/mvlpt","url":"https://github.com/sincerass/mvlpt"},{"title":"ericyinyzy/vlattack","url":"https://github.com/ericyinyzy/vlattack"},{"title":"redcaps-dataset/redcaps-downloader","url":"https://github.com/redcaps-dataset/redcaps-downloader"},{"title":"bespontaneous/proteus-pytorch","url":"https://github.com/bespontaneous/proteus-pytorch"},{"title":"shunk031/simple-aesthetics-predictor","url":"https://github.com/shunk031/simple-aesthetics-predictor"},{"title":"giantseaweed/decree","url":"https://github.com/giantseaweed/decree"},{"title":"sithu31296/simple-object-tracking","url":"https://github.com/sithu31296/simple-object-tracking"},{"title":"Gahyeonkim09/AAPL","url":"https://github.com/Gahyeonkim09/AAPL"},{"title":"filipbasara0/simple-clip","url":"https://github.com/filipbasara0/simple-clip"},{"title":"michi-3000/eyeclip","url":"https://github.com/michi-3000/eyeclip"},{"title":"baskargroup/Arboretum","url":"https://github.com/baskargroup/Arboretum"},{"title":"baskargroup/biotrove","url":"https://github.com/baskargroup/biotrove"},{"title":"kynkaat/role-of-imagenet-classes-in-fid","url":"https://github.com/kynkaat/role-of-imagenet-classes-in-fid"},{"title":"SforAiDl/CountCLIP","url":"https://github.com/SforAiDl/CountCLIP"},{"title":"mainaksingha01/applenet","url":"https://github.com/mainaksingha01/applenet"},{"title":"zhangxu0963/npc","url":"https://github.com/zhangxu0963/npc"},{"title":"mainaksingha01/odg-clip","url":"https://github.com/mainaksingha01/odg-clip"},{"title":"jhaprince/multibully","url":"https://github.com/jhaprince/multibully"},{"title":"klemens-floege/oneprot","url":"https://github.com/klemens-floege/oneprot"},{"title":"leolee99/CLIP_ITM","url":"https://github.com/leolee99/CLIP_ITM"},{"title":"madrylab/pretraining-distribution-shift-robustness","url":"https://github.com/madrylab/pretraining-distribution-shift-robustness"},{"title":"AndresPMD/Clip_CMR","url":"https://github.com/AndresPMD/Clip_CMR"},{"title":"fastscience-ai/medflamingo","url":"https://github.com/fastscience-ai/medflamingo"},{"title":"buyeah1109/KEN","url":"https://github.com/buyeah1109/KEN"},{"title":"pseulki/rococo","url":"https://github.com/pseulki/rococo"},{"title":"shkarupa-alex/tfclip","url":"https://github.com/shkarupa-alex/tfclip"},{"title":"IMvision12/keras-vision-models","url":"https://github.com/IMvision12/keras-vision-models"},{"title":"brown-palm/ObjectPrompt","url":"https://github.com/brown-palm/ObjectPrompt"},{"title":"YvanG/VQGAN-CLIP","url":"https://github.com/YvanG/VQGAN-CLIP"},{"title":"ramanakshay/clip","url":"https://github.com/ramanakshay/clip"},{"title":"NYU-DICE-Lab/open_clip","url":"https://github.com/NYU-DICE-Lab/open_clip"},{"title":"shivammehta25/clip","url":"https://github.com/shivammehta25/clip"},{"title":"minhanh151/respro","url":"https://github.com/minhanh151/respro"},{"title":"nopperl/clip_arxiv_pmc","url":"https://github.com/nopperl/clip_arxiv_pmc"},{"title":"s-a-malik/multi-few","url":"https://github.com/s-a-malik/multi-few"},{"title":"prabhupad26/100daysofML","url":"https://github.com/prabhupad26/100daysofML"},{"title":"armaank/archlectures","url":"https://github.com/armaank/archlectures"},{"title":"minhanh151/pre","url":"https://github.com/minhanh151/pre"},{"title":"yuuun/clip_pytorch","url":"https://github.com/yuuun/clip_pytorch"},{"title":"2024-MindSpore-1/Code2","url":"https://github.com/2024-MindSpore-1/Code2/tree/main/model-1/clip"},{"title":"lunaproject22/rpa","url":"https://github.com/lunaproject22/rpa"},{"title":"fiabdu/Commonly-Interesting-Images","url":"https://github.com/fiabdu/Commonly-Interesting-Images"},{"title":"iejMac/ScriptWriter","url":"https://github.com/iejMac/ScriptWriter"},{"title":"ZackPashkin/text2cartoon-pytorch-CLIP","url":"https://github.com/ZackPashkin/text2cartoon-pytorch-CLIP"},{"title":"bruthyu/bpt-vlm","url":"https://github.com/bruthyu/bpt-vlm"},{"title":"buyeah1109/finc","url":"https://github.com/buyeah1109/finc"},{"title":"pwc-1/Paper-8","url":"https://github.com/pwc-1/Paper-8/tree/main/clip"},{"title":"eify/open_clip","url":"https://github.com/eify/open_clip"},{"title":"2023-MindSpore-4/Code12","url":"https://github.com/2023-MindSpore-4/Code12/tree/main/MindFormers/clip"},{"title":"a736875071/clip-vit-large-patch14","url":"https://github.com/a736875071/clip-vit-large-patch14"},{"title":"nahidalam/open_clip","url":"https://github.com/nahidalam/open_clip"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/probability-guided-loss-for-long-tailed-multi","title":"Probability Guided Loss for Long-Tailed Multi-Label Image Classification","date":"2023-06-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/lmpt-prompt-tuning-with-class-specific","title":"LMPT: Prompt Tuning with Class-Specific Embedding Loss for Long-tailed Multi-Label Visual Recognition","date":"2023-05-08","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/long-tailed-multi-label-visual-recognition-by","title":"Long-Tailed Multi-Label Visual Recognition by Collaborative Training on Uniform and Re-Balanced Samplings","date":"2021-06-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-transferable-visual-models-from","title":"Learning Transferable Visual Models From Natural Language Supervision","date":"2021-02-26","rows_on_this_dataset":4,"code_links":82,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":20,"samples_ran":16,"samples_unverified":4,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/distribution-balanced-loss-for-multi-label","title":"Distribution-Balanced Loss for Multi-Label Classification in Long-Tailed Datasets","date":"2020-07-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-imbalanced-datasets-with-label","title":"Learning Imbalanced Datasets with Label-Distribution-Aware Margin Loss","date":"2019-06-18","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":2,"samples_unverified":9,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/large-scale-long-tailed-recognition-in-an","title":"Large-Scale Long-Tailed Recognition in an Open World","date":"2019-04-10","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/multi-label-image-recognition-with-graph","title":"Multi-Label Image Recognition with Graph Convolutional Networks","date":"2019-04-07","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/class-balanced-loss-based-on-effective-number","title":"Class-Balanced Loss Based on Effective Number of Samples","date":"2019-01-16","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":27,"samples_ran":7,"samples_unverified":20,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/focal-loss-for-dense-object-detection","title":"Focal Loss for Dense Object Detection","date":"2017-08-07","rows_on_this_dataset":1,"code_links":234,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":11,"samples_unverified":0,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/relay-backpropagation-for-effective-learning","title":"Relay Backpropagation for Effective Learning of Deep Convolutional Neural Networks","date":"2015-12-18","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":78,"samples_ran":36,"samples_unverified":42,"pointer_only_for_licence":25,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}