{"url":"/dataset/objectnet","name":"ObjectNet","full_name":null,"description_markdown":"**ObjectNet** is a test set of images collected directly using crowd-sourcing. ObjectNet is unique as the objects are captured at unusual poses in cluttered, natural scenes, which can severely degrade recognition performance. There are 50,000 images in the test set which controls for rotation, background and viewpoint. There are 313 object classes with 113 overlapping ImageNet.\r\n\r\nSource: [On Robustness and Transferability of Convolutional Neural Networks](https://arxiv.org/abs/2007.08558)\r\nImage Source: [https://objectnet.dev/](https://objectnet.dev/)","description_withheld":null,"homepage":"https://objectnet.dev/","introduced_date":"2019-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/objectnet-a-large-scale-bias-controlled","title":"ObjectNet: A large-scale bias-controlled dataset for pushing the limits of object recognition models","first_author":"Andrei Barbu","url":null},"license":{"name":"Creative Commons Attribution 4.0","url":"https://objectnet.dev/download.html"},"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Image Classification","url":"/task/image-classification","datasets_with_task":"/datasets/task/image-classification"},{"name":"Zero-Shot Transfer Image Classification","url":"/task/zero-shot-transfer-image-classification","datasets_with_task":"/datasets/task/zero-shot-transfer-image-classification"},{"name":"Unsupervised Image Classification","url":"/task/unsupervised-image-classification","datasets_with_task":"/datasets/task/unsupervised-image-classification"}],"languages":[],"variants":["ObjectNet","ObjectNet (Bounding Box)"],"data_loaders":[],"num_papers_in_archive":155,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/image-classification-on-objectnet","task":"Image Classification","dataset_variant":"ObjectNet","rows":106,"metrics":["Top-1 Accuracy","Top-5 Accuracy"],"first_row_in_archive_order":{"model":"CoCa","paper":"/paper/coca-contrastive-captioners-are-image-text","metrics":{"Top-1 Accuracy":"82.7"},"code_links":[{"title":"mlfoundations/open_clip","url":"https://github.com/mlfoundations/open_clip"},{"title":"facebookresearch/multimodal","url":"https://github.com/facebookresearch/multimodal"},{"title":"lucidrains/CoCa-pytorch","url":"https://github.com/lucidrains/CoCa-pytorch"},{"title":"amitakamath/whatsup_vlms","url":"https://github.com/amitakamath/whatsup_vlms"},{"title":"amitakamath/hard_positives","url":"https://github.com/amitakamath/hard_positives"},{"title":"Chaolei98/FreeZAD","url":"https://github.com/Chaolei98/FreeZAD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-transfer-image-classification-on-6","task":"Zero-Shot Transfer Image Classification","dataset_variant":"ObjectNet","rows":9,"metrics":["Accuracy (Private)","Accuracy (Public)","Top 5 Accuracy"],"first_row_in_archive_order":{"model":"LiT-22B","paper":"/paper/scaling-vision-transformers-to-22-billion","metrics":{"Accuracy (Private)":"87.6"},"code_links":[{"title":"lucidrains/flash-cosine-sim-attention","url":"https://github.com/lucidrains/flash-cosine-sim-attention"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-classification-on-objectnet-bounding","task":"Image Classification","dataset_variant":"ObjectNet (Bounding Box)","rows":4,"metrics":["Top 5 Accuracy"],"first_row_in_archive_order":{"model":"BiT-L (ResNet)","paper":"/paper/large-scale-learning-of-general-visual","metrics":{"Top 5 Accuracy":"85.1"},"code_links":[{"title":"google-research/big_transfer","url":"https://github.com/google-research/big_transfer"},{"title":"sayakpaul/FunMatch-Distillation","url":"https://github.com/sayakpaul/FunMatch-Distillation"},{"title":"bethgelab/InDomainGeneralizationBenchmark","url":"https://github.com/bethgelab/InDomainGeneralizationBenchmark"},{"title":"SoojungYang/supervised_pretraining_GN_WS","url":"https://github.com/SoojungYang/supervised_pretraining_GN_WS"},{"title":"sayakpaul/A-Barebones-Image-Retrieval-System","url":"https://github.com/sayakpaul/A-Barebones-Image-Retrieval-System"},{"title":"batsresearch/taglets","url":"https://github.com/batsresearch/taglets"},{"title":"MS-Mind/MS-Code-02","url":"https://github.com/MS-Mind/MS-Code-02/tree/main/configs/bit"},{"title":"2024-MindSpore-1/Code2","url":"https://github.com/2024-MindSpore-1/Code2/tree/main/model-1/bit"},{"title":"hw666666666666/BigTransfer","url":"https://github.com/hw666666666666/BigTransfer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-image-classification-on","task":"Unsupervised Image Classification","dataset_variant":"ObjectNet","rows":2,"metrics":["ARI","Accuracy (%)"],"first_row_in_archive_order":{"model":"InfoMin ResNeXt-152 + SK (PCA+k-means)","paper":"/paper/self-supervised-learning-for-large-scale","metrics":{"ARI":"1.59±0.04","Accuracy (%)":"6.53±0.19"},"code_links":[{"title":"Randl/kmeans_selfsuper","url":"https://github.com/Randl/kmeans_selfsuper"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/eva-clip-18b-scaling-clip-to-18-billion","title":"EVA-CLIP-18B: Scaling CLIP to 18 Billion Parameters","date":"2024-02-06","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":1,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/internvl-scaling-up-vision-foundation-models","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","date":"2023-12-21","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/eva-clip-improved-training-techniques-for","title":"EVA-CLIP: Improved Training Techniques for CLIP at Scale","date":"2023-03-27","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/the-effectiveness-of-mae-pre-pretraining-for","title":"The effectiveness of MAE pre-pretraining for billion-scale pretraining","date":"2023-03-23","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/scaling-vision-transformers-to-22-billion","title":"Scaling Vision Transformers to 22 Billion Parameters","date":"2023-02-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-whac-a-mole-dilemma-shortcuts-come-in","title":"A Whac-A-Mole Dilemma: Shortcuts Come in Multiples Where Mitigating One Amplifies Others","date":"2022-12-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pali-a-jointly-scaled-multilingual-language","title":"PaLI: A Jointly-Scaled Multilingual Language-Image Model","date":"2022-09-14","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/optimizing-relevance-maps-of-vision","title":"Optimizing Relevance Maps of Vision Transformers Improves Robustness","date":"2022-06-02","rows_on_this_dataset":14,"code_links":1,"syntology":null},{"paper":"/paper/matryoshka-representations-for-adaptive","title":"Matryoshka Representation Learning","date":"2022-05-26","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/coca-contrastive-captioners-are-image-text","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","date":"2022-05-04","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":9,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/data-determines-distributional-robustness-in","title":"Data Determines Distributional Robustness in Contrastive Language Image Pre-training (CLIP)","date":"2022-05-03","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/robust-cross-modal-representation-learning","title":"Robust Cross-Modal Representation Learning with Progressive Self-Distillation","date":"2022-04-10","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dilemma-self-supervised-shape-and-texture","title":"Representation Learning by Detecting Incorrect Location Embeddings","date":"2022-04-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bamboo-building-mega-scale-vision-dataset","title":"Bamboo: Building Mega-Scale Vision Dataset Continually with Human-Machine Synergy","date":"2022-03-15","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/model-soups-averaging-weights-of-multiple","title":"Model soups: averaging weights of multiple fine-tuned models improves accuracy without increasing inference time","date":"2022-03-10","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":5,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vision-models-are-more-robust-and-fair-when","title":"Vision Models Are More Robust And Fair When Pretrained On Uncurated Images Without Supervision","date":"2022-02-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/revisiting-weakly-supervised-pre-training-of","title":"Revisiting Weakly Supervised Pre-Training of Visual Perception Models","date":"2022-01-20","rows_on_this_dataset":5,"code_links":2,"syntology":null},{"paper":"/paper/pushing-the-limits-of-self-supervised-resnets","title":"Pushing the limits of self-supervised ResNets: Can we outperform supervised learning without labels on ImageNet?","date":"2022-01-13","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":0,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/optimal-representations-for-covariate-shift-1","title":"Optimal Representations for Covariate Shift","date":"2021-12-31","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pyramid-adversarial-training-improves-vit","title":"Pyramid Adversarial Training Improves ViT Performance","date":"2021-11-30","rows_on_this_dataset":22,"code_links":1,"syntology":null},{"paper":"/paper/discrete-representations-strengthen-vision-1","title":"Discrete Representations Strengthen Vision Transformer Robustness","date":"2021-11-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/combined-scaling-for-zero-shot-transfer","title":"Combined Scaling for Zero-shot Transfer Learning","date":"2021-11-19","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/lit-zero-shot-transfer-with-locked-image-text","title":"LiT: Zero-Shot Transfer with Locked-image text Tuning","date":"2021-11-15","rows_on_this_dataset":2,"code_links":5,"syntology":null},{"paper":"/paper/measuring-the-interpretability-of","title":"Measuring the Interpretability of Unsupervised Representations via Quantized Reversed Probing","date":"2021-09-29","rows_on_this_dataset":6,"code_links":0,"syntology":null},{"paper":"/paper/compressive-visual-representations","title":"Compressive Visual Representations","date":"2021-09-27","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":0,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/robust-fine-tuning-of-zero-shot-models","title":"Robust fine-tuning of zero-shot models","date":"2021-09-04","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/billion-scale-pretraining-with-vision","title":"Billion-Scale Pretraining with Vision Transformers for Multi-Task Visual Representations","date":"2021-08-12","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/recurrent-parameter-generators","title":"Compact and Optimal Deep Learning with Recurrent Parameter Generators","date":"2021-07-15","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scaling-vision-transformers","title":"Scaling Vision Transformers","date":"2021-06-08","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/leveraging-background-augmentations-to","title":"Characterizing and Improving the Robustness of Self-Supervised Learning through Background Augmentations","date":"2021-03-23","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/learning-transferable-visual-models-from","title":"Learning Transferable Visual Models From Natural Language Supervision","date":"2021-02-26","rows_on_this_dataset":2,"code_links":82,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":20,"samples_ran":16,"samples_unverified":4,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/generative-interventions-for-causal-learning","title":"Generative Interventions for Causal Learning","date":"2020-12-22","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/class-agnostic-object-detection","title":"Class-agnostic Object Detection","date":"2020-11-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","rows_on_this_dataset":1,"code_links":158,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":419,"samples_ran":281,"samples_unverified":138,"pointer_only_for_licence":154,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-supervised-learning-for-large-scale","title":"Self-Supervised Learning for Large-Scale Unsupervised Image Clustering","date":"2020-08-24","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":2,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-robustness-against-common","title":"Improving robustness against common corruptions by covariate shift adaptation","date":"2020-06-30","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/on-mixup-regularization","title":"On Mixup Regularization","date":"2020-06-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/objectnet-dataset-reanalysis-and-correction","title":"ObjectNet Dataset: Reanalysis and Correction","date":"2020-04-04","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/large-scale-learning-of-general-visual","title":"Big Transfer (BiT): General Visual Representation Learning","date":"2019-12-24","rows_on_this_dataset":6,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":3,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/objectnet-a-large-scale-bias-controlled","title":"ObjectNet: A large-scale bias-controlled dataset for pushing the limits of object recognition models","date":"2019-12-01","rows_on_this_dataset":6,"code_links":0,"syntology":null},{"paper":"/paper/context-gated-convolution","title":"Context-Gated Convolution","date":"2019-10-12","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":19,"samples_harvested":558,"samples_ran":337,"samples_unverified":221,"pointer_only_for_licence":177,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}