{"url":"/sota/object-recognition-on-shape-bias","task":{"name":"Object Recognition","url":"/task/object-recognition","note":null},"dataset":{"name":"shape bias","url":"/dataset/shape-bias"},"category":"Computer Vision","categories":["Computer Vision"],"category_note":null,"description":"Object recognition is a computer vision technique for detecting + classifying objects in images or videos. Since this is a combined task of object detection plus image classification, the state-of-the-art tables are recorded for each component task [here](https://www.paperswithcode.com/task/object-detection) and  [here](https://www.paperswithcode.com/task/image-classification2).\r\n\r\n<span style=\"color:grey; opacity: 0.6\">( Image credit: [Tensorflow Object Detection API\r\n](https://github.com/tensorflow/models/tree/master/research/object_detection) )</span>","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["shape bias"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"shape bias":null}},"counts":{"rows":18,"rows_with_code":17,"rows_with_paper_page":18,"rows_dated":18,"rows_using_additional_data":0},"rows":[{"rank_in_archive_order":1,"model":"Imagen","metrics":{"shape bias":"98.7"},"uses_additional_data":false,"paper_date":"2023-09-28","paper":"/paper/intriguing-properties-of-generative","paper_url":"https://arxiv.org/abs/2309.16779v2","paper_title":"Intriguing properties of generative classifiers","code":"https://github.com/SamsungSAILMontreal/ForestDiffusion","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":2,"n_samples":12,"n_pointer_only_licence":12}},{"rank_in_archive_order":2,"model":"Stable Diffusion","metrics":{"shape bias":"92.7"},"uses_additional_data":false,"paper_date":"2023-09-28","paper":"/paper/intriguing-properties-of-generative","paper_url":"https://arxiv.org/abs/2309.16779v2","paper_title":"Intriguing properties of generative classifiers","code":"https://github.com/SamsungSAILMontreal/ForestDiffusion","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":2,"n_samples":12,"n_pointer_only_licence":12}},{"rank_in_archive_order":3,"model":"Parti","metrics":{"shape bias":"91.7"},"uses_additional_data":false,"paper_date":"2023-09-28","paper":"/paper/intriguing-properties-of-generative","paper_url":"https://arxiv.org/abs/2309.16779v2","paper_title":"Intriguing properties of generative classifiers","code":"https://github.com/SamsungSAILMontreal/ForestDiffusion","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":2,"n_samples":12,"n_pointer_only_licence":12}},{"rank_in_archive_order":4,"model":"ViT-22B-384","metrics":{"shape bias":"86.4"},"uses_additional_data":false,"paper_date":"2023-02-10","paper":"/paper/scaling-vision-transformers-to-22-billion","paper_url":"https://arxiv.org/abs/2302.05442v1","paper_title":"Scaling Vision Transformers to 22 Billion Parameters","code":"https://github.com/lucidrains/flash-cosine-sim-attention","n_code_links":1,"syntology":null},{"rank_in_archive_order":5,"model":"ViT-22B-560","metrics":{"shape bias":"83.8"},"uses_additional_data":false,"paper_date":"2023-02-10","paper":"/paper/scaling-vision-transformers-to-22-billion","paper_url":"https://arxiv.org/abs/2302.05442v1","paper_title":"Scaling Vision Transformers to 22 Billion Parameters","code":"https://github.com/lucidrains/flash-cosine-sim-attention","n_code_links":1,"syntology":null},{"rank_in_archive_order":6,"model":"CLIP (ViT-B)","metrics":{"shape bias":"79.9"},"uses_additional_data":false,"paper_date":"2021-02-26","paper":"/paper/learning-transferable-visual-models-from","paper_url":"https://arxiv.org/abs/2103.00020v1","paper_title":"Learning Transferable Visual Models From Natural Language Supervision","code":"https://github.com/openai/CLIP","n_code_links":82,"syntology":{"n_ran":16,"n_unverified":4,"n_samples":20,"n_pointer_only_licence":16}},{"rank_in_archive_order":7,"model":"ViT-22B-224","metrics":{"shape bias":"78.0"},"uses_additional_data":false,"paper_date":"2023-02-10","paper":"/paper/scaling-vision-transformers-to-22-billion","paper_url":"https://arxiv.org/abs/2302.05442v1","paper_title":"Scaling Vision Transformers to 22 Billion Parameters","code":"https://github.com/lucidrains/flash-cosine-sim-attention","n_code_links":1,"syntology":null},{"rank_in_archive_order":8,"model":"ResNet-50 (L2 eps 5.0 adv trained)","metrics":{"shape bias":"69.5"},"uses_additional_data":false,"paper_date":"2020-07-16","paper":"/paper/do-adversarially-robust-imagenet-models","paper_url":"https://arxiv.org/abs/2007.08489v2","paper_title":"Do Adversarially Robust ImageNet Models Transfer Better?","code":"https://github.com/MadryLab/robustness","n_code_links":2,"syntology":null},{"rank_in_archive_order":9,"model":"ResNet-50 (with strong augmentations)","metrics":{"shape bias":"62.2"},"uses_additional_data":false,"paper_date":"2019-11-20","paper":"/paper/exploring-the-origins-and-prevalence-of","paper_url":"https://arxiv.org/abs/1911.09071v3","paper_title":"The Origins and Prevalence of Texture Bias in Convolutional Neural Networks","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":10,"model":"SWSL (ResNeXt-101)","metrics":{"shape bias":"49.8"},"uses_additional_data":false,"paper_date":"2019-05-02","paper":"/paper/billion-scale-semi-supervised-learning-for","paper_url":"http://arxiv.org/abs/1905.00546v1","paper_title":"Billion-scale semi-supervised learning for image classification","code":"https://github.com/facebookresearch/semi-supervised-ImageNet1K-models","n_code_links":4,"syntology":null},{"rank_in_archive_order":11,"model":"AlexNet","metrics":{"shape bias":"42.9"},"uses_additional_data":false,"paper_date":"2018-11-29","paper":"/paper/imagenet-trained-cnns-are-biased-towards","paper_url":"https://arxiv.org/abs/1811.12231v3","paper_title":"ImageNet-trained CNNs are biased towards texture; increasing shape bias improves accuracy and robustness","code":"https://github.com/rgeirhos/texture-vs-shape","n_code_links":7,"syntology":{"n_ran":0,"n_unverified":6,"n_samples":6,"n_pointer_only_licence":0}},{"rank_in_archive_order":12,"model":"SimCLR (ResNet-50x2)","metrics":{"shape bias":"41.7"},"uses_additional_data":false,"paper_date":"2020-02-13","paper":"/paper/a-simple-framework-for-contrastive-learning","paper_url":"https://arxiv.org/abs/2002.05709v3","paper_title":"A Simple Framework for Contrastive Learning of Visual Representations","code":"https://github.com/tensorflow/models/tree/master/official/vision/beta/projects/simclr","n_code_links":96,"syntology":{"n_ran":79,"n_unverified":58,"n_samples":137,"n_pointer_only_licence":52}},{"rank_in_archive_order":13,"model":"SimCLR (ResNet-50x4)","metrics":{"shape bias":"40.7"},"uses_additional_data":false,"paper_date":"2020-02-13","paper":"/paper/a-simple-framework-for-contrastive-learning","paper_url":"https://arxiv.org/abs/2002.05709v3","paper_title":"A Simple Framework for Contrastive Learning of Visual Representations","code":"https://github.com/tensorflow/models/tree/master/official/vision/beta/projects/simclr","n_code_links":96,"syntology":{"n_ran":79,"n_unverified":58,"n_samples":137,"n_pointer_only_licence":52}},{"rank_in_archive_order":14,"model":"SimCLR (ResNet-50x1)","metrics":{"shape bias":"38.9"},"uses_additional_data":false,"paper_date":"2020-02-13","paper":"/paper/a-simple-framework-for-contrastive-learning","paper_url":"https://arxiv.org/abs/2002.05709v3","paper_title":"A Simple Framework for Contrastive Learning of Visual Representations","code":"https://github.com/tensorflow/models/tree/master/official/vision/beta/projects/simclr","n_code_links":96,"syntology":{"n_ran":79,"n_unverified":58,"n_samples":137,"n_pointer_only_licence":52}},{"rank_in_archive_order":15,"model":"GoogLeNet","metrics":{"shape bias":"31.2"},"uses_additional_data":false,"paper_date":"2018-11-29","paper":"/paper/imagenet-trained-cnns-are-biased-towards","paper_url":"https://arxiv.org/abs/1811.12231v3","paper_title":"ImageNet-trained CNNs are biased towards texture; increasing shape bias improves accuracy and robustness","code":"https://github.com/rgeirhos/texture-vs-shape","n_code_links":7,"syntology":{"n_ran":0,"n_unverified":6,"n_samples":6,"n_pointer_only_licence":0}},{"rank_in_archive_order":16,"model":"SWSL (ResNet-50)","metrics":{"shape bias":"28.6"},"uses_additional_data":false,"paper_date":"2019-05-02","paper":"/paper/billion-scale-semi-supervised-learning-for","paper_url":"http://arxiv.org/abs/1905.00546v1","paper_title":"Billion-scale semi-supervised learning for image classification","code":"https://github.com/facebookresearch/semi-supervised-ImageNet1K-models","n_code_links":4,"syntology":null},{"rank_in_archive_order":17,"model":"ResNet-50","metrics":{"shape bias":"22.1"},"uses_additional_data":false,"paper_date":"2018-11-29","paper":"/paper/imagenet-trained-cnns-are-biased-towards","paper_url":"https://arxiv.org/abs/1811.12231v3","paper_title":"ImageNet-trained CNNs are biased towards texture; increasing shape bias improves accuracy and robustness","code":"https://github.com/rgeirhos/texture-vs-shape","n_code_links":7,"syntology":{"n_ran":0,"n_unverified":6,"n_samples":6,"n_pointer_only_licence":0}},{"rank_in_archive_order":18,"model":"VGG-16","metrics":{"shape bias":"17.2"},"uses_additional_data":false,"paper_date":"2018-11-29","paper":"/paper/imagenet-trained-cnns-are-biased-towards","paper_url":"https://arxiv.org/abs/1811.12231v3","paper_title":"ImageNet-trained CNNs are biased towards texture; increasing shape bias improves accuracy and robustness","code":"https://github.com/rgeirhos/texture-vs-shape","n_code_links":7,"syntology":{"n_ran":0,"n_unverified":6,"n_samples":6,"n_pointer_only_licence":0}}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":11,"rows_with_any_sample_ran":7,"distinct_papers_with_graph_line":4,"distinct_papers_with_any_sample_ran":3,"samples_over_distinct_papers":{"n_ran":105,"n_unverified":70,"n_samples":175,"n_pointer_only_licence":80,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":283,"n_unverified":208,"n_samples":491,"n_pointer_only_licence":208,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}