{"url":"/dataset/sun397","name":"SUN397","full_name":"SUN397","description_markdown":"The Scene UNderstanding (SUN) database contains 899 categories and 130,519 images. There are 397 well-sampled categories to evaluate numerous state-of-the-art algorithms for scene recognition.\r\n\r\nImage Source: [The Selection of Useful Visual Words in Class-Imbalanced Image Classification](https://www.semanticscholar.org/paper/The-Selection-of-Useful-Visual-Words-in-Image-Chimlek-Pramokchon/2b23cff4d2072dfc85cf8b09f54475791690a68d)","description_withheld":null,"homepage":"https://vision.princeton.edu/projects/2010/SUN/","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Image Classification","url":"/task/image-classification","datasets_with_task":"/datasets/task/image-classification"},{"name":"Zero-Shot Learning","url":"/task/zero-shot-learning","datasets_with_task":"/datasets/task/zero-shot-learning"},{"name":"Few-Shot Learning","url":"/task/few-shot-learning","datasets_with_task":"/datasets/task/few-shot-learning"},{"name":"Image Clustering","url":"/task/image-clustering","datasets_with_task":"/datasets/task/image-clustering"},{"name":"Fine-Grained Image Classification","url":"/task/fine-grained-image-classification","datasets_with_task":"/datasets/task/fine-grained-image-classification"},{"name":"Transductive Zero-Shot Classification","url":"/task/transductive-zero-shot-classification","datasets_with_task":"/datasets/task/transductive-zero-shot-classification"},{"name":"Prompt Engineering","url":"/task/prompt-engineering","datasets_with_task":"/datasets/task/prompt-engineering"},{"name":"Few-Shot Learning - 4 shots","url":"/task/few-shot-learning-4-shots","datasets_with_task":"/datasets/task/few-shot-learning-4-shots"},{"name":"Scene Recognition","url":"/task/scene-recognition","datasets_with_task":"/datasets/task/scene-recognition"}],"languages":[],"variants":["SUN397"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/1aurent/SUN397","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/pytorch/vision","url":"https://pytorch.org/vision/stable/generated/torchvision.datasets.SUN397.html","frameworks":["pytorch"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/sun397","frameworks":["tf","jax"]}],"num_papers_in_archive":52,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/prompt-engineering-on-sun397","task":"Prompt Engineering","dataset_variant":"SUN397","rows":14,"metrics":["Harmonic mean"],"first_row_in_archive_order":{"model":"PromptKD","paper":"/paper/promptkd-unsupervised-prompt-distillation-for","metrics":{"Harmonic mean":"82.60"},"code_links":[{"title":"zhengli97/promptkd","url":"https://github.com/zhengli97/promptkd"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/fine-grained-image-classification-on-sun397","task":"Fine-Grained Image Classification","dataset_variant":"SUN397","rows":5,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"µ2Net (ViT-L/16)","paper":"/paper/an-evolutionary-approach-to-dynamic","metrics":{"Accuracy":"84.8"},"code_links":[{"title":"google-research/google-research","url":"https://github.com/google-research/google-research/tree/master/muNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/scene-recognition-on-sun397","task":"Scene Recognition","dataset_variant":"SUN397","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"FOSNet","paper":"/paper/fosnet-an-end-to-end-trainable-deep-neural","metrics":{"Accuracy":"77.28"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-learning-on-sun397","task":"Zero-Shot Learning","dataset_variant":"SUN397","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"ZLaP*","paper":"/paper/label-propagation-for-zero-shot","metrics":{"Accuracy":"71.4"},"code_links":[{"title":"vladan-stojnic/zlap","url":"https://github.com/vladan-stojnic/zlap"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/few-shot-learning-on-sun397","task":"Few-Shot Learning","dataset_variant":"SUN397","rows":1,"metrics":["Harmonic mean"],"first_row_in_archive_order":{"model":"Variational Prompt Tuning","paper":"/paper/variational-prompt-tuning-improves","metrics":{"Harmonic mean":"78.51"},"code_links":[{"title":"saic-fi/bayesian-prompt-learning","url":"https://github.com/saic-fi/bayesian-prompt-learning"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-classification-on-sun397","task":"Image Classification","dataset_variant":"SUN397","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"TransBoost-ResNet50","paper":"/paper/transboost-improving-the-best-imagenet","metrics":{"Accuracy":"95.94%"},"code_links":[{"title":"omerb01/transboost","url":"https://github.com/omerb01/transboost"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-clustering-on-sun397","task":"Image Clustering","dataset_variant":"SUN397","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"TURTLE (CLIP + DINOv2)","paper":"/paper/let-go-of-your-labels-with-unsupervised-1","metrics":{"Accuracy":"67.9"},"code_links":[{"title":"mlbio-epfl/turtle","url":"https://github.com/mlbio-epfl/turtle"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/transductive-zero-shot-classification-on-4","task":"Transductive Zero-Shot Classification","dataset_variant":"SUN397","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"ZLaP","paper":"/paper/label-propagation-for-zero-shot","metrics":{"Accuracy":"71.9"},"code_links":[{"title":"vladan-stojnic/zlap","url":"https://github.com/vladan-stojnic/zlap"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/mmrl-parameter-efficient-and-interaction","title":"MMRL++: Parameter-Efficient and Interaction-Aware Representation Learning for Vision-Language Models","date":"2025-05-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mmrl-multi-modal-representation-learning-for","title":"MMRL: Multi-Modal Representation Learning for Vision-Language Models","date":"2025-03-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hpt-hierarchically-prompting-vision-language","title":"HPT++: Hierarchically Prompting Vision-Language Models with Multi-Granularity Knowledge Generation and Improved Structure Modeling","date":"2024-08-27","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/let-go-of-your-labels-with-unsupervised-1","title":"Let Go of Your Labels with Unsupervised Transfer","date":"2024-06-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/label-propagation-for-zero-shot","title":"Label Propagation for Zero-shot Classification with Vision-Language Models","date":"2024-04-05","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/prompt-learning-via-meta-regularization","title":"Prompt Learning via Meta-Regularization","date":"2024-04-01","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":4,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/promptkd-unsupervised-prompt-distillation-for","title":"PromptKD: Unsupervised Prompt Distillation for Vision-Language Models","date":"2024-03-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-hierarchical-prompt-with-structured","title":"Learning Hierarchical Prompt with Structured Linguistic Knowledge for Vision-Language Models","date":"2023-12-11","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":3,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dept-decoupled-prompt-tuning","title":"DePT: Decoupled Prompt Tuning","date":"2023-09-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":4,"samples_unverified":4,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/read-only-prompt-optimization-for-vision","title":"Read-only Prompt Optimization for Vision-Language Few-shot Learning","date":"2023-08-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-regulating-prompts-foundational-model","title":"Self-regulating Prompts: Foundational Model Adaptation without Forgetting","date":"2023-07-13","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":7,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/consistency-guided-prompt-learning-for-vision","title":"Consistency-guided Prompt Learning for Vision-Language Models","date":"2023-06-01","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":3,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-domain-invariant-prompt-for-vision","title":"Learning Domain Invariant Prompt for Vision-Language Models","date":"2022-12-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/maple-multi-modal-prompt-learning","title":"MaPLe: Multi-modal Prompt Learning","date":"2022-10-06","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/variational-prompt-tuning-improves","title":"Bayesian Prompt Learning for Image-Language Model Generalization","date":"2022-10-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/transboost-improving-the-best-imagenet","title":"TransBoost: Improving the Best ImageNet Performance using Deep Transduction","date":"2022-05-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/an-evolutionary-approach-to-dynamic","title":"An Evolutionary Approach to Dynamic Introduction of Tasks in Large-scale Multitask Learning Systems","date":"2022-05-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bamboo-building-mega-scale-vision-dataset","title":"Bamboo: Building Mega-Scale Vision Dataset Continually with Human-Machine Synergy","date":"2022-03-15","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/conditional-prompt-learning-for-vision","title":"Conditional Prompt Learning for Vision-Language Models","date":"2022-03-10","rows_on_this_dataset":1,"code_links":12,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vision-models-are-more-robust-and-fair-when","title":"Vision Models Are More Robust And Fair When Pretrained On Uncurated Images Without Supervision","date":"2022-02-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/self-supervised-learning-by-estimating-twin-1","title":"Self-Supervised Learning by Estimating Twin Class Distributions","date":"2021-10-14","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":5,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/with-a-little-help-from-my-friends-nearest","title":"With a Little Help from My Friends: Nearest-Neighbor Contrastive Learning of Visual Representations","date":"2021-04-29","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-transferable-visual-models-from","title":"Learning Transferable Visual Models From Natural Language Supervision","date":"2021-02-26","rows_on_this_dataset":1,"code_links":82,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":20,"samples_ran":16,"samples_unverified":4,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/semantic-aware-scene-recognition","title":"Semantic-Aware Scene Recognition","date":"2019-09-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fosnet-an-end-to-end-trainable-deep-neural","title":"FOSNet: An End-to-End Trainable Deep Neural Network for Scene Recognition","date":"2019-07-17","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":17,"samples_harvested":127,"samples_ran":69,"samples_unverified":58,"pointer_only_for_licence":31,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}