{"url":"/task/scene-classification","name":"Scene Classification","slug":"scene-classification","description_markdown":"**Scene Classification** is a task in which scenes from photographs are categorically classified. Unlike object classification, which focuses on classifying prominent objects in the foreground, Scene Classification uses the layout of objects within the scene, in addition to the ambient context, for classification.\n\n\n<span class=\"description-source\">Source: [Scene classification with Convolutional Neural Networks ](http://cs231n.stanford.edu/reports/2017/pdfs/102.pdf)</span>","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":453,"papers_with_code":148,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":23,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/scene-classification-on-uc-merced-land-use","slug":"scene-classification-on-uc-merced-land-use","dataset":"UC Merced Land Use Dataset","dataset_url":"/dataset/uc-merced-land-use-dataset","rows_in_archive":6,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"µ2Net+ (ViT-L/16)","paper_title":"A Continual Development Methodology for Large-scale Multitask Dynamic ML Systems","paper_url":"/paper/a-continual-development-methodology-for-large","paper_date":"2022-09-15","arxiv_id":"2209.07326","code_links":[{"title":"google-research/google-research","url":"https://github.com/google-research/google-research/tree/master/muNet"}],"syntology":null}},{"leaderboard":"/sota/scene-classification-on-places365-standard","slug":"scene-classification-on-places365-standard","dataset":"Places365-Standard","dataset_url":"/dataset/places365","rows_in_archive":2,"metrics":["Top 1 Error","Top 5 Error"],"first_row_in_archive_order":{"model":"WaveMix","paper_title":"WaveMix: A Resource-efficient Neural Network for Image Analysis","paper_url":"/paper/wavemix-lite-a-resource-efficient-neural","paper_date":"2022-05-28","arxiv_id":"2205.14375","code_links":[{"title":"pranavphoenix/WaveMix","url":"https://github.com/pranavphoenix/WaveMix"}],"syntology":null}}],"datasets":[{"url":"/dataset/resisc45","name":"RESISC45","full_name":"RESISC45","num_papers_in_archive":187},{"url":"/dataset/bigearthnet","name":"BigEarthNet","full_name":"","num_papers_in_archive":85},{"url":"/dataset/rsicd","name":"RSICD","full_name":"Remote Sensing Image Captioning Dataset","num_papers_in_archive":70},{"url":"/dataset/places365","name":"Places365","full_name":"","num_papers_in_archive":65},{"url":"/dataset/million-aid","name":"Million-AID","full_name":"","num_papers_in_archive":41},{"url":"/dataset/aid","name":"AID","full_name":"Aerial Image Dataset","num_papers_in_archive":40},{"url":"/dataset/mtg-jamendo","name":"MTG-Jamendo","full_name":null,"num_papers_in_archive":39},{"url":"/dataset/sen12ms","name":"SEN12MS","full_name":"","num_papers_in_archive":34},{"url":"/dataset/dcase-2016","name":"DCASE 2016","full_name":"DCASE 2016","num_papers_in_archive":31},{"url":"/dataset/uc-merced-land-use-dataset","name":"UC Merced Land Use Dataset","full_name":"","num_papers_in_archive":23},{"url":"/dataset/mlrsnet","name":"MLRSNet","full_name":"","num_papers_in_archive":17},{"url":"/dataset/tau-urban-acoustic-scenes-2019","name":"TAU Urban Acoustic Scenes 2019","full_name":"TAU Urban Acoustic Scenes 2019","num_papers_in_archive":14},{"url":"/dataset/tut-acoustic-scenes-2017","name":"TUT Acoustic Scenes 2017","full_name":"TUT Acoustic Scenes 2017","num_papers_in_archive":13},{"url":"/dataset/rice","name":"RICE","full_name":"Remote sensing Image Cloud rEmoving","num_papers_in_archive":12},{"url":"/dataset/dcase-2013","name":"DCASE 2013","full_name":"DCASE 2013","num_papers_in_archive":11},{"url":"/dataset/aider","name":"AIDER","full_name":"","num_papers_in_archive":8},{"url":"/dataset/cochlscene","name":"CochlScene","full_name":"","num_papers_in_archive":7},{"url":"/dataset/dcase-2019-mobile","name":"DCASE 2019 Mobile","full_name":"TAU Urban Acoustic Scenes 2019 Mobile","num_papers_in_archive":5},{"url":"/dataset/skyeye-968k","name":"SkyEye-968k","full_name":"","num_papers_in_archive":5},{"url":"/dataset/litis-rouen","name":"LITIS Rouen","full_name":"LITIS Rouen","num_papers_in_archive":3},{"url":"/dataset/openstreetmap-multi-sensor-scene","name":"OpenStreetMap Multi-Sensor Scene Classification","full_name":"","num_papers_in_archive":1},{"url":"/dataset/surveillance-camera-fight-dataset","name":"Surveillance Camera Fight Dataset","full_name":null,"num_papers_in_archive":1},{"url":"/dataset/rs-ns92","name":"RS_NS92","full_name":"Remote Sensing Natural Scenes 92 (classes)","num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":148,"tagged_in_all":453,"items":[{"url":"/paper/spatial-information-considered-network-for","title":"Spatial Information Considered Network for Scene Classification","date":"2020-05-18","arxiv_id":null,"repositories_listed":4,"syntology":null},{"url":"/paper/remote-sensing-image-scene-classification","title":"Remote Sensing Image Scene Classification: Benchmark and State of the Art","date":"2017-03-01","arxiv_id":"1703.00121","repositories_listed":4,"syntology":null},{"url":"/paper/rsmamba-remote-sensing-image-classification","title":"RSMamba: Remote Sensing Image Classification with State Space Model","date":"2024-03-28","arxiv_id":"2403.19654","repositories_listed":3,"syntology":{"n":10,"n_ran":8,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/vision-language-models-in-remote-sensing","title":"Vision-Language Models in Remote Sensing: Current Progress and Future Trends","date":"2023-05-09","arxiv_id":"2305.05726","repositories_listed":3,"syntology":null},{"url":"/paper/generalized-scene-classification-from-small","title":"Generalized Scene Classification from Small-Scale Datasets with Multi-Task Learning","date":"2021-10-08","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/scene-graph-augmented-data-driven-risk","title":"Scene-Graph Augmented Data-Driven Risk Assessment of Autonomous Vehicle Decisions","date":"2020-08-31","arxiv_id":"2009.06435","repositories_listed":3,"syntology":null},{"url":"/paper/the-receptive-field-as-a-regularizer-in-deep","title":"The Receptive Field as a Regularizer in Deep Convolutional Neural Networks for Acoustic Scene Classification","date":"2019-07-03","arxiv_id":"1907.01803","repositories_listed":3,"syntology":null},{"url":"/paper/sen12ms-a-curated-dataset-of-georeferenced","title":"SEN12MS -- A Curated Dataset of Georeferenced Multi-Spectral Sentinel-1/2 Imagery for Deep Learning and Data Fusion","date":"2019-06-18","arxiv_id":"1906.07789","repositories_listed":3,"syntology":null},{"url":"/paper/deep-cnns-meet-global-covariance-pooling","title":"Deep CNNs Meet Global Covariance Pooling: Better Representation and Generalization","date":"2019-04-15","arxiv_id":"1904.06836","repositories_listed":3,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/let-there-be-color-joint-end-to-end-learning","title":"Let there be color!: joint end-to-end learning of global and local image priors for automatic image colorization with simultaneous classification","date":"2016-07-01","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/h2rsvlm-towards-helpful-and-honest-remote","title":"VHM: Versatile and Honest Vision Language Model for Remote Sensing Image Analysis","date":"2024-03-29","arxiv_id":"2403.20213","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/mtp-advancing-remote-sensing-foundation-model","title":"MTP: Advancing Remote Sensing Foundation Model via Multi-Task Pretraining","date":"2024-03-20","arxiv_id":"2403.13430","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/efficient-multi-resolution-fusion-for-remote","title":"Efficient Multi-Resolution Fusion for Remote Sensing Data with Label Uncertainty","date":"2024-02-07","arxiv_id":"2402.05045","repositories_listed":2,"syntology":null},{"url":"/paper/decur-decoupling-common-unique","title":"Decoupling Common and Unique Representations for Multimodal Self-supervised Learning","date":"2023-09-11","arxiv_id":"2309.05300","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/efficient-multi-task-scene-analysis-with-rgb","title":"Efficient Multi-Task Scene Analysis with RGB-D Transformers","date":"2023-06-08","arxiv_id":"2306.05242","repositories_listed":2,"syntology":null},{"url":"/paper/efficient-multi-task-rgb-d-scene-analysis-for","title":"Efficient Multi-Task RGB-D Scene Analysis for Indoor Environments","date":"2022-07-10","arxiv_id":"2207.04526","repositories_listed":2,"syntology":null},{"url":"/paper/debiased-pseudo-labeling-in-self-training","title":"Debiased Self-Training for Semi-Supervised Learning","date":"2022-02-15","arxiv_id":"2202.07136","repositories_listed":2,"syntology":null},{"url":"/paper/a-system-of-vision-sensor-based-deep-neural","title":"A system of vision sensor based deep neural networks for complex driving scene analysis in support of crash risk assessment and prevention","date":"2021-06-18","arxiv_id":"2106.10319","repositories_listed":2,"syntology":null},{"url":"/paper/understanding-the-role-of-individual-units-in","title":"Understanding the Role of Individual Units in a Deep Neural Network","date":"2020-09-10","arxiv_id":"2009.05041","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/emergent-properties-of-foveated-perceptual","title":"Emergent Properties of Foveated Perceptual Systems","date":"2020-06-14","arxiv_id":"2006.07991","repositories_listed":2,"syntology":{"n":9,"n_ran":4,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/receptive-field-regularized-cnn-variants-for","title":"Receptive-field-regularized CNN variants for acoustic scene classification","date":"2019-09-05","arxiv_id":"1909.02859","repositories_listed":2,"syntology":null},{"url":"/paper/deep-learning-based-aerial-image","title":"Deep-Learning-Based Aerial Image Classification for Emergency Response Applications Using Unmanned Aerial Vehicles","date":"2019-06-20","arxiv_id":"1906.08716","repositories_listed":2,"syntology":null},{"url":"/paper/a-remote-sensing-image-dataset-for-cloud","title":"A Remote Sensing Image Dataset for Cloud Removal","date":"2019-01-03","arxiv_id":"1901.00600","repositories_listed":2,"syntology":null},{"url":"/paper/training-neural-audio-classifiers-with-few","title":"Training neural audio classifiers with few data","date":"2018-10-24","arxiv_id":"1810.10274","repositories_listed":2,"syntology":null},{"url":"/paper/a-multi-device-dataset-for-urban-acoustic","title":"A multi-device dataset for urban acoustic scene classification","date":"2018-07-25","arxiv_id":"1807.09840","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/a-simple-fusion-of-deep-and-shallow-learning","title":"A Simple Fusion of Deep and Shallow Learning for Acoustic Scene Classification","date":"2018-06-19","arxiv_id":"1806.07506","repositories_listed":2,"syntology":null},{"url":"/paper/exploring-models-and-data-for-remote-sensing","title":"Exploring Models and Data for Remote Sensing Image Caption Generation","date":"2017-12-21","arxiv_id":"1712.07835","repositories_listed":2,"syntology":null},{"url":"/paper/knowledge-guided-disambiguation-for-large","title":"Knowledge Guided Disambiguation for Large-Scale Scene Classification with Multi-Resolution CNNs","date":"2016-10-04","arxiv_id":"1610.01119","repositories_listed":2,"syntology":null},{"url":"/paper/a-challenge-to-build-neuro-symbolic-video","title":"A Challenge to Build Neuro-Symbolic Video Agents","date":"2025-05-20","arxiv_id":"2505.13851","repositories_listed":1,"syntology":null},{"url":"/paper/low-complexity-acoustic-scene-classification-4","title":"Low-Complexity Acoustic Scene Classification with Device Information in the DCASE 2025 Challenge","date":"2025-05-03","arxiv_id":"2505.01747","repositories_listed":1,"syntology":null}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}