{"url":"/dataset/sun-rgb-d","name":"SUN RGB-D","full_name":"SUN RGB-D","description_markdown":"The SUN RGBD dataset contains 10335 real RGB-D images of room scenes. Each RGB image has a corresponding depth and segmentation map. As many as 700 object categories are labeled. The training and testing sets contain 5285 and 5050 images, respectively.\r\n\r\nSource: [Mix and match networks: multi-domain alignment for unpaired image-to-image translation](https://arxiv.org/abs/1903.04294)\r\nImage Source: [https://rgbd.cs.princeton.edu/](https://rgbd.cs.princeton.edu/)","description_withheld":null,"homepage":"https://rgbd.cs.princeton.edu/","introduced_date":"2015-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/sun-rgb-d-a-rgb-d-scene-understanding","title":"SUN RGB-D: A RGB-D Scene Understanding Benchmark Suite","first_author":"Shuran Song","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Interactive","url":"/datasets/modality/interactive"},{"name":"Point cloud","url":"/datasets/modality/point-cloud"},{"name":"RGB-D","url":"/datasets/modality/rgb-d"}],"tasks":[{"name":"Object Detection","url":"/task/object-detection","datasets_with_task":"/datasets/task/object-detection"},{"name":"Semantic Segmentation","url":"/task/semantic-segmentation","datasets_with_task":"/datasets/task/semantic-segmentation"},{"name":"3D Object Detection","url":"/task/3d-object-detection","datasets_with_task":"/datasets/task/3d-object-detection"},{"name":"Panoptic Segmentation","url":"/task/panoptic-segmentation","datasets_with_task":"/datasets/task/panoptic-segmentation"},{"name":"Monocular Depth Estimation","url":"/task/monocular-depth-estimation","datasets_with_task":"/datasets/task/monocular-depth-estimation"},{"name":"Monocular 3D Object Detection","url":"/task/monocular-3d-object-detection","datasets_with_task":"/datasets/task/monocular-3d-object-detection"},{"name":"Scene Segmentation","url":"/task/scene-segmentation","datasets_with_task":"/datasets/task/scene-segmentation"},{"name":"Scene Recognition","url":"/task/scene-recognition","datasets_with_task":"/datasets/task/scene-recognition"},{"name":"Scene Classification (unified classes)","url":"/task/scene-classification-unified-classes","datasets_with_task":"/datasets/task/scene-classification-unified-classes"},{"name":"Robust Semi-Supervised RGBD Semantic Segmentation","url":"/task/robust-semi-supervised-rgbd-semantic","datasets_with_task":"/datasets/task/robust-semi-supervised-rgbd-semantic"},{"name":"Object Detection In Indoor Scenes","url":"/task/object-detection-in-indoor-scenes","datasets_with_task":"/datasets/task/object-detection-in-indoor-scenes"},{"name":"Room Layout Estimation","url":"/task/room-layout-estimation","datasets_with_task":"/datasets/task/room-layout-estimation"},{"name":"Panoptic Segmentation (PanopticNDT instances)","url":"/task/panoptic-segmentation-panopticndt-instances","datasets_with_task":"/datasets/task/panoptic-segmentation-panopticndt-instances"}],"languages":[],"variants":["SUN-RGBD","SUN-RGBD val","SUN RGB-D"],"data_loaders":[{"repo":"https://github.com/open-mmlab/mmdetection3d","url":"https://github.com/open-mmlab/mmdetection3d/blob/master/docs/data_preparation.md","frameworks":["pytorch"]}],"num_papers_in_archive":477,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/semantic-segmentation-on-sun-rgbd","task":"Semantic Segmentation","dataset_variant":"SUN-RGBD","rows":44,"metrics":["Mean IoU","Mean IoU (test)"],"first_row_in_archive_order":{"model":"GeminiFusion (Swin-Large)","paper":"/paper/geminifusion-efficient-pixel-wise-multimodal","metrics":{"Mean IoU":"54.6"},"code_links":[{"title":"jiadingcn/geminifusion","url":"https://github.com/jiadingcn/geminifusion"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/3d-object-detection-on-sun-rgbd-val","task":"3D Object Detection","dataset_variant":"SUN-RGBD val","rows":32,"metrics":["mAP@0.25","mAP@0.5","Inference Speed (s)"],"first_row_in_archive_order":{"model":"Point-GCC+TR3D+FF","paper":"/paper/point-gcc-universal-self-supervised-3d-scene","metrics":{"mAP@0.25":"69.7","mAP@0.5":"54.0"},"code_links":[{"title":"asterisci/point-gcc","url":"https://github.com/asterisci/point-gcc"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/3d-object-detection-on-sun-rgbd","task":"3D Object Detection","dataset_variant":"SUN-RGBD","rows":8,"metrics":["mAP@0.25","mAP@0.5","Inference Speed (s)"],"first_row_in_archive_order":{"model":"Uni3DETR","paper":"/paper/uni3detr-unified-3d-detection-transformer-1","metrics":{"mAP@0.25":"67.0","mAP@0.5":"50.3"},"code_links":[{"title":"zhenyuw16/uni3detr","url":"https://github.com/zhenyuw16/uni3detr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/monocular-3d-object-detection-on-sun-rgb-d","task":"Monocular 3D Object Detection","dataset_variant":"SUN RGB-D","rows":7,"metrics":["AP@0.15 (NYU-37)","AP@0.15 (10 / NYU-37)","AP@0.15 (10 / PNet-30)"],"first_row_in_archive_order":{"model":"IM3D","paper":"/paper/holistic-3d-scene-understanding-from-a-single-1","metrics":{"AP@0.15 (10 / NYU-37)":"45.21","AP@0.15 (NYU-37)":"24.10"},"code_links":[{"title":"chengzhag/Implicit3DUnderstanding","url":"https://github.com/chengzhag/Implicit3DUnderstanding"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-detection-in-indoor-scenes-on-sun-rgb","task":"Object Detection In Indoor Scenes","dataset_variant":"SUN RGB-D","rows":7,"metrics":["AP 0.5"],"first_row_in_archive_order":{"model":"YONOD + CPPM (RGB + Depth)","paper":"/paper/you-only-need-one-detector-unified-object","metrics":{"AP 0.5":"58.1"},"code_links":[{"title":"liketheflower/uoddm","url":"https://github.com/liketheflower/uoddm"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/room-layout-estimation-on-sun-rgb-d","task":"Room Layout Estimation","dataset_variant":"SUN RGB-D","rows":7,"metrics":["IoU","Camera Pitch","Camera Roll"],"first_row_in_archive_order":{"model":"IM3D","paper":"/paper/holistic-3d-scene-understanding-from-a-single-1","metrics":{"Camera Pitch":"2.98","Camera Roll":"2.11","IoU":"64.4"},"code_links":[{"title":"chengzhag/Implicit3DUnderstanding","url":"https://github.com/chengzhag/Implicit3DUnderstanding"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/scene-segmentation-on-sun-rgbd","task":"Scene Segmentation","dataset_variant":"SUN-RGBD","rows":5,"metrics":["Mean IoU"],"first_row_in_archive_order":{"model":"ICM","paper":"/paper/scene-parsing-via-integrated-classification","metrics":{"Mean IoU":"50.60"},"code_links":[{"title":"shihengcan/ICM-matcaffe","url":"https://github.com/shihengcan/ICM-matcaffe"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/monocular-depth-estimation-on-sun-rgbd","task":"Monocular Depth Estimation","dataset_variant":"SUN-RGBD","rows":3,"metrics":["Delta < 1.25","Delta < 1.25^2","Delta < 1.25^3","RMSE","absolute relative error","log 10"],"first_row_in_archive_order":{"model":"RPSF","paper":"/paper/end-to-end-learning-for-joint-depth-and-image","metrics":{"Delta < 1.25":"0.937","Delta < 1.25^2":"0.981","Delta < 1.25^3":"0.992","RMSE":"0.335","absolute relative error":"0.114","log 10":"0.034"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/scene-recognition-on-sun-rgbd","task":"Scene Recognition","dataset_variant":"SUN-RGBD","rows":2,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"OMNIVORE (Swin-B)","paper":"/paper/omnivore-a-single-model-for-many-visual","metrics":{"Accuracy (%)":"67.2"},"code_links":[{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"facebookresearch/omnivore","url":"https://github.com/facebookresearch/omnivore"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-detection-on-sun-rgbd-val","task":"Object Detection","dataset_variant":"SUN-RGBD val","rows":1,"metrics":["MAP"],"first_row_in_archive_order":{"model":"CDSSD","paper":"/paper/how-to-extract-fashion-trends-from-social","metrics":{"MAP":"7"},"code_links":[{"title":"trhgu/awesome-fashion-contents","url":"https://github.com/trhgu/awesome-fashion-contents"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/panoptic-segmentation-on-sun-rgbd","task":"Panoptic Segmentation","dataset_variant":"SUN-RGBD","rows":1,"metrics":["PQ"],"first_row_in_archive_order":{"model":"EMSANet","paper":"/paper/efficient-multi-task-rgb-d-scene-analysis-for","metrics":{"PQ":"52.84"},"code_links":[{"title":"tui-nicr/emsanet","url":"https://github.com/tui-nicr/emsanet"},{"title":"tui-nicr/nicr-scene-analysis-datasets","url":"https://github.com/tui-nicr/nicr-scene-analysis-datasets"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/hdbformer-efficient-rgb-d-semantic","title":"HDBFormer: Efficient RGB-D Semantic Segmentation with A Heterogeneous Dual-Branch Framework","date":"2025-04-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dformerv2-geometry-self-attention-for-rgbd","title":"DFormerv2: Geometry Self-Attention for RGBD Semantic Segmentation","date":"2025-04-07","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/state-space-model-meets-transformer-a-new-1","title":"State Space Model Meets Transformer: A New Paradigm for 3D Object Detection","date":"2025-03-18","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":2,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/diffusion-based-rgb-d-semantic-segmentation","title":"Diffusion-based RGB-D Semantic Segmentation with Deformable Attention Transformer","date":"2024-09-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/scaledepth-decomposing-metric-depth","title":"ScaleDepth: Decomposing Metric Depth Estimation into Scale Prediction and Relative Depth Estimation","date":"2024-07-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/geminifusion-efficient-pixel-wise-multimodal","title":"GeminiFusion: Efficient Pixel-wise Multimodal Fusion for Vision Transformer","date":"2024-06-03","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":7,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/spgroup3d-superpoint-grouping-network-for","title":"SPGroup3D: Superpoint Grouping Network for Indoor 3D Object Detection","date":"2023-12-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/efficient-multimodal-semantic-segmentation","title":"Efficient Multimodal Semantic Segmentation via Dual-Prompt Learning","date":"2023-12-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/uni3detr-unified-3d-detection-transformer-1","title":"Uni3DETR: Unified 3D Detection Transformer","date":"2023-10-09","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/asymformer-asymmetrical-cross-modal","title":"AsymFormer: Asymmetrical Cross-Modal Representation Learning for Mobile Platform Real-Time RGB-D Semantic Segmentation","date":"2023-09-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/panopticndt-efficient-and-robust-panoptic","title":"PanopticNDT: Efficient and Robust Panoptic Mapping","date":"2023-09-24","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":0,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dformer-rethinking-rgbd-representation","title":"DFormer: Rethinking RGBD Representation Learning for Semantic Segmentation","date":"2023-09-18","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/v-detr-detr-with-vertex-relative-position","title":"V-DETR: DETR with Vertex Relative Position Encoding for 3D Object Detection","date":"2023-08-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":10,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/efficient-multi-task-scene-analysis-with-rgb","title":"Efficient Multi-Task Scene Analysis with RGB-D Transformers","date":"2023-06-08","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/point-gcc-universal-self-supervised-3d-scene","title":"Point-GCC: Universal Self-supervised 3D Scene Pre-training via Geometry-Color Contrast","date":"2023-05-31","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/octformer-octree-based-transformers-for-3d","title":"OctFormer: Octree-based Transformers for 3D Point Clouds","date":"2023-05-04","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":2,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/missing-modality-robustness-in-semi","title":"Missing Modality Robustness in Semi-Supervised Multi-Modal Semantic Segmentation","date":"2023-04-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ddp-diffusion-model-for-dense-visual","title":"DDP: Diffusion Model for Dense Visual Prediction","date":"2023-03-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pixel-difference-convolutional-network-for","title":"Pixel Difference Convolutional Network for RGB-D Semantic Segmentation","date":"2023-02-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/tr3d-towards-real-time-indoor-3d-object","title":"TR3D: Towards Real-Time Indoor 3D Object Detection","date":"2023-02-06","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/lcpformer-towards-effective-3d-point-cloud","title":"LCPFormer: Towards Effective 3D Point Cloud Analysis via Local Context Propagation in Transformers","date":"2022-10-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dcanet-differential-convolution-attention","title":"DCANet: Differential Convolution Attention Network for RGB-D Semantic Segmentation","date":"2022-10-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cagroup3d-class-aware-grouping-for-3d-object","title":"CAGroup3D: Class-Aware Grouping for 3D Object Detection on Point Clouds","date":"2022-10-09","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/boosting-3d-object-detection-via-object","title":"Boosting 3D Object Detection via Object-Focused Image Fusion","date":"2022-07-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/efficient-multi-task-rgb-d-scene-analysis-for","title":"Efficient Multi-Task RGB-D Scene Analysis for Indoor Environments","date":"2022-07-10","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/you-only-need-one-detector-unified-object","title":"Unified Object Detector for Different Modalities based on Vision Transformers","date":"2022-07-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/surface-representation-for-point-clouds","title":"Surface Representation for Point Clouds","date":"2022-05-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":2,"samples_unverified":13,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multimodal-token-fusion-for-vision","title":"Multimodal Token Fusion for Vision Transformers","date":"2022-04-19","rows_on_this_dataset":3,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/end-to-end-learning-for-joint-depth-and-image","title":"End-to-end Learning for Joint Depth and Image Reconstruction from Diffracted Rotation","date":"2022-04-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/rbgnet-ray-based-grouping-for-3d-object","title":"RBGNet: Ray-based Grouping for 3D Object Detection","date":"2022-04-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/simcrosstrans-a-simple-cross-modality","title":"simCrossTrans: A Simple Cross-Modality Transfer Learning for Object Detection with ConvNets or Vision Transformers","date":"2022-03-20","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/cmx-cross-modal-fusion-for-rgb-x-semantic","title":"CMX: Cross-Modal Fusion for RGB-X Semantic Segmentation with Transformers","date":"2022-03-09","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/omnivore-a-single-model-for-many-visual","title":"Omnivore: A Single Model for Many Visual Modalities","date":"2022-01-20","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/attention-based-dual-supervised-decoder-for","title":"Attention-based Dual Supervised Decoder for RGBD Semantic Segmentation","date":"2022-01-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/point-cloud-pre-training-with-natural-3d","title":"Point Cloud Pre-Training With Natural 3D Structures","date":"2022-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/fcaf3d-fully-convolutional-anchor-free-3d","title":"FCAF3D: Fully Convolutional Anchor-Free 3D Object Detection","date":"2021-12-01","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/an-end-to-end-transformer-model-for-3d-object","title":"An End-to-End Transformer Model for 3D Object Detection","date":"2021-09-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/spatio-temporal-self-supervised","title":"Spatio-temporal Self-Supervised Representation Learning for 3D Point Clouds","date":"2021-09-01","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/shapeconv-shape-aware-convolutional-layer-for","title":"ShapeConv: Shape-aware Convolutional Layer for Indoor RGB-D Semantic Segmentation","date":"2021-08-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ci-net-contextual-information-for-joint","title":"CI-Net: Contextual Information for Joint Semantic Segmentation and Depth Estimation","date":"2021-07-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/imvoxelnet-image-to-voxels-projection-for","title":"ImVoxelNet: Image to Voxels Projection for Monocular and Multi-View General-Purpose 3D Object Detection","date":"2021-06-02","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deep-feature-selection-and-fusion-for-rgb-d","title":"Deep feature selection-and-fusion for RGB-D semantic segmentation","date":"2021-05-10","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/back-tracing-representative-points-for-voting","title":"Back-tracing Representative Points for Voting-based 3D Object Detection in Point Clouds","date":"2021-04-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/group-free-3d-object-detection-via","title":"Group-Free 3D Object Detection via Transformers","date":"2021-04-01","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":9,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/holistic-3d-scene-understanding-from-a-single-1","title":"Holistic 3D Scene Understanding from a Single Image with Implicit Representation","date":"2021-03-11","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/3d-object-detection-and-instance-segmentation","title":"3D Object Detection and Instance Segmentation from 3D Range and 2D Color Images","date":"2021-02-09","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/efficient-rgb-d-semantic-segmentation-for","title":"Efficient RGB-D Semantic Segmentation for Indoor Scene Analysis","date":"2020-11-13","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/bi-directional-cross-modality-feature","title":"Bi-directional Cross-Modality Feature Propagation with Separation-and-Aggregation Gate for RGB-D Semantic Segmentation","date":"2020-07-17","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/h3dnet-3d-object-detection-using-hybrid","title":"H3DNet: 3D Object Detection Using Hybrid Geometric Primitives","date":"2020-06-10","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/pattern-structure-diffusion-for-multi-task","title":"Pattern-Structure Diffusion for Multi-Task Learning","date":"2020-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-hierarchical-graph-network-for-3d-object","title":"A Hierarchical Graph Network for 3D Object Detection on Point Clouds","date":"2020-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/when-cnns-meet-random-rnns-towards-multi","title":"When CNNs Meet Random RNNs: Towards Multi-Level Analysis for RGB-D Object and Scene Recognition","date":"2020-04-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/spatial-information-guided-convolution-for","title":"Spatial Information Guided Convolution for Real-Time RGBD Semantic Segmentation","date":"2020-04-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/total3dunderstanding-joint-layout-object-pose","title":"Total3DUnderstanding: Joint Layout, Object Pose and Mesh Reconstruction for Indoor Scenes from a Single Image","date":"2020-02-27","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":2,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/attention-guided-chained-context-aggregation","title":"Attention-guided Chained Context Aggregation for Semantic Segmentation","date":"2020-02-27","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/imvotenet-boosting-3d-object-detection-in","title":"ImVoteNet: Boosting 3D Object Detection in Point Clouds with Image Votes","date":"2020-01-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-modal-attention-based-fusion-model-for","title":"Multi-Modal Attention-based Fusion Model for Semantic Segmentation of RGB-Depth Images","date":"2019-12-25","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/perspectivenet-3d-object-detection-from-a-1","title":"PerspectiveNet: 3D Object Detection from a Single RGB Image via Perspective Points","date":"2019-12-16","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/frustum-voxnet-for-3d-object-detection-from","title":"Frustum VoxNet for 3D object detection from RGB-D or Depth images","date":"2019-10-12","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/index-network","title":"Index Network","date":"2019-08-11","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/clouds-of-oriented-gradients-for-3d-detection","title":"Clouds of Oriented Gradients for 3D Detection of Objects, Surfaces, and Indoor Scene Layouts","date":"2019-06-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/scene-parsing-via-integrated-classification","title":"Scene Parsing via Integrated Classification Model and Variance-Based Regularization","date":"2019-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/acnet-attention-based-network-to-exploit","title":"ACNet: Attention Based Network to Exploit Complementary Features for RGBD Semantic Segmentation","date":"2019-05-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deep-hough-voting-for-3d-object-detection-in","title":"Deep Hough Voting for 3D Object Detection in Point Clouds","date":"2019-04-21","rows_on_this_dataset":1,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":2,"samples_unverified":8,"pointer_only_for_licence":10,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cooperative-holistic-scene-understanding","title":"Cooperative Holistic Scene Understanding: Unifying 3D Object, Layout, and Camera Pose Estimation","date":"2018-10-31","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":1,"samples_unverified":17,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-supervised-model-adaptation-for","title":"Self-Supervised Model Adaptation for Multimodal Semantic Segmentation","date":"2018-08-11","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/holistic-3d-scene-parsing-and-reconstruction","title":"Holistic 3D Scene Parsing and Reconstruction from a Single RGB Image","date":"2018-08-07","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/how-to-extract-fashion-trends-from-social","title":"How To Extract Fashion Trends From Social Media? A Robust Object Detector With Support For Unsupervised Learning","date":"2018-06-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rednet-residual-encoder-decoder-network-for","title":"RedNet: Residual Encoder-Decoder Network for indoor RGB-D Semantic Segmentation","date":"2018-06-04","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/context-contrasted-feature-and-gated-multi","title":"Context Contrasted Feature and Gated Multi-Scale Aggregation for Scene Segmentation","date":"2018-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/3d-object-detection-with-latent-support","title":"3D Object Detection With Latent Support Surfaces","date":"2018-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/depth-aware-cnn-for-rgb-d-segmentation","title":"Depth-aware CNN for RGB-D Segmentation","date":"2018-03-19","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":1,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/frustum-pointnets-for-3d-object-detection","title":"Frustum PointNets for 3D Object Detection from RGB-D Data","date":"2017-11-22","rows_on_this_dataset":3,"code_links":68,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rdfnet-rgb-d-multi-level-residual-feature","title":"RDFNet: RGB-D Multi-Level Residual Feature Fusion for Indoor Semantic Segmentation","date":"2017-10-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/3d-graph-neural-networks-for-rgbd-semantic","title":"3D Graph Neural Networks for RGBD Semantic Segmentation","date":"2017-10-01","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/2d-driven-3d-object-detection-in-rgb-d-images","title":"2D-Driven 3D Object Detection in RGB-D Images","date":"2017-10-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/recurrent-scene-parsing-with-perspective","title":"Recurrent Scene Parsing with Perspective Understanding in the Loop","date":"2017-05-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/three-dimensional-object-detection-and-layout","title":"Three-Dimensional Object Detection and Layout Prediction Using Clouds of Oriented Gradients","date":"2016-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/fully-convolutional-networks-for-semantic","title":"Fully Convolutional Networks for Semantic Segmentation","date":"2016-05-20","rows_on_this_dataset":1,"code_links":37,"syntology":null},{"paper":"/paper/deep-sliding-shapes-for-amodal-3d-object","title":"Deep Sliding Shapes for Amodal 3D Object Detection in RGB-D Images","date":"2015-11-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/segnet-a-deep-convolutional-encoder-decoder","title":"SegNet: A Deep Convolutional Encoder-Decoder Architecture for Image Segmentation","date":"2015-11-02","rows_on_this_dataset":1,"code_links":74,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":44,"samples_ran":9,"samples_unverified":35,"pointer_only_for_licence":10,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/semantic-image-segmentation-with-deep","title":"Semantic Image Segmentation with Deep Convolutional Nets and Fully Connected CRFs","date":"2014-12-22","rows_on_this_dataset":1,"code_links":18,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-rich-features-from-rgb-d-images-for","title":"Learning Rich Features from RGB-D Images for Object Detection and Segmentation","date":"2014-07-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/understanding-indoor-scenes-using-3d","title":"Understanding Indoor Scenes Using 3D Geometric Phrases","date":"2013-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":28,"samples_harvested":237,"samples_ran":69,"samples_unverified":168,"pointer_only_for_licence":29,"papers_with_no_sample_that_ran":7,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}