{"url":"/dataset/2d-3d-s","name":"2D-3D-S","full_name":"2D-3D-Semantic","description_markdown":"The **2D-3D-S** dataset provides a variety of mutually registered modalities from 2D, 2.5D and 3D domains, with instance-level semantic and geometric annotations. It covers over 6,000 m2 collected in 6 large-scale indoor areas that originate from 3 different buildings. It contains over 70,000 RGB images, along with the corresponding depths, surface normals, semantic annotations, global XYZ images (all in forms of both regular and 360° equirectangular images) as well as camera information. It also includes registered raw and semantically annotated 3D meshes and point clouds. The dataset enables development of joint and cross-modal learning models and potentially unsupervised approaches utilizing the regularities present in large-scale indoor spaces.\r\n\r\nSource: [https://github.com/alexsax/2D-3D-Semantics](https://github.com/alexsax/2D-3D-Semantics)\r\nImage Source: [https://github.com/alexsax/2D-3D-Semantics](https://github.com/alexsax/2D-3D-Semantics)","description_withheld":null,"homepage":"https://github.com/alexsax/2D-3D-Semantics","introduced_date":"2017-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/joint-2d-3d-semantic-data-for-indoor-scene","title":"Joint 2D-3D-Semantic Data for Indoor Scene Understanding","first_author":"Iro Armeni","url":null},"license":{"name":"Custom","url":"https://github.com/alexsax/2D-3D-Semantics#download"},"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Semantic Segmentation","url":"/task/semantic-segmentation","datasets_with_task":"/datasets/task/semantic-segmentation"},{"name":"Depth Estimation","url":"/task/depth-estimation","datasets_with_task":"/datasets/task/depth-estimation"},{"name":"Semi-Supervised Semantic Segmentation","url":"/task/semi-supervised-semantic-segmentation","datasets_with_task":"/datasets/task/semi-supervised-semantic-segmentation"},{"name":"Self-Supervised Learning","url":"/task/self-supervised-learning","datasets_with_task":"/datasets/task/self-supervised-learning"},{"name":"Visual Navigation","url":"/task/visual-navigation","datasets_with_task":"/datasets/task/visual-navigation"},{"name":"3D Room Layouts From A Single RGB Panorama","url":"/task/3d-room-layouts-from-a-single-rgb-panorama","datasets_with_task":"/datasets/task/3d-room-layouts-from-a-single-rgb-panorama"},{"name":"Robust Semi-Supervised RGBD Semantic Segmentation","url":"/task/robust-semi-supervised-rgbd-semantic","datasets_with_task":"/datasets/task/robust-semi-supervised-rgbd-semantic"},{"name":"Semi-Supervised RGBD Semantic Segmentation","url":"/task/semi-supervised-rgbd-semantic-segmentation","datasets_with_task":"/datasets/task/semi-supervised-rgbd-semantic-segmentation"}],"languages":[],"variants":["2D-3D-S","Stanford2D3D","Stanford2D3D Panoramic","Stanford2D3D Panoramic - RGBD","Stanford2D3D - RGBD"],"data_loaders":[{"repo":"https://github.com/alexsax/2D-3D-Semantics","url":"https://github.com/alexsax/2D-3D-Semantics","frameworks":[]}],"num_papers_in_archive":147,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/semantic-segmentation-on-stanford2d3d-1","task":"Semantic Segmentation","dataset_variant":"Stanford2D3D Panoramic","rows":25,"metrics":["mIoU","mAcc"],"first_row_in_archive_order":{"model":"SFSS-MMSI (RGB+HHA)","paper":"/paper/single-frame-semantic-segmentation-using","metrics":{"mAcc":"70.68","mIoU":"60.6%"},"code_links":[{"title":"sguttikon/SFSS-MMSI","url":"https://github.com/sguttikon/SFSS-MMSI"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/depth-estimation-on-stanford2d3d-panoramic","task":"Depth Estimation","dataset_variant":"Stanford2D3D Panoramic","rows":18,"metrics":["RMSE","absolute relative error"],"first_row_in_archive_order":{"model":"HiMODE","paper":"/paper/himode-a-hybrid-monocular-omnidirectional","metrics":{"RMSE":"0.2619","absolute relative error":"0.0532"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/3d-room-layouts-from-a-single-rgb-panorama-on-3","task":"3D Room Layouts From A Single RGB Panorama","dataset_variant":"Stanford2D3D Panoramic","rows":9,"metrics":["3DIoU","Corner Error","Pixel Error"],"first_row_in_archive_order":{"model":"DMH-Net","paper":"/paper/3d-room-layout-estimation-from-a-cubemap-of","metrics":{"3DIoU":"84.93","Corner Error":"0.67","Pixel Error":"1.93"},"code_links":[{"title":"starrah/dmh-net","url":"https://github.com/starrah/dmh-net"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semantic-segmentation-on-stanford2d3d-rgbd","task":"Semantic Segmentation","dataset_variant":"Stanford2D3D - RGBD","rows":6,"metrics":["mIoU","Pixel Accuracy","mAcc"],"first_row_in_archive_order":{"model":"CMX (SegFormer-B4)","paper":"/paper/cmx-cross-modal-fusion-for-rgb-x-semantic","metrics":{"Pixel Accuracy":"82.6","mIoU":"62.1"},"code_links":[{"title":"huaaaliu/rgbx_semantic_segmentation","url":"https://github.com/huaaaliu/rgbx_semantic_segmentation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semantic-segmentation-on-stanford2d3d-2","task":"Semantic Segmentation","dataset_variant":"Stanford2D3D Panoramic - RGBD","rows":3,"metrics":["mAcc","mIoU"],"first_row_in_archive_order":{"model":"CBFC","paper":"/paper/complementary-bi-directional-feature","metrics":{"mAcc":"70.8","mIoU":"56.7"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semi-supervised-semantic-segmentation-on-2d","task":"Semi-Supervised Semantic Segmentation","dataset_variant":"2D-3D-S","rows":1,"metrics":["mIoU (0.1% labels)","mIoU (0.2% labels)","mIoU (1% labels)"],"first_row_in_archive_order":{"model":"M3L (Linear Fusion B2)","paper":"/paper/missing-modality-robustness-in-semi","metrics":{"mIoU (0.1% labels)":"40.05","mIoU (0.2% labels)":"44.62","mIoU (1% labels)":"49.28"},"code_links":[{"title":"harshm121/m3l","url":"https://github.com/harshm121/m3l"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/single-frame-semantic-segmentation-using","title":"Single Frame Semantic Segmentation Using Multi-Modal Spherical Images","date":"2023-08-18","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/sgat4pass-spherical-geometry-aware","title":"SGAT4PASS: Spherical Geometry-Aware Transformer for PAnoramic Semantic Segmentation","date":"2023-06-06","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":8,"samples_unverified":3,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/missing-modality-robustness-in-semi","title":"Missing Modality Robustness in Semi-Supervised Multi-Modal Semantic Segmentation","date":"2023-04-21","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/interpolated-selectionconv-for-spherical","title":"Interpolated SelectionConv for Spherical Images and Surfaces","date":"2022-10-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/fredsnet-joint-monocular-depth-and-semantic","title":"FreDSNet: Joint Monocular Depth and Semantic Segmentation with Fast Fourier Convolutions","date":"2022-10-04","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/bifuse-self-supervised-and-efficient-bi","title":"BiFuse++: Self-supervised and Efficient Bi-projection Fusion for 360 Depth Estimation","date":"2022-09-07","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":1,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/spheredepth-panorama-depth-estimation-from","title":"SphereDepth: Panorama Depth Estimation from Spherical Domain","date":"2022-08-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/neural-contourlet-network-for-monocular-360","title":"Neural Contourlet Network for Monocular 360 Depth Estimation","date":"2022-08-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/behind-every-domain-there-is-a-shift-adapting","title":"Behind Every Domain There is a Shift: Adapting Distortion-aware Vision Transformers for Panoramic Semantic Segmentation","date":"2022-07-25","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/3d-room-layout-estimation-from-a-cubemap-of","title":"3D Room Layout Estimation from a Cubemap of Panorama Image via Deep Manhattan Hough Transform","date":"2022-07-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":7,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/complementary-bi-directional-feature","title":"Complementary Bi-directional Feature Compression for Indoor 360° Semantic Segmentation with Self-distillation","date":"2022-07-06","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/himode-a-hybrid-monocular-omnidirectional","title":"HiMODE: A Hybrid Monocular Omnidirectional Depth Estimation Model","date":"2022-04-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/panoformer-panorama-transformer-for-indoor","title":"PanoFormer: Panorama Transformer for Indoor 360 Depth Estimation","date":"2022-03-17","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":0,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cmx-cross-modal-fusion-for-rgb-x-semantic","title":"CMX: Cross-Modal Fusion for RGB-X Semantic Segmentation with Transformers","date":"2022-03-09","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/omnifusion-360-monocular-depth-estimation-via","title":"OmniFusion: 360 Monocular Depth Estimation via Geometry-Aware Fusion","date":"2022-03-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bending-reality-distortion-aware-transformers","title":"Bending Reality: Distortion-aware Transformers for Adapting to Panoramic Semantic Segmentation","date":"2022-03-02","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/glpanodepth-global-to-local-panoramic-depth","title":"GLPanoDepth: Global-to-Local Panoramic Depth Estimation","date":"2022-02-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/panodepth-a-two-stage-approach-for-monocular","title":"PanoDepth: A Two-Stage Approach for Monocular Omnidirectional Depth Estimation","date":"2022-02-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/acdnet-adaptively-combined-dilated","title":"ACDNet: Adaptively Combined Dilated Convolution for Monocular Panorama Depth Estimation","date":"2021-12-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-360-monocular-depth-estimation-via","title":"Improving 360 Monocular Depth Estimation via Non-local Dense Prediction Transformer and Joint Supervised and Self-supervised Learning","date":"2021-09-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/shapeconv-shape-aware-convolutional-layer-for","title":"ShapeConv: Shape-aware Convolutional Layer for Indoor RGB-D Semantic Segmentation","date":"2021-08-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/slicenet-deep-dense-depth-estimation-from-a","title":"SliceNet: Deep Dense Depth Estimation From a Single Indoor Panorama Using a Slice-Based Representation","date":"2021-06-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/omnilayout-room-layout-reconstruction-from","title":"OmniLayout: Room Layout Reconstruction from Indoor Spherical Panoramas","date":"2021-04-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/led2-net-monocular-360-layout-estimation-via","title":"LED2-Net: Monocular 360 Layout Estimation via Differentiable Depth Rendering","date":"2021-04-01","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":1,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sslayout360-semi-supervised-indoor-layout","title":"SSLayout360: Semi-Supervised Indoor Layout Estimation from 360-Degree Panorama","date":"2021-03-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unifuse-unidirectional-fusion-for-360-circ","title":"UniFuse: Unidirectional Fusion for 360$^{\\circ}$ Panorama Depth Estimation","date":"2021-02-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hohonet-360-indoor-holistic-understanding","title":"HoHoNet: 360 Indoor Holistic Understanding with Latent Horizontal Features","date":"2020-11-23","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/atlantanet-inferring-the-3d-indoor-layout","title":"AtlantaNet: Inferring the 3D Indoor Layout from a Single 360(∘) Image beyond the Manhattan World Assumption","date":"2020-08-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/spin-weighted-spherical-cnns","title":"Spin-Weighted Spherical CNNs","date":"2020-06-18","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/geometric-structure-based-and-regularized","title":"Geometric Structure Based and Regularized Depth Estimation From 360 Indoor Imagery","date":"2020-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/bifuse-monocular-360-depth-estimation-via-bi","title":"BiFuse: Monocular 360 Depth Estimation via Bi-Projection Fusion","date":"2020-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multi-modal-attention-based-fusion-model-for","title":"Multi-Modal Attention-based Fusion Model for Semantic Segmentation of RGB-Depth Images","date":"2019-12-25","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/tangent-images-for-mitigating-spherical","title":"Tangent Images for Mitigating Spherical Distortion","date":"2019-12-19","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/orientation-aware-semantic-segmentation-on","title":"Orientation-aware Semantic Segmentation on Icosahedron Spheres","date":"2019-07-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gauge-equivariant-convolutional-networks-and","title":"Gauge Equivariant Convolutional Networks and the Icosahedral CNN","date":"2019-02-11","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":1,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/horizonnet-learning-room-layout-with-1d","title":"HorizonNet: Learning Room Layout with 1D Representation and Pano Stretch Data Augmentation","date":"2019-01-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/spherical-cnns-on-unstructured-grids","title":"Spherical CNNs on Unstructured Grids","date":"2019-01-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dula-net-a-dual-projection-network-for","title":"DuLa-Net: A Dual-Projection Network for Estimating Room Layouts from a Single RGB Panorama","date":"2018-11-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/distortion-aware-convolutional-filters-for","title":"Distortion-Aware Convolutional Filters for Dense Prediction in Panoramic Images","date":"2018-09-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/omnidepth-dense-depth-estimation-for-indoors","title":"OmniDepth: Dense Depth Estimation for Indoors Spherical Panoramas","date":"2018-07-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/layoutnet-reconstructing-the-3d-room-layout","title":"LayoutNet: Reconstructing the 3D Room Layout from a Single RGB Image","date":"2018-03-23","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":27,"samples_ran":1,"samples_unverified":26,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/depth-aware-cnn-for-rgb-d-segmentation","title":"Depth-aware CNN for RGB-D Segmentation","date":"2018-03-19","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":1,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":16,"samples_harvested":139,"samples_ran":36,"samples_unverified":103,"pointer_only_for_licence":11,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}