{"url":"/dataset/nyuv2","name":"NYUv2","full_name":"NYU-Depth V2","description_markdown":"The **NYU-Depth V2** data set is comprised of video sequences from a variety of indoor scenes as recorded by both the RGB and Depth cameras from the Microsoft Kinect. It features:\r\n\r\n* 1449 densely labeled pairs of aligned RGB and depth images\r\n* 464 new scenes taken from 3 cities\r\n* 407,024 new unlabeled frames\r\n* Each object is labeled with a class and an instance number.\r\nThe dataset has several components:\r\n* Labeled: A subset of the video data accompanied by dense multi-class labels. This data has also been preprocessed to fill in missing depth labels.\r\n* Raw: The raw RGB, depth and accelerometer data as provided by the Kinect.\r\n* Toolbox: Useful functions for manipulating the data and labels.\r\n\r\nSource: [https://cs.nyu.edu/~silberman/datasets/nyu_depth_v2.html](https://cs.nyu.edu/~silberman/datasets/nyu_depth_v2.html)\r\nImage Source: [https://cs.nyu.edu/~silberman/datasets/nyu_depth_v2.html](https://cs.nyu.edu/~silberman/datasets/nyu_depth_v2.html)","description_withheld":null,"homepage":"https://cs.nyu.edu/~silberman/datasets/nyu_depth_v2.html","introduced_date":"2012-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"Indoor Segmentation and Support Inference from RGBD Images","first_author":null,"url":"http://arxiv.org/pdf/1301.3572.pdf"},"license":{"name":"Unknown","url":null},"modalities":[{"name":"RGB-D","url":"/datasets/modality/rgb-d"}],"tasks":[{"name":"Semantic Segmentation","url":"/task/semantic-segmentation","datasets_with_task":"/datasets/task/semantic-segmentation"},{"name":"Instance Segmentation","url":"/task/instance-segmentation","datasets_with_task":"/datasets/task/instance-segmentation"},{"name":"3D Object Detection","url":"/task/3d-object-detection","datasets_with_task":"/datasets/task/3d-object-detection"},{"name":"Depth Estimation","url":"/task/depth-estimation","datasets_with_task":"/datasets/task/depth-estimation"},{"name":"Panoptic Segmentation","url":"/task/panoptic-segmentation","datasets_with_task":"/datasets/task/panoptic-segmentation"},{"name":"Monocular Depth Estimation","url":"/task/monocular-depth-estimation","datasets_with_task":"/datasets/task/monocular-depth-estimation"},{"name":"Multi-Task Learning","url":"/task/multi-task-learning","datasets_with_task":"/datasets/task/multi-task-learning"},{"name":"Scene Segmentation","url":"/task/scene-segmentation","datasets_with_task":"/datasets/task/scene-segmentation"},{"name":"Depth Completion","url":"/task/depth-completion","datasets_with_task":"/datasets/task/depth-completion"},{"name":"Boundary Detection","url":"/task/boundary-detection","datasets_with_task":"/datasets/task/boundary-detection"},{"name":"Real-Time Semantic Segmentation","url":"/task/real-time-semantic-segmentation","datasets_with_task":"/datasets/task/real-time-semantic-segmentation"},{"name":"Surface Normals Estimation","url":"/task/surface-normals-estimation","datasets_with_task":"/datasets/task/surface-normals-estimation"},{"name":"3D Semantic Scene Completion","url":"/task/3d-semantic-scene-completion","datasets_with_task":"/datasets/task/3d-semantic-scene-completion"},{"name":"3D Semantic Scene Completion from a single RGB image","url":"/task/3d-semantic-scene-completion-from-a-single","datasets_with_task":"/datasets/task/3d-semantic-scene-completion-from-a-single"},{"name":"Surface Normal Estimation","url":"/task/surface-normal-estimation","datasets_with_task":"/datasets/task/surface-normal-estimation"},{"name":"Scene Classification (unified classes)","url":"/task/scene-classification-unified-classes","datasets_with_task":"/datasets/task/scene-classification-unified-classes"},{"name":"Plane Instance Segmentation","url":"/task/plane-instance-segmentation","datasets_with_task":"/datasets/task/plane-instance-segmentation"},{"name":"Zero-shot Scene Classification (unified classes)","url":"/task/zero-shot-scene-classification-unified","datasets_with_task":"/datasets/task/zero-shot-scene-classification-unified"}],"languages":[],"variants":["NYU-Depth V2","NYU Depth v2","NYU-Depth V2 Surface Normals","NYUv2","NYU-Depth V2 self-supervised"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/0jl/NYUv2","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/sayakpaul/nyu_depth_v2","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/nyu_depth_v2","frameworks":["tf","jax"]}],"num_papers_in_archive":986,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/semantic-segmentation-on-nyu-depth-v2","task":"Semantic Segmentation","dataset_variant":"NYU Depth v2","rows":121,"metrics":["Mean IoU","Mean Accuracy"],"first_row_in_archive_order":{"model":"OmniVec2","paper":"/paper/omnivec2-a-novel-transformer-based-network","metrics":{"Mean IoU":"63.6"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/monocular-depth-estimation-on-nyu-depth-v2","task":"Monocular Depth Estimation","dataset_variant":"NYU-Depth V2","rows":85,"metrics":["absolute relative error","RMSE","log 10","Delta < 1.25","Delta < 1.25^2","Delta < 1.25^3"],"first_row_in_archive_order":{"model":"HybridDepth","paper":"/paper/hybriddepth-robust-depth-fusion-for-mobile-ar","metrics":{"Delta < 1.25":"0.988","Delta < 1.25^2":"1.000","Delta < 1.25^3":"1.000","RMSE":"0.128","absolute relative error":"0.026"},"code_links":[{"title":"cake-lab/hybriddepth","url":"https://github.com/cake-lab/hybriddepth"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/3d-semantic-scene-completion-on-nyuv2","task":"3D Semantic Scene Completion","dataset_variant":"NYUv2","rows":28,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"SG-SSC","paper":"/paper/2d-semantic-guided-semantic-scene-completion","metrics":{"mIoU":"55.4"},"code_links":[{"title":"aipixel/SG-SSC","url":"https://github.com/aipixel/SG-SSC"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/depth-estimation-on-nyu-depth-v2","task":"Depth Estimation","dataset_variant":"NYU-Depth V2","rows":17,"metrics":["RMS","RMSE","mAP"],"first_row_in_archive_order":{"model":"EVP","paper":"/paper/evp-enhanced-visual-perception-using-inverse","metrics":{"RMS":"0.224"},"code_links":[{"title":"lavreniuk/evp","url":"https://github.com/lavreniuk/evp"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/real-time-semantic-segmentation-on-nyu-depth-1","task":"Real-Time Semantic Segmentation","dataset_variant":"NYU Depth v2","rows":10,"metrics":["mIoU","Speed(ms/f)","Speed  (FPS)"],"first_row_in_archive_order":{"model":"AsymFormer","paper":"/paper/asymformer-asymmetrical-cross-modal","metrics":{"Speed  (FPS)":"65.5 (3090)","Speed(ms/f)":"15.3","mIoU":"54.1"},"code_links":[{"title":"Fourier7754/AsymFormer","url":"https://github.com/Fourier7754/AsymFormer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/monocular-depth-estimation-on-nyu-depth-v2-4","task":"Monocular Depth Estimation","dataset_variant":"NYU-Depth V2 self-supervised","rows":8,"metrics":["Root mean square error (RMSE)","Absolute relative error (AbsRel)","delta_1","delta_2","delta_3"],"first_row_in_archive_order":{"model":"IndoorDepth","paper":"/paper/deeper-into-self-supervised-monocular-indoor","metrics":{"Absolute relative error (AbsRel)":"0.126","Root mean square error (RMSE)":"0.494","delta_1":"84.5","delta_2":"96.5","delta_3":"99.1"},"code_links":[{"title":"fcntes/indoordepth","url":"https://github.com/fcntes/indoordepth"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/3d-semantic-scene-completion-from-a-single","task":"3D Semantic Scene Completion from a single RGB image","dataset_variant":"NYUv2","rows":6,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"ISO","paper":"/paper/monocular-occupancy-prediction-for-scalable","metrics":{"mIoU":"31.25"},"code_links":[{"title":"hongxiaoy/ISO","url":"https://github.com/hongxiaoy/ISO"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/surface-normals-estimation-on-nyu-depth-v2-1","task":"Surface Normals Estimation","dataset_variant":"NYU Depth v2","rows":6,"metrics":["% < 11.25","% < 22.5","% < 30","Mean Angle Error","RMSE"],"first_row_in_archive_order":{"model":"Metric3Dv2(L, FT)","paper":"/paper/metric3d-v2-a-versatile-monocular-geometric-1","metrics":{"% < 11.25":"68.8","% < 22.5":"84.9","% < 30":"89.8","Mean Angle Error":"12.0","RMSE":"19.2"},"code_links":[{"title":"yvanyin/metric3d","url":"https://github.com/yvanyin/metric3d"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/boundary-detection-on-nyu-depth-v2","task":"Boundary Detection","dataset_variant":"NYU-Depth V2","rows":3,"metrics":["odsF"],"first_row_in_archive_order":{"model":"InvPT","paper":"/paper/inverted-pyramid-multi-task-transformer-for","metrics":{"odsF":"78.1"},"code_links":[{"title":"prismformore/InvPT","url":"https://github.com/prismformore/InvPT"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/depth-completion-on-nyu-depth-v2","task":"Depth Completion","dataset_variant":"NYU-Depth V2","rows":3,"metrics":["RMSE","REL"],"first_row_in_archive_order":{"model":"NLSPN","paper":"/paper/non-local-spatial-propagation-network-for","metrics":{"REL":"0.012","RMSE":"0.092"},"code_links":[{"title":"zzangjinsun/NLSPN_ECCV20","url":"https://github.com/zzangjinsun/NLSPN_ECCV20"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-task-learning-on-nyuv2","task":"Multi-Task Learning","dataset_variant":"NYUv2","rows":2,"metrics":["Mean IoU"],"first_row_in_archive_order":{"model":"SwinMTL","paper":"/paper/swinmtl-a-shared-architecture-for","metrics":{"Mean IoU":"58.14"},"code_links":[{"title":"pardistaghavi/swinmtl","url":"https://github.com/pardistaghavi/swinmtl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/panoptic-segmentation-on-nyu-depth-v2","task":"Panoptic Segmentation","dataset_variant":"NYU Depth v2","rows":2,"metrics":["PQ"],"first_row_in_archive_order":{"model":"EMSANet (2x ResNet-34 NBt1D, PanopticNDT version, finetuned)","paper":"/paper/panopticndt-efficient-and-robust-panoptic","metrics":{"PQ":"51.15"},"code_links":[{"title":"tui-nicr/emsanet","url":"https://github.com/tui-nicr/emsanet"},{"title":"tui-nicr/panoptic-mapping","url":"https://github.com/tui-nicr/panoptic-mapping"},{"title":"tui-nicr/nicr-scene-analysis-datasets","url":"https://github.com/tui-nicr/nicr-scene-analysis-datasets"},{"title":"tui-nicr/emsaformer","url":"https://github.com/tui-nicr/emsaformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/3d-object-detection-on-nyu-depth-v2","task":"3D Object Detection","dataset_variant":"NYU Depth v2","rows":1,"metrics":["MAP"],"first_row_in_archive_order":{"model":"SGPN-CNN","paper":"/paper/sgpn-similarity-group-proposal-network-for-3d","metrics":{"MAP":"41.3"},"code_links":[{"title":"laughtervv/SGPN","url":"https://github.com/laughtervv/SGPN"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/instance-segmentation-on-nyu-depth-v2","task":"Instance Segmentation","dataset_variant":"NYU Depth v2","rows":1,"metrics":["mAP@0.5"],"first_row_in_archive_order":{"model":"SGPN-CNN","paper":"/paper/sgpn-similarity-group-proposal-network-for-3d","metrics":{"mAP@0.5":"30.5"},"code_links":[{"title":"laughtervv/SGPN","url":"https://github.com/laughtervv/SGPN"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/scene-segmentation-on-nyu-depth-v2","task":"Scene Segmentation","dataset_variant":"NYU Depth v2","rows":1,"metrics":["Mean IoU"],"first_row_in_archive_order":{"model":"Dilated FCN-2s RGB","paper":"/paper/efficient-yet-deep-convolutional-neural","metrics":{"Mean IoU":"32.3%"},"code_links":[{"title":"SharifAmit/DilatedFCNSegmentation","url":"https://github.com/SharifAmit/DilatedFCNSegmentation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/surface-normals-estimation-on-nyu-depth-v2","task":"Surface Normals Estimation","dataset_variant":"NYU-Depth V2 Surface Normals","rows":1,"metrics":["RMSE"],"first_row_in_archive_order":{"model":"DSN","paper":"/paper/on-deep-learning-techniques-to-boost","metrics":{"RMSE":"12.2"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/depthmatch-semi-supervised-rgb-d-scene","title":"DepthMatch: Semi-Supervised RGB-D Scene Parsing through Depth-Guided Regularization","date":"2025-05-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/hdbformer-efficient-rgb-d-semantic","title":"HDBFormer: Efficient RGB-D Semantic Segmentation with A Heterogeneous Dual-Branch Framework","date":"2025-04-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dformerv2-geometry-self-attention-for-rgbd","title":"DFormerv2: Geometry Self-Attention for RGBD Semantic Segmentation","date":"2025-04-07","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unik3d-universal-camera-monocular-3d","title":"UniK3D: Universal Camera Monocular 3D Estimation","date":"2025-03-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unidepthv2-universal-monocular-metric-depth","title":"UniDepthV2: Universal Monocular Metric Depth Estimation Made Simpler","date":"2025-02-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/distill-any-depth-distillation-creates-a","title":"Distill Any Depth: Distillation Creates a Stronger Monocular Depth Estimator","date":"2025-02-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hspformer-hierarchical-spatial-perception","title":"HSPFormer: Hierarchical Spatial Perception Transformer for Semantic Segmentation","date":"2025-01-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/depthmaster-taming-diffusion-models-for","title":"DepthMaster: Taming Diffusion Models for Monocular Depth Estimation","date":"2025-01-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/diffusion-based-rgb-d-semantic-segmentation","title":"Diffusion-based RGB-D Semantic Segmentation with Deformable Attention Transformer","date":"2024-09-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/fine-tuning-image-conditional-diffusion","title":"Fine-Tuning Image-Conditional Diffusion Models is Easier than You Think","date":"2024-09-17","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":16,"samples_unverified":1,"pointer_only_for_licence":17,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/grin-zero-shot-metric-depth-with-pixel-level","title":"GRIN: Zero-Shot Metric Depth with Pixel-Level Diffusion","date":"2024-09-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/primedepth-efficient-monocular-depth","title":"PrimeDepth: Efficient Monocular Depth Estimation with a Stable Diffusion Preimage","date":"2024-09-13","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":13,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/2d-semantic-guided-semantic-scene-completion","title":"2D Semantic-Guided Semantic Scene Completion","date":"2024-09-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hybriddepth-robust-depth-fusion-for-mobile-ar","title":"HybridDepth: Robust Metric Depth Fusion by Leveraging Depth from Focus and Single-Image Priors","date":"2024-07-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/monocular-occupancy-prediction-for-scalable","title":"Monocular Occupancy Prediction for Scalable Indoor Scenes","date":"2024-07-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scaledepth-decomposing-metric-depth","title":"ScaleDepth: Decomposing Metric Depth Estimation into Scale Prediction and Relative Depth Estimation","date":"2024-07-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/geminifusion-efficient-pixel-wise-multimodal","title":"GeminiFusion: Efficient Pixel-wise Multimodal Fusion for Vision Transformer","date":"2024-06-03","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":7,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hapnet-toward-superior-rgb-thermal-scene","title":"HAPNet: Toward Superior RGB-Thermal Scene Parsing via Hybrid, Asymmetric, and Progressive Heterogeneous Feature Fusion","date":"2024-04-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unidepth-universal-monocular-metric-depth","title":"UniDepth: Universal Monocular Metric Depth Estimation","date":"2024-03-27","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":26,"samples_ran":15,"samples_unverified":11,"pointer_only_for_licence":25,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ecodepth-effective-conditioning-of-diffusion","title":"ECoDepth: Effective Conditioning of Diffusion Models for Monocular Depth Estimation","date":"2024-03-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":11,"samples_unverified":4,"pointer_only_for_licence":15,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/metric3d-v2-a-versatile-monocular-geometric-1","title":"Metric3Dv2: A Versatile Monocular Geometric Foundation Model for Zero-shot Metric Depth and Surface Normal Estimation","date":"2024-03-22","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/futuredepth-learning-to-predict-the-future","title":"FutureDepth: Learning to Predict the Future Improves Video Depth Estimation","date":"2024-03-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/swinmtl-a-shared-architecture-for","title":"SwinMTL: A Shared Architecture for Simultaneous Depth Estimation and Semantic Segmentation from Monocular Camera Images","date":"2024-03-15","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/depth-anything-unleashing-the-power-of-large","title":"Depth Anything: Unleashing the Power of Large-Scale Unlabeled Data","date":"2024-01-19","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":3,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/context-aware-interaction-network-for-rgb-t","title":"Context-Aware Interaction Network for RGB-T Semantic Segmentation","date":"2024-01-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/omnivec2-a-novel-transformer-based-network","title":"OmniVec2 - A Novel Transformer based Network for Large Scale Multimodal and Multitask Learning","date":"2024-01-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/harnessing-diffusion-models-for-visual","title":"Harnessing Diffusion Models for Visual Perception with Meta Prompts","date":"2023-12-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zero-shot-metric-depth-with-a-field-of-view","title":"Zero-Shot Metric Depth with a Field-of-View Conditioned Diffusion Model","date":"2023-12-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/evp-enhanced-visual-perception-using-inverse","title":"EVP: Enhanced Visual Perception using Inverse Multi-Attentive Feature Refinement and Regularized Image-Text Alignment","date":"2023-12-13","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/repurposing-diffusion-based-image-generators","title":"Repurposing Diffusion-Based Image Generators for Monocular Depth Estimation","date":"2023-12-04","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":26,"samples_ran":14,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deeper-into-self-supervised-monocular-indoor","title":"Deeper into Self-Supervised Monocular Indoor Depth Estimation","date":"2023-12-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/efficient-multimodal-semantic-segmentation","title":"Efficient Multimodal Semantic Segmentation via Dual-Prompt Learning","date":"2023-12-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/optimizing-rgb-d-semantic-segmentation","title":"Optimizing rgb-d semantic segmentation through multi-modal interaction and pooling attention","date":"2023-11-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/polymax-general-dense-prediction-with-mask","title":"PolyMaX: General Dense Prediction with Mask Transformer","date":"2023-11-09","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/omnivec-learning-robust-representations-with","title":"OmniVec: Learning robust representations with cross modal sharing","date":"2023-11-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/analysis-of-nan-divergence-in-training","title":"Analysis of NaN Divergence in Training Monocular Depth Estimation Model","date":"2023-11-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/haarnet-large-scale-linear-morphological","title":"HaarNet: Large-scale Linear-Morphological Hybrid Network for RGB-D Semantic Segmentation","date":"2023-10-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/mesa-masked-geometric-and-supervised-pre","title":"MeSa: Masked, Geometric, and Supervised Pre-training for Monocular Depth Estimation","date":"2023-10-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/text-image-alignment-for-diffusion-based","title":"Text-image Alignment for Diffusion-based Perception","date":"2023-09-29","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/ndc-scene-boost-monocular-3d-semantic-scene","title":"NDC-Scene: Boost Monocular 3D Semantic Scene Completion in Normalized Device Coordinates Space","date":"2023-09-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":5,"samples_unverified":2,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/iebins-iterative-elastic-bins-for-monocular-1","title":"IEBins: Iterative Elastic Bins for Monocular Depth Estimation","date":"2023-09-25","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":8,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/asymformer-asymmetrical-cross-modal","title":"AsymFormer: Asymmetrical Cross-Modal Representation Learning for Mobile Platform Real-Time RGB-D Semantic Segmentation","date":"2023-09-25","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/panopticndt-efficient-and-robust-panoptic","title":"PanopticNDT: Efficient and Robust Panoptic Mapping","date":"2023-09-24","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":0,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/nddepth-normal-distance-assisted-monocular","title":"NDDepth: Normal-Distance Assisted Monocular Depth Estimation","date":"2023-09-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/large-scale-monocular-depth-estimation-in-the","title":"Large-scale Monocular Depth Estimation in the Wild","date":"2023-09-18","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dformer-rethinking-rgbd-representation","title":"DFormer: Rethinking RGBD Representation Learning for Semantic Segmentation","date":"2023-09-18","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/understanding-dark-scenes-by-contrasting","title":"Understanding Dark Scenes by Contrasting Multi-Modal Observations","date":"2023-08-23","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/spatial-information-guided-adaptive-context","title":"Spatial-information Guided Adaptive Context-aware Network for Efficient RGB-D Semantic Segmentation","date":"2023-08-11","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/prompt-guided-transformer-for-multi-task","title":"Prompt Guided Transformer for Multi-Task Dense Prediction","date":"2023-07-28","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/comptr-towards-diverse-bi-source-dense","title":"ComPtr: Towards Diverse Bi-source Dense Prediction Tasks via A Simple yet General Complementary Transformer","date":"2023-07-23","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/metric3d-towards-zero-shot-metric-3d","title":"Metric3D: Towards Zero-shot Metric 3D Prediction from A Single Image","date":"2023-07-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/neural-video-depth-stabilizer","title":"NVDS+: Towards Efficient and Versatile Neural Stabilizer for Video Depth Estimation","date":"2023-07-17","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/efficient-multi-task-scene-analysis-with-rgb","title":"Efficient Multi-Task Scene Analysis with RGB-D Transformers","date":"2023-06-08","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/mmanet-margin-aware-distillation-and-modality","title":"MMANet: Margin-aware Distillation and Modality-aware Regularization for Incomplete Multimodal Learning","date":"2023-04-17","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":0,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dinov2-learning-robust-visual-features","title":"DINOv2: Learning Robust Visual Features without Supervision","date":"2023-04-14","rows_on_this_dataset":2,"code_links":26,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":46,"samples_ran":21,"samples_unverified":25,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/idisc-internal-discretization-for-monocular","title":"iDisc: Internal Discretization for Monocular Depth Estimation","date":"2023-04-13","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ddp-diffusion-model-for-dense-visual","title":"DDP: Diffusion Model for Dense Visual Prediction","date":"2023-03-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unleashing-text-to-image-diffusion-models-for-1","title":"Unleashing Text-to-Image Diffusion Models for Visual Perception","date":"2023-03-03","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/delivering-arbitrary-modal-semantic","title":"Delivering Arbitrary-Modal Semantic Segmentation","date":"2023-03-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/monocular-depth-estimation-using-diffusion","title":"Monocular Depth Estimation using Diffusion Models","date":"2023-02-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/occdepth-a-depth-aware-method-for-3d-semantic","title":"OccDepth: A Depth-Aware Method for 3D Semantic Scene Completion","date":"2023-02-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zoedepth-zero-shot-transfer-by-combining","title":"ZoeDepth: Zero-shot Transfer by Combining Relative and Metric Depth","date":"2023-02-23","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":10,"samples_unverified":6,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pixel-difference-convolutional-network-for","title":"Pixel Difference Convolutional Network for RGB-D Semantic Segmentation","date":"2023-02-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/urcdc-depth-uncertainty-rectified-cross","title":"URCDC-Depth: Uncertainty Rectified Cross-Distillation with CutFlip for Monocular Depth Estimation","date":"2023-02-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/va-depthnet-a-variational-approach-to-single","title":"VA-DepthNet: A Variational Approach to Single Image Depth Prediction","date":"2023-02-13","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-deep-regression-with-ordinal","title":"Improving Deep Regression with Ordinal Entropy","date":"2023-01-21","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/all-in-tokens-unifying-output-space-of-visual","title":"All in Tokens: Unifying Output Space of Visual Tasks via Soft Token","date":"2023-01-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/attention-attention-everywhere-monocular","title":"Attention Attention Everywhere: Monocular Depth Prediction with Skip Attention","date":"2022-10-17","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-task-meta-learning-learn-how-to-adapt","title":"Multi-Task Meta Learning: learn how to adapt to unseen tasks","date":"2022-10-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dcanet-differential-convolution-attention","title":"DCANet: Differential Convolution Attention Network for RGB-D Semantic Segmentation","date":"2022-10-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/composite-learning-for-robust-and-effective","title":"Composite Learning for Robust and Effective Dense Predictions","date":"2022-10-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/irondepth-iterative-refinement-of-single-view","title":"IronDepth: Iterative Refinement of Single-View Depth using Surface Normal and its Uncertainty","date":"2022-10-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/toward-edge-efficient-dense-predictions-with","title":"Toward Edge-Efficient Dense Predictions with Synergistic Multi-Task Neural Architecture Search","date":"2022-10-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/masked-supervised-learning-for-semantic","title":"Masked Supervised Learning for Semantic Segmentation","date":"2022-10-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/depth-map-decomposition-for-monocular-depth","title":"Depth Map Decomposition for Monocular Depth Estimation","date":"2022-08-23","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/focal-wnet-an-architecture-unifying","title":"Focal-WNet: An Architecture Unifying Convolution and Attention for Depth Estimation","date":"2022-07-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/efficient-multi-task-rgb-d-scene-analysis-for","title":"Efficient Multi-Task RGB-D Scene Analysis for Indoor Environments","date":"2022-07-10","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/depthformer-multiscale-vision-transformer-for","title":"Depthformer : Multiscale Vision Transformer For Monocular Depth Estimation With Local Global Information Fusion","date":"2022-07-10","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cross-task-attention-mechanism-for-dense","title":"DenseMTL: Cross-task Attention Mechanism for Dense Multi-task Learning","date":"2022-06-17","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/depth-adapted-cnns-for-rgb-d-semantic","title":"Depth-Adapted CNNs for RGB-D Semantic Segmentation","date":"2022-06-08","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/mmformer-multimodal-medical-transformer-for","title":"mmFormer: Multimodal Medical Transformer for Incomplete Multimodal Learning of Brain Tumor Segmentation","date":"2022-06-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":6,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/revealing-the-dark-secrets-of-masked-image","title":"Revealing the Dark Secrets of Masked Image Modeling","date":"2022-05-26","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/struct-mdc-mesh-refined-unsupervised-depth","title":"Struct-MDC: Mesh-Refined Unsupervised Depth Completion Leveraging Structural Regularities from Visual SLAM","date":"2022-04-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multimodal-token-fusion-for-vision","title":"Multimodal Token Fusion for Vision Transformers","date":"2022-04-19","rows_on_this_dataset":2,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/p3depth-monocular-depth-estimation-with-a","title":"P3Depth: Monocular Depth Estimation with a Piecewise Planarity Prior","date":"2022-04-05","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/multimae-multi-modal-multi-task-masked","title":"MultiMAE: Multi-modal Multi-task Masked Autoencoders","date":"2022-04-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/binsformer-revisiting-adaptive-bins-for","title":"BinsFormer: Revisiting Adaptive Bins for Monocular Depth Estimation","date":"2022-04-03","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":1,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dynamic-multimodal-fusion","title":"Dynamic Multimodal Fusion","date":"2022-03-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/localbins-improving-depth-estimation-by","title":"LocalBins: Improving Depth Estimation by Learning Local Distributions","date":"2022-03-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":11,"samples_unverified":3,"pointer_only_for_licence":14,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/depthformer-exploiting-long-range-correlation","title":"DepthFormer: Exploiting Long-Range Correlation and Local Information for Accurate Monocular Depth Estimation","date":"2022-03-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/inverted-pyramid-multi-task-transformer-for","title":"InvPT: Inverted Pyramid Multi-task Transformer for Dense Scene Understanding","date":"2022-03-15","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cmx-cross-modal-fusion-for-rgb-x-semantic","title":"CMX: Cross-Modal Fusion for RGB-X Semantic Segmentation with Transformers","date":"2022-03-09","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/new-crfs-neural-window-fully-connected-crfs-1","title":"NeW CRFs: Neural Window Fully-connected CRFs for Monocular Depth Estimation","date":"2022-03-03","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":5,"samples_unverified":5,"pointer_only_for_licence":10,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-task-learning-as-a-bargaining-game","title":"Multi-Task Learning as a Bargaining Game","date":"2022-02-02","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/omnivore-a-single-model-for-many-visual","title":"Omnivore: A Single Model for Many Visual Modalities","date":"2022-01-20","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/global-local-path-networks-for-monocular","title":"Global-Local Path Networks for Monocular Depth Estimation with Vertical CutDepth","date":"2022-01-19","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/attention-based-dual-supervised-decoder-for","title":"Attention-based Dual Supervised Decoder for RGBD Semantic Segmentation","date":"2022-01-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/not-all-voxels-are-equal-semantic-scene","title":"Not All Voxels Are Equal: Semantic Scene Completion from the Point-Voxel Perspective","date":"2021-12-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/nvs-monodepth-improving-monocular-depth","title":"NVS-MonoDepth: Improving Monocular Depth Prediction with Novel View Synthesis","date":"2021-12-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/toward-practical-self-supervised-monocular","title":"Toward Practical Monocular Indoor Depth Estimation","date":"2021-12-04","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/channel-exchanging-networks-for-multimodal","title":"Channel Exchanging Networks for Multimodal and Multitask Dense Image Prediction","date":"2021-12-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/monoscene-monocular-3d-semantic-scene","title":"MonoScene: Monocular 3D Semantic Scene Completion","date":"2021-12-01","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/data-augmented-3d-semantic-scene-completion","title":"Data Augmented 3D Semantic Scene Completion with 2D Segmentation Priors","date":"2021-11-26","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/cerberus-transformer-joint-semantic","title":"Cerberus Transformer: Joint Semantic, Affordance and Attribute Parsing","date":"2021-11-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hs3-learning-with-proper-task-complexity-in","title":"HS3: Learning with Proper Task Complexity in Hierarchically Supervised Semantic Segmentation","date":"2021-11-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/d-net-a-generalised-and-optimised-deep","title":"D-Net: A Generalised and Optimised Deep Network for Monocular Depth Estimation","date":"2021-09-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/warp-refine-propagation-semi-supervised-auto","title":"Warp-Refine Propagation: Semi-Supervised Auto-labeling via Cycle-consistency","date":"2021-09-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/estimating-and-exploiting-the-aleatoric","title":"Estimating and Exploiting the Aleatoric Uncertainty in Surface Normal Estimation","date":"2021-09-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":6,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/shapeconv-shape-aware-convolutional-layer-for","title":"ShapeConv: Shape-aware Convolutional Layer for Indoor RGB-D Semantic Segmentation","date":"2021-08-24","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/structdepth-leveraging-the-structural","title":"StructDepth: Leveraging the structural regularities for self-supervised indoor depth estimation","date":"2021-08-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-deep-multimodal-feature","title":"Learning Deep Multimodal Feature Representation with Asymmetric Multi-layer Fusion","date":"2021-08-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":0,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ci-net-contextual-information-for-joint","title":"CI-Net: Contextual Information for Joint Semantic Segmentation and Depth Estimation","date":"2021-07-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/monoindoor-towards-good-practice-of-self","title":"MonoIndoor: Towards Good Practice of Self-Supervised Monocular Depth Estimation for Indoor Environments","date":"2021-07-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cutdepth-edge-aware-data-augmentation-in","title":"CutDepth:Edge-aware Data Augmentation in Depth Estimation","date":"2021-07-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/contrastive-multimodal-fusion-with","title":"Contrastive Multimodal Fusion with TupleInfoNCE","date":"2021-07-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":0,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-multi-modal-learning-with-uni-modal","title":"Improving Multi-Modal Learning with Uni-Modal Teachers","date":"2021-06-21","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/unsupervised-scale-consistent-depth-learning","title":"Unsupervised Scale-consistent Depth Learning from Video","date":"2021-05-25","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/deep-feature-selection-and-fusion-for-rgb-d","title":"Deep feature selection-and-fusion for RGB-D semantic segmentation","date":"2021-05-10","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/exploring-relational-context-for-multi-task","title":"Exploring Relational Context for Multi-Task Dense Prediction","date":"2021-04-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/semantic-scene-completion-via-integrating","title":"Semantic Scene Completion via Integrating Instances and Scene in-the-Loop","date":"2021-04-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/inverseform-a-loss-function-for-structured","title":"InverseForm: A Loss Function for Structured Boundary-Aware Segmentation","date":"2021-04-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multimodal-knowledge-expansion","title":"Multimodal Knowledge Expansion","date":"2021-03-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vision-transformers-for-dense-prediction","title":"Vision Transformers for Dense Prediction","date":"2021-03-24","rows_on_this_dataset":1,"code_links":15,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":116,"samples_ran":54,"samples_unverified":62,"pointer_only_for_licence":15,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/transformers-solve-the-limited-receptive","title":"Transformer-Based Attention Networks for Continuous Pixel-Wise Prediction","date":"2021-03-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":7,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/3d-semantic-scene-completion-a-survey","title":"3D Semantic Scene Completion: a Survey","date":"2021-03-12","rows_on_this_dataset":5,"code_links":0,"syntology":null},{"paper":"/paper/sosd-net-joint-semantic-object-segmentation","title":"SOSD-Net: Joint Semantic Object Segmentation and Depth Estimation from Monocular images","date":"2021-01-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/monocular-depth-estimation-using-laplacian","title":"Monocular Depth Estimation Using Laplacian Pyramid-Based Depth Residuals","date":"2021-01-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rfnet-region-aware-fusion-network-for","title":"RFNet: Region-Aware Fusion Network for Incomplete Multi-Modal Brain Tumor Segmentation","date":"2021-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-to-recover-3d-scene-shape-from-a","title":"Learning to Recover 3D Scene Shape from a Single Image","date":"2020-12-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/adabins-depth-estimation-using-adaptive-bins","title":"AdaBins: Depth Estimation using Adaptive Bins","date":"2020-11-28","rows_on_this_dataset":2,"code_links":11,"syntology":null},{"paper":"/paper/efficient-rgb-d-semantic-segmentation-for","title":"Efficient RGB-D Semantic Segmentation for Indoor Scene Analysis","date":"2020-11-13","rows_on_this_dataset":2,"code_links":4,"syntology":null},{"paper":"/paper/multi-layer-feature-aggregation-for-deep","title":"Multi-layer Feature Aggregation for Deep Scene Parsing Models","date":"2020-11-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/on-deep-learning-techniques-to-boost","title":"On Deep Learning Techniques to Boost Monocular Depth Estimation for Autonomous Navigation","date":"2020-10-13","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/semantic-point-completion-network-for-3d","title":"Semantic Point Completion Network for 3D Semantic Scene Completion","date":"2020-08-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/lmscnet-lightweight-multiscale-3d-semantic","title":"LMSCNet: Lightweight Multiscale 3D Semantic Completion","date":"2020-08-24","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/non-local-spatial-propagation-network-for","title":"Non-Local Spatial Propagation Network for Depth Completion","date":"2020-07-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":1,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/malleable-2-5d-convolution-learning-receptive","title":"Malleable 2.5D Convolution: Learning Receptive Fields along the Depth-axis for RGB-D Scene Parsing","date":"2020-07-18","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":4,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bi-directional-cross-modality-feature","title":"Bi-directional Cross-Modality Feature Propagation with Separation-and-Aggregation Gate for RGB-D Semantic Segmentation","date":"2020-07-17","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/p-2-net-patch-match-and-plane-regularization","title":"P$^{2}$Net: Patch-match and Plane-regularization for Unsupervised Indoor Depth Estimation","date":"2020-07-15","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unsupervised-depth-learning-in-challenging","title":"Auto-Rectify Network for Unsupervised Indoor Depth Estimation","date":"2020-06-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/variational-context-deformable-convnets-for","title":"Variational Context-Deformable ConvNets for Indoor Scene Parsing","date":"2020-06-01","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/sdc-depth-semantic-divide-and-conquer-network","title":"SDC-Depth: Semantic Divide-and-Conquer Network for Monocular Depth Estimation","date":"2020-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/pattern-structure-diffusion-for-multi-task","title":"Pattern-Structure Diffusion for Multi-Task Learning","date":"2020-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/focus-on-defocus-bridging-the-synthetic-to","title":"Focus on defocus: bridging the synthetic to real domain gap for depth estimation","date":"2020-05-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/spatial-information-guided-convolution-for","title":"Spatial Information Guided Convolution for Real-Time RGBD Semantic Segmentation","date":"2020-04-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/anisotropic-convolutional-networks-for-3d","title":"Anisotropic Convolutional Networks for 3D Semantic Scene Completion","date":"2020-04-05","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/towards-better-generalization-joint-depth","title":"Towards Better Generalization: Joint Depth-Pose Learning without PoseNet","date":"2020-04-03","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/temporally-distributed-networks-for-fast","title":"Temporally Distributed Networks for Fast Video Semantic Segmentation","date":"2020-04-03","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/attention-based-multi-modal-fusion-network","title":"Attention-based Multi-modal Fusion Network for Semantic Scene Completion","date":"2020-03-31","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/3d-sketch-aware-semantic-scene-completion-via","title":"3D Sketch-aware Semantic Scene Completion via Semi-supervised Structure Prior","date":"2020-03-31","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/3d-gated-recurrent-fusion-for-semantic-scene","title":"3D Gated Recurrent Fusion for Semantic Scene Completion","date":"2020-02-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/rgb-based-semantic-segmentation-using-self","title":"RGB-based Semantic Segmentation Using Self-Supervised Depth Pre-Training","date":"2020-02-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/mti-net-multi-scale-task-interaction-networks","title":"MTI-Net: Multi-Scale Task Interaction Networks for Multi-Task Learning","date":"2020-01-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/single-image-depth-estimation-trained-via-1","title":"Single Image Depth Estimation Trained via Depth from Defocus Cues","date":"2020-01-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multi-modal-attention-based-fusion-model-for","title":"Multi-Modal Attention-based Fusion Model for Semantic Segmentation of RGB-Depth Images","date":"2019-12-25","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/adashare-learning-what-to-share-for-efficient","title":"AdaShare: Learning What To Share For Efficient Deep Multi-Task Learning","date":"2019-11-27","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/moving-indoor-unsupervised-video-depth","title":"Moving Indoor: Unsupervised Video Depth Learning in Challenging Environments","date":"2019-10-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/3d-ken-burns-effect-from-a-single-image","title":"3D Ken Burns Effect from a Single Image","date":"2019-09-12","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/structure-attentioned-memory-network-for","title":"Structure-Attentioned Memory Network for Monocular Depth Estimation","date":"2019-09-10","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/forknet-multi-branch-volumetric-semantic","title":"ForkNet: Multi-branch Volumetric Semantic Completion from a Single Depth Image","date":"2019-09-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a2j-anchor-to-joint-regression-network-for-3d","title":"A2J: Anchor-to-Joint Regression Network for 3D Articulated Pose Estimation from a Single Depth Image","date":"2019-08-27","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":6,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/index-network","title":"Index Network","date":"2019-08-11","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/edgenet-semantic-scene-completion-from-rgb-d","title":"EdgeNet: Semantic Scene Completion from a Single RGB-D Image","date":"2019-08-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/cascaded-context-pyramid-for-full-resolution","title":"Cascaded Context Pyramid for Full-Resolution 3D Semantic Scene Completion","date":"2019-08-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/enforcing-geometric-constraints-of-virtual","title":"Enforcing geometric constraints of virtual normal for depth prediction","date":"2019-07-29","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/from-big-to-small-multi-scale-local-planar","title":"From Big to Small: Multi-Scale Local Planar Guidance for Monocular Depth Estimation","date":"2019-07-24","rows_on_this_dataset":2,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":5,"samples_unverified":10,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/structure-aware-residual-pyramid-network-for","title":"Structure-Aware Residual Pyramid Network for Monocular Depth Estimation","date":"2019-07-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/efficient-semantic-scene-completion-network-1","title":"Efficient Semantic Scene Completion Network with Spatial Group Convolution","date":"2019-07-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/floors-are-flat-leveraging-semantics-for-real","title":"Floors are Flat: Leveraging Semantics for Real-Time Surface Normal Prediction","date":"2019-06-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/generating-and-exploiting-probabilistic","title":"Generating and Exploiting Probabilistic Monocular Depth Estimates","date":"2019-06-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pattern-affinitive-propagation-across-depth-1","title":"Pattern-Affinitive Propagation across Depth, Surface Normal and Semantic Segmentation","date":"2019-06-08","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/scene-parsing-via-integrated-classification","title":"Scene Parsing via Integrated Classification Model and Variance-Based Regularization","date":"2019-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/monocular-depth-estimation-using-relative","title":"Monocular Depth Estimation Using Relative Depth Maps","date":"2019-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/acnet-attention-based-network-to-exploit","title":"ACNet: Attention Based Network to Exploit Complementary Features for RGBD Semantic Segmentation","date":"2019-05-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-fully-dense-neural-networks-for","title":"Learning Fully Dense Neural Networks for Image Semantic Segmentation","date":"2019-05-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/190508598","title":"SharpNet: Fast and Accurate Recovery of Occluding Contours in Monocular Depth Estimation","date":"2019-05-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deep-optics-for-monocular-depth-estimation","title":"Deep Optics for Monocular Depth Estimation and 3D Object Detection","date":"2019-04-18","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/rgbd-based-dimensional-decomposition-residual","title":"RGBD Based Dimensional Decomposition Residual Network for 3D Semantic Scene Completion","date":"2019-03-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/attention-based-context-aggregation-network","title":"Attention-based Context Aggregation Network for Monocular Depth Estimation","date":"2019-01-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":0,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/high-quality-monocular-depth-estimation-via","title":"High Quality Monocular Depth Estimation via Transfer Learning","date":"2018-12-31","rows_on_this_dataset":1,"code_links":45,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":23,"samples_ran":5,"samples_unverified":18,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fast-neural-architecture-search-of-compact","title":"Fast Neural Architecture Search of Compact Semantic Segmentation Models via Auxiliary Cells","date":"2018-10-25","rows_on_this_dataset":3,"code_links":4,"syntology":null},{"paper":"/paper/light-weight-refinenet-for-real-time-semantic","title":"Light-Weight RefineNet for Real-Time Semantic Segmentation","date":"2018-10-08","rows_on_this_dataset":6,"code_links":2,"syntology":null},{"paper":"/paper/real-time-joint-semantic-segmentation-and","title":"Real-Time Joint Semantic Segmentation and Depth Estimation Using Asymmetric Annotations","date":"2018-09-13","rows_on_this_dataset":3,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":6,"samples_unverified":0,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/joint-task-recursive-learning-for-semantic","title":"Joint Task-Recursive Learning for Semantic Segmentation and Depth Estimation","date":"2018-09-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/view-volume-network-for-semantic-scene","title":"View-volume Network for Semantic Scene Completion from a Single Depth Image","date":"2018-06-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/deep-ordinal-regression-network-for-monocular","title":"Deep Ordinal Regression Network for Monocular Depth Estimation","date":"2018-06-06","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rednet-residual-encoder-decoder-network-for","title":"RedNet: Residual Encoder-Decoder Network for indoor RGB-D Semantic Segmentation","date":"2018-06-04","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dense-decoder-shortcut-connections-for-single","title":"Dense Decoder Shortcut Connections for Single-Pass Semantic Segmentation","date":"2018-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/pad-net-multi-tasks-guided-prediction-and","title":"PAD-Net: Multi-Tasks Guided Prediction-and-Distillation Network for Simultaneous Depth Estimation and Scene Parsing","date":"2018-05-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/two-stream-3d-semantic-scene-completion","title":"Two Stream 3D Semantic Scene Completion","date":"2018-04-10","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/revisiting-single-image-depth-estimation","title":"Revisiting Single Image Depth Estimation: Toward Higher Resolution Maps with Accurate Object Boundaries","date":"2018-03-23","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/depth-aware-cnn-for-rgb-d-segmentation","title":"Depth-aware CNN for RGB-D Segmentation","date":"2018-03-19","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":1,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/semantic-scene-completion-combining-colour","title":"Semantic Scene Completion Combining Colour and Depth: preliminary experiments","date":"2018-02-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/nddr-cnn-layer-wise-feature-fusing-in-multi","title":"NDDR-CNN: Layerwise Feature Fusing in Multi-Task CNNs by Neural Discriminative Dimensionality Reduction","date":"2018-01-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/sgpn-similarity-group-proposal-network-for-3d","title":"SGPN: Similarity Group Proposal Network for 3D Point Cloud Instance Segmentation","date":"2017-11-23","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/cascaded-feature-network-for-semantic","title":"Cascaded Feature Network for Semantic Segmentation of RGB-D Images","date":"2017-10-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/3d-graph-neural-networks-for-rgbd-semantic","title":"3D Graph Neural Networks for RGBD Semantic Segmentation","date":"2017-10-01","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/efficient-yet-deep-convolutional-neural","title":"Efficient Yet Deep Convolutional Neural Networks for Semantic Segmentation","date":"2017-07-26","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/semantic-segmentation-with-reverse-attention","title":"Semantic Segmentation with Reverse Attention","date":"2017-07-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/locality-sensitive-deconvolution-networks","title":"Locality-Sensitive Deconvolution Networks With Gated Fusion for RGB-D Indoor Semantic Segmentation","date":"2017-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/recurrent-scene-parsing-with-perspective","title":"Recurrent Scene Parsing with Perspective Understanding in the Loop","date":"2017-05-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multi-scale-continuous-crfs-as-sequential","title":"Multi-Scale Continuous CRFs as Sequential Deep Networks for Monocular Depth Estimation","date":"2017-04-07","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/what-uncertainties-do-we-need-in-bayesian","title":"What Uncertainties Do We Need in Bayesian Deep Learning for Computer Vision?","date":"2017-03-15","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pyramid-scene-parsing-network","title":"Pyramid Scene Parsing Network","date":"2016-12-04","rows_on_this_dataset":3,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":29,"samples_ran":7,"samples_unverified":22,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/semantic-scene-completion-from-a-single-depth","title":"Semantic Scene Completion from a Single Depth Image","date":"2016-11-28","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/refinenet-multi-path-refinement-networks-for","title":"RefineNet: Multi-Path Refinement Networks for High-Resolution Semantic Segmentation","date":"2016-11-20","rows_on_this_dataset":1,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":1,"samples_unverified":6,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hemis-hetero-modal-image-segmentation","title":"HeMIS: Hetero-Modal Image Segmentation","date":"2016-07-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-two-streamed-network-for-estimating-fine","title":"A Two-Streamed Network for Estimating Fine-Scaled Depth Maps from Single RGB Images","date":"2016-07-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/fully-convolutional-networks-for-semantic","title":"Fully Convolutional Networks for Semantic Segmentation","date":"2016-05-20","rows_on_this_dataset":1,"code_links":37,"syntology":null},{"paper":"/paper/cross-stitch-networks-for-multi-task-learning","title":"Cross-stitch Networks for Multi-task Learning","date":"2016-04-12","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/std2p-rgbd-semantic-segmentation-using-spatio","title":"STD2P: RGBD Semantic Segmentation Using Spatio-Temporal Data-Driven Pooling","date":"2016-04-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/predicting-depth-surface-normals-and-semantic","title":"Predicting Depth, Surface Normals and Semantic Labels with a Common Multi-Scale Convolutional Architecture","date":"2014-11-18","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":75,"samples_harvested":736,"samples_ran":347,"samples_unverified":389,"pointer_only_for_licence":186,"papers_with_no_sample_that_ran":14,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}