{"url":"/sota/depth-estimation-on-nyu-depth-v2","task":{"name":"Depth Estimation","url":"/task/depth-estimation","note":null},"dataset":{"name":"NYU-Depth V2","url":"/dataset/nyuv2"},"category":"Computer Vision","categories":["Computer Vision"],"category_note":null,"description":"**Depth Estimation** is the task of measuring the distance of each pixel relative to the camera. Depth is extracted from either monocular (single) or stereo (multiple views of a scene) images. Traditional methods use multi-view geometry to find the relationship between the images. Newer methods can directly estimate depth by minimizing the regression loss, or by learning to generate a novel view from a sequence. The most popular benchmarks are KITTI and NYUv2. Models are typically evaluated according to a RMS metric.\r\n\r\n<span class=\"description-source\">Source: [DIODE: A Dense Indoor and Outdoor DEpth Dataset ](https://arxiv.org/abs/1908.00463)</span>","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["RMS","RMSE","mAP"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"RMS":null,"RMSE":"lower","mAP":"higher"}},"counts":{"rows":17,"rows_with_code":14,"rows_with_paper_page":17,"rows_dated":17,"rows_using_additional_data":1},"rows":[{"rank_in_archive_order":1,"model":"EVP","metrics":{"RMS":"0.224"},"uses_additional_data":false,"paper_date":"2023-12-13","paper":"/paper/evp-enhanced-visual-perception-using-inverse","paper_url":"https://arxiv.org/abs/2312.08548v1","paper_title":"EVP: Enhanced Visual Perception using Inverse Multi-Attentive Feature Refinement and Regularized Image-Text Alignment","code":"https://github.com/lavreniuk/evp","n_code_links":1,"syntology":null},{"rank_in_archive_order":2,"model":"DINOv2 (ViT-g/14 frozen, w/ DPT decoder)","metrics":{"RMS":"0.279"},"uses_additional_data":true,"paper_date":"2023-04-14","paper":"/paper/dinov2-learning-robust-visual-features","paper_url":"https://arxiv.org/abs/2304.07193v2","paper_title":"DINOv2: Learning Robust Visual Features without Supervision","code":"https://github.com/huggingface/transformers","n_code_links":26,"syntology":{"n_ran":21,"n_unverified":25,"n_samples":46,"n_pointer_only_licence":12}},{"rank_in_archive_order":3,"model":"SwinV2-L 1K-MIM","metrics":{"RMS":"0.287"},"uses_additional_data":false,"paper_date":"2022-05-26","paper":"/paper/revealing-the-dark-secrets-of-masked-image","paper_url":"https://arxiv.org/abs/2205.13543v2","paper_title":"Revealing the Dark Secrets of Masked Image Modeling","code":"https://github.com/SwinTransformer/MIM-Depth-Estimation","n_code_links":1,"syntology":{"n_ran":4,"n_unverified":1,"n_samples":5,"n_pointer_only_licence":0}},{"rank_in_archive_order":4,"model":"Semantic-aware NN","metrics":{"RMS":"0.30"},"uses_additional_data":false,"paper_date":"2019-09-12","paper":"/paper/3d-ken-burns-effect-from-a-single-image","paper_url":"https://arxiv.org/abs/1909.05483v1","paper_title":"3D Ken Burns Effect from a Single Image","code":"https://github.com/sniklaus/3d-ken-burns","n_code_links":4,"syntology":{"n_ran":0,"n_unverified":2,"n_samples":2,"n_pointer_only_licence":0}},{"rank_in_archive_order":5,"model":"SwinV2-B 1K-MIM","metrics":{"RMS":"0.304"},"uses_additional_data":false,"paper_date":"2022-05-26","paper":"/paper/revealing-the-dark-secrets-of-masked-image","paper_url":"https://arxiv.org/abs/2205.13543v2","paper_title":"Revealing the Dark Secrets of Masked Image Modeling","code":"https://github.com/SwinTransformer/MIM-Depth-Estimation","n_code_links":1,"syntology":{"n_ran":4,"n_unverified":1,"n_samples":5,"n_pointer_only_licence":0}},{"rank_in_archive_order":6,"model":"P3Depth","metrics":{"RMS":"0.356"},"uses_additional_data":false,"paper_date":"2022-04-05","paper":"/paper/p3depth-monocular-depth-estimation-with-a","paper_url":"https://arxiv.org/abs/2204.02091v1","paper_title":"P3Depth: Monocular Depth Estimation with a Piecewise Planarity Prior","code":"https://github.com/syscv/p3depth","n_code_links":1,"syntology":null},{"rank_in_archive_order":7,"model":"AdaBins","metrics":{"RMS":"0.364"},"uses_additional_data":false,"paper_date":"2020-11-28","paper":"/paper/adabins-depth-estimation-using-adaptive-bins","paper_url":"https://arxiv.org/abs/2011.14141v1","paper_title":"AdaBins: Depth Estimation using Adaptive Bins","code":"https://github.com/shariqfarooq123/AdaBins","n_code_links":11,"syntology":null},{"rank_in_archive_order":8,"model":"TransDepth (AGD+ ViT)","metrics":{"RMS":"0.365"},"uses_additional_data":false,"paper_date":"2021-03-22","paper":"/paper/transformers-solve-the-limited-receptive","paper_url":"https://arxiv.org/abs/2103.12091v2","paper_title":"Transformer-Based Attention Networks for Continuous Pixel-Wise Prediction","code":"https://github.com/ygjwd12345/TransDepth","n_code_links":1,"syntology":{"n_ran":7,"n_unverified":4,"n_samples":11,"n_pointer_only_licence":0}},{"rank_in_archive_order":9,"model":"BTS","metrics":{"RMS":"0.407"},"uses_additional_data":false,"paper_date":"2019-07-24","paper":"/paper/from-big-to-small-multi-scale-local-planar","paper_url":"https://arxiv.org/abs/1907.10326v6","paper_title":"From Big to Small: Multi-Scale Local Planar Guidance for Monocular Depth Estimation","code":"https://github.com/cleinc/bts","n_code_links":14,"syntology":{"n_ran":5,"n_unverified":10,"n_samples":15,"n_pointer_only_licence":4}},{"rank_in_archive_order":10,"model":"VNL","metrics":{"RMS":"0.416"},"uses_additional_data":false,"paper_date":"2019-07-29","paper":"/paper/enforcing-geometric-constraints-of-virtual","paper_url":"https://arxiv.org/abs/1907.12209v2","paper_title":"Enforcing geometric constraints of virtual normal for depth prediction","code":"https://github.com/aim-uofa/AdelaiDepth","n_code_links":3,"syntology":null},{"rank_in_archive_order":11,"model":"Optimized, freeform","metrics":{"RMS":"0.4325"},"uses_additional_data":false,"paper_date":"2019-04-18","paper":"/paper/deep-optics-for-monocular-depth-estimation","paper_url":"http://arxiv.org/abs/1904.08601v1","paper_title":"Deep Optics for Monocular Depth Estimation and 3D Object Detection","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":12,"model":"Freeform","metrics":{"RMS":"0.433"},"uses_additional_data":false,"paper_date":"2019-04-18","paper":"/paper/deep-optics-for-monocular-depth-estimation","paper_url":"http://arxiv.org/abs/1904.08601v1","paper_title":"Deep Optics for Monocular Depth Estimation and 3D Object Detection","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":13,"model":"DORN","metrics":{"RMS":"0.509"},"uses_additional_data":false,"paper_date":"2018-06-06","paper":"/paper/deep-ordinal-regression-network-for-monocular","paper_url":"http://arxiv.org/abs/1806.02446v1","paper_title":"Deep Ordinal Regression Network for Monocular Depth Estimation","code":"https://github.com/hufu6371/DORN","n_code_links":5,"syntology":{"n_ran":0,"n_unverified":3,"n_samples":3,"n_pointer_only_licence":0}},{"rank_in_archive_order":14,"model":"MS-CRF","metrics":{"RMS":"0.586"},"uses_additional_data":false,"paper_date":"2017-04-07","paper":"/paper/multi-scale-continuous-crfs-as-sequential","paper_url":"http://arxiv.org/abs/1704.02157v1","paper_title":"Multi-Scale Continuous CRFs as Sequential Deep Networks for Monocular Depth Estimation","code":"https://github.com/danxuhk/ContinuousCRF-CNN","n_code_links":2,"syntology":null},{"rank_in_archive_order":15,"model":"PAD-Net","metrics":{"RMS":"0.792"},"uses_additional_data":false,"paper_date":"2018-05-11","paper":"/paper/pad-net-multi-tasks-guided-prediction-and","paper_url":"http://arxiv.org/abs/1805.04409v1","paper_title":"PAD-Net: Multi-Tasks Guided Prediction-and-Distillation Network for Simultaneous Depth Estimation and Scene Parsing","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":16,"model":"Defocus/DepthNet (Normalized)","metrics":{"RMSE":"0.013"},"uses_additional_data":false,"paper_date":"2020-05-19","paper":"/paper/focus-on-defocus-bridging-the-synthetic-to","paper_url":"https://arxiv.org/abs/2005.09623v1","paper_title":"Focus on defocus: bridging the synthetic to real domain gap for depth estimation","code":"https://github.com/dvl-tum/defocus-net","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":4,"n_samples":4,"n_pointer_only_licence":0}},{"rank_in_archive_order":17,"model":"A2J","metrics":{"mAP":"8.61"},"uses_additional_data":false,"paper_date":"2019-08-27","paper":"/paper/a2j-anchor-to-joint-regression-network-for-3d","paper_url":"https://arxiv.org/abs/1908.09999v1","paper_title":"A2J: Anchor-to-Joint Regression Network for 3D Articulated Pose Estimation from a Single Depth Image","code":"https://github.com/zhangboshen/A2J","n_code_links":2,"syntology":{"n_ran":6,"n_unverified":3,"n_samples":9,"n_pointer_only_licence":0}}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":9,"rows_with_any_sample_ran":6,"distinct_papers_with_graph_line":8,"distinct_papers_with_any_sample_ran":5,"samples_over_distinct_papers":{"n_ran":43,"n_unverified":52,"n_samples":95,"n_pointer_only_licence":16,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":47,"n_unverified":53,"n_samples":100,"n_pointer_only_licence":16,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}