{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/learning-to-recover-3d-scene-shape-from-a","title":"Learning to Recover 3D Scene Shape from a Single Image","arxiv_id":"2012.09365","date":"2020-12-17","proceeding":"CVPR 2021 1","authors":["Wei Yin","Jianming Zhang","Oliver Wang","Simon Niklaus","Long Mai","Simon Chen","Chunhua Shen"],"abstract":"Despite significant progress in monocular depth estimation in the wild, recent state-of-the-art methods cannot be used to recover accurate 3D scene shape due to an unknown depth shift induced by shift-invariant reconstruction losses used in mixed-data depth prediction training, and possible unknown camera focal length. We investigate this problem in detail, and propose a two-stage framework that first predicts depth up to an unknown scale and shift from a single monocular image, and then use 3D point cloud encoders to predict the missing depth shift and focal length that allow us to recover a realistic 3D scene shape. In addition, we propose an image-level normalized regression loss and a normal-based geometry loss to enhance depth prediction models trained on mixed datasets. We test our depth model on nine unseen datasets and achieve state-of-the-art performance on zero-shot dataset generalization. Code is available at: https://git.io/Depth","url_abs":"https://arxiv.org/abs/2012.09365v1","url_pdf":"https://arxiv.org/pdf/2012.09365v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"learning-to-recover-3d-scene-shape-from-a","repo_url":"https://github.com/aim-uofa/AdelaiDepth","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}}],"tasks":[{"task_slug":"3d-scene-reconstruction","task_name":"3D Scene Reconstruction"},{"task_slug":"depth-estimation","task_name":"Depth Estimation"},{"task_slug":"depth-prediction","task_name":"Depth Prediction"},{"task_slug":"indoor-monocular-depth-estimation","task_name":"Indoor Monocular Depth Estimation"},{"task_slug":"monocular-depth-estimation","task_name":"Monocular Depth Estimation"},{"task_slug":"single-view-3d-reconstruction","task_name":"Single-View 3D Reconstruction"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/depth-estimation-on-diode","task":"Depth Estimation","dataset":"DIODE","model":"LeRes","rank_in_archive_order":2,"of":2,"metrics":{"Delta < 1.25":"0.234"},"uses_additional_data":false},{"leaderboard":"/sota/depth-estimation-on-scannetv2","task":"Depth Estimation","dataset":"ScanNetV2","model":"LeReS","rank_in_archive_order":3,"of":3,"metrics":{"absolute relative error":"0.095"},"uses_additional_data":false},{"leaderboard":"/sota/indoor-monocular-depth-estimation-on-diode","task":"Indoor Monocular Depth Estimation","dataset":"DIODE","model":"LeReS","rank_in_archive_order":1,"of":2,"metrics":{"Delta < 1.25^3":"0.900"},"uses_additional_data":true},{"leaderboard":"/sota/monocular-depth-estimation-on-eth3d","task":"Monocular Depth Estimation","dataset":"ETH3D","model":"LeReS","rank_in_archive_order":9,"of":10,"metrics":{"Delta < 1.25":"0.0777","absolute relative error":"0.0171"},"uses_additional_data":false},{"leaderboard":"/sota/monocular-depth-estimation-on-kitti-eigen","task":"Monocular Depth Estimation","dataset":"KITTI Eigen split","model":"LeReS","rank_in_archive_order":77,"of":79,"metrics":{"Delta < 1.25":"0.784","absolute relative error":"0.149"},"uses_additional_data":false},{"leaderboard":"/sota/monocular-depth-estimation-on-nyu-depth-v2","task":"Monocular Depth Estimation","dataset":"NYU-Depth V2","model":"LeReS","rank_in_archive_order":39,"of":85,"metrics":{"Delta < 1.25":"0.916","absolute relative error":"0.09"},"uses_additional_data":true}],"syntology":{"syntology_url":"https://syntology.ai/paper/2012.09365","atlas_url":"https://app.syntology.ai/?focus=2012.09365","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}