{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/nimbled-enhancing-self-supervised-monocular","title":"NimbleD: Enhancing Self-supervised Monocular Depth Estimation with Pseudo-labels and Large-scale Video Pre-training","arxiv_id":"2408.14177","date":"2024-08-26","proceeding":null,"authors":["Albert Luginov","Muhammad Shahzad"],"abstract":"We introduce NimbleD, an efficient self-supervised monocular depth estimation learning framework that incorporates supervision from pseudo-labels generated by a large vision model. This framework does not require camera intrinsics, enabling large-scale pre-training on publicly available videos. Our straightforward yet effective learning strategy significantly enhances the performance of fast and lightweight models without introducing any overhead, allowing them to achieve performance comparable to state-of-the-art self-supervised monocular depth estimation models. This advancement is particularly beneficial for virtual and augmented reality applications requiring low latency inference. The source code, model weights, and acknowledgments are available at https://github.com/xapaxca/nimbled .","url_abs":"https://arxiv.org/abs/2408.14177v1","url_pdf":"https://arxiv.org/pdf/2408.14177v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"nimbled-enhancing-self-supervised-monocular","repo_url":"https://github.com/xapaxca/nimbled","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"depth-estimation","task_name":"Depth Estimation"},{"task_slug":"monocular-depth-estimation","task_name":"Monocular Depth Estimation"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/monocular-depth-estimation-on-kitti-eigen-1","task":"Monocular Depth Estimation","dataset":"KITTI Eigen split unsupervised","model":"NimbleD-LiteMono-8M","rank_in_archive_order":9,"of":55,"metrics":{"Delta < 1.25":"0.910","Delta < 1.25^2":"0.970","Delta < 1.25^3":"0.986","Mono":"O","RMSE":"4.194","RMSE log":"0.165","Resolution":"640x192","Sq Rel":"0.646","absolute relative error":"0.092"},"uses_additional_data":false},{"leaderboard":"/sota/monocular-depth-estimation-on-kitti-eigen-1","task":"Monocular Depth Estimation","dataset":"KITTI Eigen split unsupervised","model":"NimbleD-LiteMono","rank_in_archive_order":18,"of":55,"metrics":{"Delta < 1.25":"0.903","Delta < 1.25^2":"0.969","Delta < 1.25^3":"0.986","Mono":"O","RMSE":"4.304","RMSE log":"0.171","Resolution":"640x192","Sq Rel":"0.684","absolute relative error":"0.096"},"uses_additional_data":false},{"leaderboard":"/sota/monocular-depth-estimation-on-kitti-eigen-1","task":"Monocular Depth Estimation","dataset":"KITTI Eigen split unsupervised","model":"Nimbled-SwiftDepth","rank_in_archive_order":19,"of":55,"metrics":{"Delta < 1.25":"0.905","Delta < 1.25^2":"0.969","Delta < 1.25^3":"0.986","Mono":"O","RMSE":"4.333","RMSE log":"0.171","Resolution":"640x192","Sq Rel":"0.697","absolute relative error":"0.096"},"uses_additional_data":false},{"leaderboard":"/sota/monocular-depth-estimation-on-kitti-eigen-1","task":"Monocular Depth Estimation","dataset":"KITTI Eigen split unsupervised","model":"Nimbled-MD2-R50","rank_in_archive_order":21,"of":55,"metrics":{"Delta < 1.25":"0.904","Delta < 1.25^2":"0.968","Delta < 1.25^3":"0.985","Mono":"O","RMSE":"4.377","RMSE log":"0.172","Resolution":"640x192","Sq Rel":"0.721","absolute relative error":"0.097"},"uses_additional_data":false},{"leaderboard":"/sota/monocular-depth-estimation-on-kitti-eigen-1","task":"Monocular Depth Estimation","dataset":"KITTI Eigen split unsupervised","model":"Nimbled-SwiftDepth-S","rank_in_archive_order":23,"of":55,"metrics":{"Delta < 1.25":"0.901","Delta < 1.25^2":"0.968","Delta < 1.25^3":"0.985","Mono":"O","RMSE":"4.401","RMSE log":"0.174","Resolution":"640x192","Sq Rel":"0.733","absolute relative error":"0.098"},"uses_additional_data":false},{"leaderboard":"/sota/monocular-depth-estimation-on-kitti-eigen-1","task":"Monocular Depth Estimation","dataset":"KITTI Eigen split unsupervised","model":"NimbleD-LiteMono-S","rank_in_archive_order":25,"of":55,"metrics":{"Delta < 1.25":"0.898","Delta < 1.25^2":"0.967","Delta < 1.25^3":"0.986","Mono":"O","RMSE":"4.370","RMSE log":"0.172","Resolution":"640x192","Sq Rel":"0.709","absolute relative error":"0.099"},"uses_additional_data":false},{"leaderboard":"/sota/monocular-depth-estimation-on-kitti-eigen-1","task":"Monocular Depth Estimation","dataset":"KITTI Eigen split unsupervised","model":"Nimbled-MD2-R18","rank_in_archive_order":29,"of":55,"metrics":{"Delta < 1.25":"0.898","Delta < 1.25^2":"0.967","Delta < 1.25^3":"0.985","Mono":"O","RMSE":"4.440","RMSE log":"0.175","Resolution":"640x192","Sq Rel":"0.739","absolute relative error":"0.100"},"uses_additional_data":false}],"syntology":{"syntology_url":null,"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}