{"url":"/task/3d-object-detection-from-monocular-images","name":"3D Object Detection From Monocular Images","slug":"3d-object-detection-from-monocular-images","description_markdown":"This is the task of detecting 3D objects from monocular images (as opposed to LiDAR based counterparts). It is usually associated with autonomous driving based tasks.\r\n\r\n<span style=\"color:grey; opacity: 0.6\">( Image credit: [Orthographic Feature Transform for Monocular 3D Object Detection](https://arxiv.org/pdf/1811.08188v1.pdf) )</span>","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":17,"papers_with_code":12,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/3d-object-detection-from-monocular-images-on-7","slug":"3d-object-detection-from-monocular-images-on-7","dataset":"KITTI-360","dataset_url":"/dataset/kitti-360","rows_in_archive":11,"metrics":["AP50","AP25"],"first_row_in_archive_order":{"model":"SeaBird + PanopticBEV","paper_title":"SeaBird: Segmentation in Bird's View with Dice Loss Improves Monocular 3D Detection of Large Objects","paper_url":"/paper/seabird-segmentation-in-bird-s-view-with-dice","paper_date":"2024-03-29","arxiv_id":"2403.20318","code_links":[{"title":"abhi1kumar/seabird","url":"https://github.com/abhi1kumar/seabird"}],"syntology":null}},{"leaderboard":"/sota/3d-object-detection-from-monocular-images-on-6","slug":"3d-object-detection-from-monocular-images-on-6","dataset":"Waymo Open Dataset","dataset_url":"/dataset/waymo-open-dataset","rows_in_archive":3,"metrics":["3D mAPH Vehicle (Front Camera Only)"],"first_row_in_archive_order":{"model":"DEVIANT","paper_title":"DEVIANT: Depth EquiVarIAnt NeTwork for Monocular 3D Object Detection","paper_url":"/paper/deviant-depth-equivariant-network-for","paper_date":"2022-07-21","arxiv_id":"2207.10758","code_links":[{"title":"abhi1kumar/deviant","url":"https://github.com/abhi1kumar/deviant"},{"title":"abhi1kumar/seabird","url":"https://github.com/abhi1kumar/seabird"}],"syntology":{"n":15,"n_ran":9,"n_unverified":6,"n_pointer_only":0}}},{"leaderboard":"/sota/3d-object-detection-from-monocular-images-on-4","slug":"3d-object-detection-from-monocular-images-on-4","dataset":"nuScenes Cars","dataset_url":null,"rows_in_archive":1,"metrics":["AP 2.0m","AP 0.5m","AP 1.0m","AP 4.0m","ATE","ASE","AOE"],"first_row_in_archive_order":{"model":"MonoDIS","paper_title":"Disentangling Monocular 3D Object Detection","paper_url":"/paper/disentangling-monocular-3d-object-detection","paper_date":"2019-05-29","arxiv_id":"1905.12365","code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/waymo-open-dataset","name":"Waymo Open Dataset","full_name":"","num_papers_in_archive":481},{"url":"/dataset/kitti-360","name":"KITTI-360","full_name":"","num_papers_in_archive":246},{"url":"/dataset/3d-pop","name":"3D-POP","full_name":"","num_papers_in_archive":3}],"subtasks":[],"parent_tasks":[{"url":"/task/object-detection","name":"Object Detection"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":12,"of":12,"tagged_in_all":17,"items":[{"url":"/paper/deep-hough-voting-for-3d-object-detection-in","title":"Deep Hough Voting for 3D Object Detection in Point Clouds","date":"2019-04-21","arxiv_id":"1904.09664","repositories_listed":13,"syntology":{"n":10,"n_ran":2,"n_unverified":8,"n_pointer_only":10}},{"url":"/paper/m3d-rpn-monocular-3d-region-proposal-network","title":"M3D-RPN: Monocular 3D Region Proposal Network for Object Detection","date":"2019-07-13","arxiv_id":"1907.06038","repositories_listed":4,"syntology":{"n":21,"n_ran":1,"n_unverified":20,"n_pointer_only":0}},{"url":"/paper/deviant-depth-equivariant-network-for","title":"DEVIANT: Depth EquiVarIAnt NeTwork for Monocular 3D Object Detection","date":"2022-07-21","arxiv_id":"2207.10758","repositories_listed":2,"syntology":{"n":15,"n_ran":9,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/seabird-segmentation-in-bird-s-view-with-dice","title":"SeaBird: Segmentation in Bird's View with Dice Loss Improves Monocular 3D Detection of Large Objects","date":"2024-03-29","arxiv_id":"2403.20318","repositories_listed":1,"syntology":null},{"url":"/paper/omni3d-a-large-benchmark-and-model-for-3d","title":"Omni3D: A Large Benchmark and Model for 3D Object Detection in the Wild","date":"2022-07-21","arxiv_id":"2207.10660","repositories_listed":1,"syntology":null},{"url":"/paper/monodetr-depth-aware-transformer-for","title":"MonoDETR: Depth-guided Transformer for Monocular 3D Object Detection","date":"2022-03-24","arxiv_id":"2203.13310","repositories_listed":1,"syntology":null},{"url":"/paper/monodtr-monocular-3d-object-detection-with","title":"MonoDTR: Monocular 3D Object Detection with Depth-Aware Transformer","date":"2022-03-21","arxiv_id":"2203.10981","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/roca-robust-cad-model-retrieval-and-alignment","title":"ROCA: Robust CAD Model Retrieval and Alignment from a Single Image","date":"2021-12-03","arxiv_id":"2112.01988","repositories_listed":1,"syntology":null},{"url":"/paper/geometry-uncertainty-projection-network-for","title":"Geometry Uncertainty Projection Network for Monocular 3D Object Detection","date":"2021-07-29","arxiv_id":"2107.13774","repositories_listed":1,"syntology":{"n":21,"n_ran":8,"n_unverified":13,"n_pointer_only":0}},{"url":"/paper/groomed-nms-grouped-mathematically","title":"GrooMeD-NMS: Grouped Mathematically Differentiable NMS for Monocular 3D Object Detection","date":"2021-03-31","arxiv_id":"2103.17202","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/delving-into-localization-errors-for","title":"Delving into Localization Errors for Monocular 3D Object Detection","date":"2021-03-30","arxiv_id":"2103.16237","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/orthographic-feature-transform-for-monocular","title":"Orthographic Feature Transform for Monocular 3D Object Detection","date":"2018-11-20","arxiv_id":"1811.08188","repositories_listed":1,"syntology":{"n":14,"n_ran":2,"n_unverified":12,"n_pointer_only":0}}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}