{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/objectron-a-large-scale-dataset-of-object","title":"Objectron: A Large Scale Dataset of Object-Centric Videos in the Wild with Pose Annotations","arxiv_id":"2012.09988","date":"2020-12-18","proceeding":"CVPR 2021 1","authors":["Adel Ahmadyan","Liangkai Zhang","Jianing Wei","Artsiom Ablavatski","Matthias Grundmann"],"abstract":"3D object detection has recently become popular due to many applications in robotics, augmented reality, autonomy, and image retrieval. We introduce the Objectron dataset to advance the state of the art in 3D object detection and foster new research and applications, such as 3D object tracking, view synthesis, and improved 3D shape representation. The dataset contains object-centric short videos with pose annotations for nine categories and includes 4 million annotated images in 14,819 annotated videos. We also propose a new evaluation metric, 3D Intersection over Union, for 3D object detection. We demonstrate the usefulness of our dataset in 3D object detection tasks by providing baseline models trained on this dataset. Our dataset and evaluation source code are available online at http://www.objectron.dev","url_abs":"https://arxiv.org/abs/2012.09988v1","url_pdf":"https://arxiv.org/pdf/2012.09988v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"objectron-a-large-scale-dataset-of-object","repo_url":"https://github.com/google-research-datasets/Objectron","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"tf","reach":{"status":"unanswered"}}],"tasks":[{"task_slug":"3d-object-detection","task_name":"3D Object Detection"},{"task_slug":"3d-object-tracking","task_name":"3D Object Tracking"},{"task_slug":"3d-shape-representation","task_name":"3D Shape Representation"},{"task_slug":"image-retrieval","task_name":"Image Retrieval"},{"task_slug":"monocular-3d-object-detection","task_name":"Monocular 3D Object Detection"},{"task_slug":"object","task_name":"Object"},{"task_slug":"object-detection","task_name":"Object Detection"},{"task_slug":"object-tracking","task_name":"Object Tracking"},{"task_slug":"retrieval","task_name":"Retrieval"},{"task_slug":"object-detection-1","task_name":"object-detection"}],"methods":[{"method_slug":"1x1-convolution","method_name":"1x1 Convolution"},{"method_slug":"average-pooling","method_name":"Average Pooling"},{"method_slug":"batch-normalization","method_name":"Batch Normalization"},{"method_slug":"convolution","method_name":"Convolution"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"depthwise-convolution","method_name":"Depthwise Convolution"},{"method_slug":"depthwise-separable-convolution","method_name":"Depthwise Separable Convolution"},{"method_slug":"dropout","method_name":"Dropout"},{"method_slug":"efficientnet","method_name":"EfficientNet"},{"method_slug":"inverted-residual-block","method_name":"Inverted Residual Block"},{"method_slug":"pointwise-convolution","method_name":"Pointwise Convolution"},{"method_slug":"rmsprop","method_name":"RMSProp"},{"method_slug":"relu","method_name":"ReLU"},{"method_slug":"sigmoid-activation","method_name":"Sigmoid Activation"},{"method_slug":"squeeze-and-excitation-block","method_name":"Squeeze-and-Excitation Block"}],"datasets_introduced":[{"slug":"objectron","name":"Objectron","full_name":null}],"methods_introduced":[],"results":[{"leaderboard":"/sota/monocular-3d-object-detection-on-google","task":"Monocular 3D Object Detection","dataset":"Google Objectron","model":"EfficientNetLite + keypoint regressor","rank_in_archive_order":2,"of":3,"metrics":{"AP at 10' Elevation error":"0.8584","AP at 15' Azimuth error":"0.7844","Average Precision at 0.5 3D IoU":"0.6512","MPE":"0.0467"},"uses_additional_data":false}],"syntology":{"syntology_url":"https://syntology.ai/paper/2012.09988","atlas_url":"https://app.syntology.ai/?focus=2012.09988","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}