{"url":"/dataset/dair-v2x","name":"DAIR-V2X","full_name":null,"description_markdown":"**DAIR-V2X** is a  large-scale, multi-modality, multi-view dataset from real scenarios for VICAD. DAIR-V2X comprises 71254 LiDAR frames and 71254 Camera frames, and all frames are captured from real scenes with 3D annotations.","description_withheld":null,"homepage":"https://github.com/AIR-THU/DAIR-V2X","introduced_date":"2022-04-12","introduced_date_note":null,"introduced_by":{"paper":"/paper/dair-v2x-a-large-scale-dataset-for-vehicle","title":"DAIR-V2X: A Large-Scale Dataset for Vehicle-Infrastructure Cooperative 3D Object Detection","first_author":"Haibao Yu","url":null},"license":{"name":"Apache-2.0 license","url":"https://github.com/AIR-THU/DAIR-V2X/blob/main/LICENSE"},"modalities":[],"tasks":[{"name":"3D Object Detection","url":"/task/3d-object-detection","datasets_with_task":"/datasets/task/3d-object-detection"}],"languages":[],"variants":["DAIR-V2X","DAIR-V2X-I"],"data_loaders":[],"num_papers_in_archive":38,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/3d-object-detection-on-dair-v2x-i","task":"3D Object Detection","dataset_variant":"DAIR-V2X-I","rows":9,"metrics":["AP|R40(moderate)","AP|R40(easy)","AP|R40(hard)"],"first_row_in_archive_order":{"model":"MonoUNI","paper":"/paper/monouni-a-unified-vehicle-and-infrastructure","metrics":{"AP|R40(easy)":"90.92","AP|R40(hard)":"87.2","AP|R40(moderate)":"87.2"},"code_links":[{"title":"Traffic-X/MonoUNI","url":"https://github.com/Traffic-X/MonoUNI"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/3d-object-detection-on-dair-v2x","task":"3D Object Detection","dataset_variant":"DAIR-V2X","rows":2,"metrics":["AP50"],"first_row_in_archive_order":{"model":"CoBEVFlow","paper":null,"metrics":{"AP50":"80.7"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/cobev-elevating-roadside-3d-object-detection","title":"CoBEV: Elevating Roadside 3D Object Detection with Depth and Height Complementarity","date":"2023-10-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/monouni-a-unified-vehicle-and-infrastructure","title":"MonoUNI: A Unified Vehicle and Infrastructure-side Monocular 3D Object Detection Network with Sufficient Depth Clues","date":"2023-09-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bevheight-a-robust-framework-for-vision-based","title":"BEVHeight: A Robust Framework for Vision-based Roadside 3D Object Detection","date":"2023-03-15","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":15,"samples_ran":12,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/calibration-free-bev-representation-for","title":"Calibration-free BEV Representation for Infrastructure Perception","date":"2023-03-07","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/where2comm-communication-efficient","title":"Where2comm: Communication-Efficient Collaborative Perception via Spatial Confidence Maps","date":"2022-09-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bevdepth-acquisition-of-reliable-depth-for","title":"BEVDepth: Acquisition of Reliable Depth for Multi-view 3D Object Detection","date":"2022-06-21","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/bevformer-learning-bird-s-eye-view","title":"BEVFormer: Learning Bird's-Eye-View Representation from Multi-Camera Images via Spatiotemporal Transformers","date":"2022-03-31","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/imvoxelnet-image-to-voxels-projection-for","title":"ImVoxelNet: Image to Voxels Projection for Monocular and Multi-View General-Purpose 3D Object Detection","date":"2021-06-02","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mvx-net-multimodal-voxelnet-for-3d-object","title":"MVX-Net: Multimodal VoxelNet for 3D Object Detection","date":"2019-04-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pointpillars-fast-encoders-for-object","title":"PointPillars: Fast Encoders for Object Detection from Point Clouds","date":"2018-12-14","rows_on_this_dataset":1,"code_links":18,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":15,"samples_ran":2,"samples_unverified":13,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":5,"samples_harvested":40,"samples_ran":22,"samples_unverified":18,"pointer_only_for_licence":8,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}