{"url":"/task/autonomous-driving","name":"Autonomous Driving","slug":"autonomous-driving","description_markdown":"Autonomous driving is the task of driving a vehicle without human conduction. \r\n\r\nMany of the state-of-the-art results can be found at more general task pages such as [3D Object Detection](https://paperswithcode.com/task/3d-object-detection) and [Semantic Segmentation](https://paperswithcode.com/task/semantic-segmentation).\r\n\r\n<span style=\"color:grey; opacity: 0.5\">(Image credit: [Exploring the Limitations of Behavior Cloning for Autonomous Driving](https://arxiv.org/pdf/1904.08980v1.pdf))</span>","categories":[{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Miscellaneous","url":"/area/miscellaneous"},{"name":"Robots","url":"/area/robots"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":6092,"papers_with_code":2091,"benchmarks":4,"benchmark_tables_in_archive":4,"benchmark_tables_shown":4,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":70,"subtasks":6,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/autonomous-driving-on-carla-leaderboard","slug":"autonomous-driving-on-carla-leaderboard","dataset":"CARLA Leaderboard","dataset_url":"/dataset/carla","rows_in_archive":18,"metrics":["Driving Score","Route Completion","Infraction penalty"],"first_row_in_archive_order":{"model":"ReasonNet","paper_title":"ReasonNet: End-to-End Driving with Temporal and Global Reasoning","paper_url":"/paper/reasonnet-end-to-end-driving-with-temporal-1","paper_date":"2023-05-17","arxiv_id":"2305.10507","code_links":[],"syntology":null}},{"leaderboard":"/sota/autonomous-driving-on-town05-long","slug":"autonomous-driving-on-town05-long","dataset":"Town05 Long","dataset_url":null,"rows_in_archive":2,"metrics":["RC","DS"],"first_row_in_archive_order":{"model":"Geometric Fusion","paper_title":"Multi-Modal Fusion Transformer for End-to-End Autonomous Driving","paper_url":"/paper/multi-modal-fusion-transformer-for-end-to-end","paper_date":"2021-04-19","arxiv_id":"2104.09224","code_links":[{"title":"autonomousvision/transfuser","url":"https://github.com/autonomousvision/transfuser"},{"title":"Kin-Zhang/mmfn","url":"https://github.com/Kin-Zhang/mmfn"}],"syntology":{"n":15,"n_ran":9,"n_unverified":6,"n_pointer_only":0}}},{"leaderboard":"/sota/autonomous-driving-on-town05-short","slug":"autonomous-driving-on-town05-short","dataset":"Town05 Short","dataset_url":null,"rows_in_archive":2,"metrics":["RC","DS"],"first_row_in_archive_order":{"model":"Geometric Fusion","paper_title":"Multi-Modal Fusion Transformer for End-to-End Autonomous Driving","paper_url":"/paper/multi-modal-fusion-transformer-for-end-to-end","paper_date":"2021-04-19","arxiv_id":"2104.09224","code_links":[{"title":"autonomousvision/transfuser","url":"https://github.com/autonomousvision/transfuser"},{"title":"Kin-Zhang/mmfn","url":"https://github.com/Kin-Zhang/mmfn"}],"syntology":{"n":15,"n_ran":9,"n_unverified":6,"n_pointer_only":0}}},{"leaderboard":"/sota/autonomous-driving-on-apollocar3d","slug":"autonomous-driving-on-apollocar3d","dataset":"ApolloCar3D","dataset_url":"/dataset/apollocar3d","rows_in_archive":1,"metrics":["A3DP"],"first_row_in_archive_order":{"model":"GSNet","paper_title":"GSNet: Joint Vehicle Pose and Shape Reconstruction with Geometrical and Scene-aware Supervision","paper_url":"/paper/gsnet-joint-vehicle-pose-and-shape","paper_date":"2020-07-26","arxiv_id":"2007.13124","code_links":[{"title":"lkeab/gsnet","url":"https://github.com/lkeab/gsnet"}],"syntology":null}}],"datasets":[{"url":"/dataset/carla","name":"CARLA","full_name":"Car Learning to Act","num_papers_in_archive":1345},{"url":"/dataset/waymo-open-dataset","name":"Waymo Open Dataset","full_name":"","num_papers_in_archive":481},{"url":"/dataset/airsim","name":"AirSim","full_name":"","num_papers_in_archive":285},{"url":"/dataset/virtual-kitti","name":"Virtual KITTI","full_name":"","num_papers_in_archive":133},{"url":"/dataset/idd","name":"IDD","full_name":"Indian Driving Dataset","num_papers_in_archive":98},{"url":"/dataset/torcs","name":"TORCS","full_name":"The Open Racing Car Simulator","num_papers_in_archive":96},{"url":"/dataset/interaction-dataset","name":"INTERACTION Dataset","full_name":"","num_papers_in_archive":81},{"url":"/dataset/apolloscape-1","name":"ApolloScape","full_name":"","num_papers_in_archive":74},{"url":"/dataset/semanticposs","name":"SemanticPOSS","full_name":"","num_papers_in_archive":71},{"url":"/dataset/lost-and-found","name":"Lost and Found","full_name":"","num_papers_in_archive":57},{"url":"/dataset/woodscape","name":"WoodScape","full_name":"","num_papers_in_archive":55},{"url":"/dataset/pandaset","name":"PandaSet","full_name":"","num_papers_in_archive":54},{"url":"/dataset/uavid","name":"UAVid","full_name":"","num_papers_in_archive":54},{"url":"/dataset/drivingstereo","name":"DrivingStereo","full_name":"","num_papers_in_archive":50},{"url":"/dataset/bdd-x","name":"BDD-X","full_name":"Berkeley Deep Drive-X (eXplanation)","num_papers_in_archive":47},{"url":"/dataset/wilddash","name":"WildDash","full_name":"","num_papers_in_archive":47},{"url":"/dataset/talk2car","name":"Talk2Car","full_name":"","num_papers_in_archive":45},{"url":"/dataset/ddd17","name":"DDD17","full_name":"DAVIS Driving Dataset 2017","num_papers_in_archive":41},{"url":"/dataset/kitti-road","name":"KITTI Road","full_name":"","num_papers_in_archive":41},{"url":"/dataset/argoverse-2-motion-forecasting","name":"Argoverse 2 Motion Forecasting","full_name":"","num_papers_in_archive":39},{"url":"/dataset/h3d","name":"H3D","full_name":"Honda Research Institute 3D","num_papers_in_archive":39},{"url":"/dataset/hdd","name":"HDD","full_name":"Honda Research Institute Driving Dataset","num_papers_in_archive":38},{"url":"/dataset/a-3d","name":"A*3D","full_name":null,"num_papers_in_archive":37},{"url":"/dataset/dr-eye-ve","name":"DR(eye)VE","full_name":"","num_papers_in_archive":34},{"url":"/dataset/road","name":"ROAD","full_name":"ROAD: The ROad event Awareness Dataset for Autonomous Driving","num_papers_in_archive":27},{"url":"/dataset/toronto-3d","name":"Toronto-3D","full_name":null,"num_papers_in_archive":24},{"url":"/dataset/carrada","name":"CARRADA","full_name":"","num_papers_in_archive":22},{"url":"/dataset/apollocar3d","name":"ApolloCar3D","full_name":"","num_papers_in_archive":17},{"url":"/dataset/4seasons","name":"4Seasons","full_name":"","num_papers_in_archive":15},{"url":"/dataset/titan","name":"TITAN","full_name":"","num_papers_in_archive":15},{"url":"/dataset/kitti-depth","name":"KITTI-Depth","full_name":null,"num_papers_in_archive":14},{"url":"/dataset/urbanloco","name":"UrbanLoco","full_name":"","num_papers_in_archive":14},{"url":"/dataset/synwoodscape","name":"SynWoodScape","full_name":"Synthetic Surround-view Fisheye Camera Dataset for Autonomous Driving","num_papers_in_archive":13},{"url":"/dataset/once-3dlanes","name":"ONCE-3DLanes","full_name":"Monocular 3D Lane Detection Dataset","num_papers_in_archive":12},{"url":"/dataset/soda10m","name":"SODA10M","full_name":"","num_papers_in_archive":12},{"url":"/dataset/dawn","name":"DAWN","full_name":"","num_papers_in_archive":10},{"url":"/dataset/opendd","name":"openDD","full_name":"","num_papers_in_archive":10},{"url":"/dataset/blvd","name":"BLVD","full_name":"","num_papers_in_archive":9},{"url":"/dataset/synthcity","name":"SynthCity","full_name":"","num_papers_in_archive":8},{"url":"/dataset/durlar","name":"DurLAR","full_name":"A High-Fidelity 128-Channel LiDAR Dataset with Panoramic Ambient and Reflectivity Imagery","num_papers_in_archive":5},{"url":"/dataset/elas","name":"ELAS","full_name":"","num_papers_in_archive":5},{"url":"/dataset/pedx","name":"PedX","full_name":"","num_papers_in_archive":5},{"url":"/dataset/soda-d","name":"SODA-D","full_name":"","num_papers_in_archive":5},{"url":"/dataset/brno-urban-dataset","name":"Brno-Urban-Dataset","full_name":null,"num_papers_in_archive":4},{"url":"/dataset/carlane-benchmark","name":"CARLANE Benchmark","full_name":"","num_papers_in_archive":4},{"url":"/dataset/muad","name":"MUAD","full_name":"Multiple Uncertainties for Autonomous Driving","num_papers_in_archive":4},{"url":"/dataset/swiss3dcities","name":"Swiss3DCities","full_name":"","num_papers_in_archive":4},{"url":"/dataset/tcg","name":"TCG","full_name":"Traffic Control Gesture","num_papers_in_archive":4},{"url":"/dataset/long-term-visual-localization","name":"Long-term visual localization","full_name":"","num_papers_in_archive":3},{"url":"/dataset/sdn","name":"SDN","full_name":"Situated Dialogue Navigation","num_papers_in_archive":3},{"url":"/dataset/tusimple-lane","name":"TuSimple Lane","full_name":"","num_papers_in_archive":3},{"url":"/dataset/undd","name":"UNDD","full_name":"Urban Night Driving Dataset","num_papers_in_archive":3},{"url":"/dataset/occ-traj120","name":"Occ-Traj120","full_name":null,"num_papers_in_archive":2},{"url":"/dataset/stereomsi","name":"StereoMSI","full_name":"","num_papers_in_archive":2},{"url":"/dataset/talk2nav","name":"Talk2Nav","full_name":"","num_papers_in_archive":2},{"url":"/dataset/tas-nir","name":"TAS-NIR","full_name":"","num_papers_in_archive":2},{"url":"/dataset/virtual-gallery","name":"Virtual Gallery","full_name":"","num_papers_in_archive":2},{"url":"/dataset/apron-dataset","name":"Apron Dataset","full_name":"","num_papers_in_archive":1},{"url":"/dataset/autonomous-driving-streaming-perception","name":"Autonomous-driving Streaming Perception Benchmarrk","full_name":"","num_papers_in_archive":1},{"url":"/dataset/citr-dataset","name":"CITR Dataset","full_name":"","num_papers_in_archive":1},{"url":"/dataset/coool-challenge-of-out-of-label-a-novel","name":"COOOL: Challenge Of Out-Of-Label A Novel Benchmark for Autonomous Driving","full_name":"","num_papers_in_archive":1},{"url":"/dataset/drivingweather","name":"Driving Weather","full_name":"","num_papers_in_archive":1},{"url":"/dataset/lidar-cs","name":"LiDAR-CS","full_name":"","num_papers_in_archive":1},{"url":"/dataset/meteor","name":"METEOR","full_name":"","num_papers_in_archive":1},{"url":"/dataset/panoramic-video-panoptic-segmentation-dataset","name":"Panoramic Video Panoptic Segmentation Dataset","full_name":"","num_papers_in_archive":1},{"url":"/dataset/precog","name":"PRECOG","full_name":"PREdiction of Clinical Outcomes from Genomic Profiles","num_papers_in_archive":1},{"url":"/dataset/reasonable-crowd","name":"Reasonable Crowd","full_name":"","num_papers_in_archive":1},{"url":"/dataset/humans-in-3d","name":"Humans in 3D","full_name":"","num_papers_in_archive":0},{"url":"/dataset/hyper-drive","name":"Hyper Drive","full_name":"Hyperspectral Driving Dataset","num_papers_in_archive":0},{"url":"/dataset/loli-street","name":"LoLI-Street","full_name":"Low-Light Images of Streets","num_papers_in_archive":0}],"subtasks":[{"url":"/task/3d-pedestrian-tracking","name":"3D Pedestrian Tracking"},{"url":"/task/bench2drive","name":"Bench2Drive"},{"url":"/task/carla-map-leaderboard","name":"CARLA MAP Leaderboard"},{"url":"/task/dead-reckoning-prediction","name":"Dead-Reckoning Prediction"},{"url":"/task/motion-forecasting","name":"Motion Forecasting"},{"url":"/task/navsim","name":"NavSim"}],"parent_tasks":[{"url":"/task/autonomous-vehicles","name":"Autonomous Vehicles"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":2091,"tagged_in_all":6092,"items":[{"url":"/paper/yolox-exceeding-yolo-series-in-2021","title":"YOLOX: Exceeding YOLO Series in 2021","date":"2021-07-18","arxiv_id":"2107.08430","repositories_listed":42,"syntology":{"n":23,"n_ran":1,"n_unverified":22,"n_pointer_only":0}},{"url":"/paper/pointpillars-fast-encoders-for-object","title":"PointPillars: Fast Encoders for Object Detection from Point Clouds","date":"2018-12-14","arxiv_id":"1812.05784","repositories_listed":18,"syntology":{"n":15,"n_ran":2,"n_unverified":13,"n_pointer_only":1}},{"url":"/paper/nuscenes-a-multimodal-dataset-for-autonomous","title":"nuScenes: A multimodal dataset for autonomous driving","date":"2019-03-26","arxiv_id":"1903.11027","repositories_listed":16,"syntology":{"n":17,"n_ran":1,"n_unverified":16,"n_pointer_only":0}},{"url":"/paper/multinet-real-time-joint-semantic-reasoning","title":"MultiNet: Real-time Joint Semantic Reasoning for Autonomous Driving","date":"2016-12-22","arxiv_id":"1612.07695","repositories_listed":15,"syntology":{"n":42,"n_ran":6,"n_unverified":36,"n_pointer_only":0}},{"url":"/paper/awq-activation-aware-weight-quantization-for","title":"AWQ: Activation-aware Weight Quantization for LLM Compression and Acceleration","date":"2023-06-01","arxiv_id":"2306.00978","repositories_listed":12,"syntology":{"n":18,"n_ran":12,"n_unverified":6,"n_pointer_only":2}},{"url":"/paper/squeezedet-unified-small-low-power-fully","title":"SqueezeDet: Unified, Small, Low Power Fully Convolutional Neural Networks for Real-Time Object Detection for Autonomous Driving","date":"2016-12-04","arxiv_id":"1612.01051","repositories_listed":12,"syntology":{"n":30,"n_ran":0,"n_unverified":30,"n_pointer_only":0}},{"url":"/paper/key-points-estimation-and-point-instance","title":"Key Points Estimation and Point Instance Segmentation Approach for Lane Detection","date":"2020-02-16","arxiv_id":"2002.06604","repositories_listed":10,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/complex-yolo-real-time-3d-object-detection-on","title":"Complex-YOLO: Real-time 3D Object Detection on Point Clouds","date":"2018-03-16","arxiv_id":"1803.06199","repositories_listed":10,"syntology":null},{"url":"/paper/fcos3d-fully-convolutional-one-stage","title":"FCOS3D: Fully Convolutional One-Stage Monocular 3D Object Detection","date":"2021-04-22","arxiv_id":"2104.10956","repositories_listed":9,"syntology":{"n":22,"n_ran":7,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/multi-modality-cut-and-paste-for-3d-object","title":"Exploring Data Augmentation for Multi-Modality 3D Object Detection","date":"2020-12-23","arxiv_id":"2012.12741","repositories_listed":9,"syntology":null},{"url":"/paper/learning-by-cheating","title":"Learning by Cheating","date":"2019-12-27","arxiv_id":"1912.12294","repositories_listed":9,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/scalability-in-perception-for-autonomous","title":"Scalability in Perception for Autonomous Driving: Waymo Open Dataset","date":"2019-12-10","arxiv_id":"1912.04838","repositories_listed":9,"syntology":{"n":10,"n_ran":0,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/the-double-sphere-camera-model","title":"The Double Sphere Camera Model","date":"2018-07-24","arxiv_id":"1807.08957","repositories_listed":9,"syntology":null},{"url":"/paper/learning-to-drive-in-a-day","title":"Learning to Drive in a Day","date":"2018-07-01","arxiv_id":"1807.00412","repositories_listed":8,"syntology":{"n":22,"n_ran":0,"n_unverified":22,"n_pointer_only":0}},{"url":"/paper/explaining-how-a-deep-neural-network-trained","title":"Explaining How a Deep Neural Network Trained with End-to-End Learning Steers a Car","date":"2017-04-25","arxiv_id":"1704.07911","repositories_listed":8,"syntology":null},{"url":"/paper/afdet-anchor-free-one-stage-3d-object","title":"AFDet: Anchor Free One Stage 3D Object Detection","date":"2020-06-23","arxiv_id":"2006.12671","repositories_listed":7,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/efficientvit-enhanced-linear-attention-for","title":"EfficientViT: Multi-Scale Linear Attention for High-Resolution Dense Prediction","date":"2022-05-29","arxiv_id":"2205.14756","repositories_listed":6,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/multi-path-segmentation-network","title":"ShelfNet for Fast Semantic Segmentation","date":"2018-11-27","arxiv_id":"1811.11254","repositories_listed":6,"syntology":null},{"url":"/paper/deeptraffic-crowdsourced-hyperparameter","title":"DeepTraffic: Crowdsourced Hyperparameter Tuning of Deep Reinforcement Learning Systems for Multi-Agent Dense Traffic Navigation","date":"2018-01-09","arxiv_id":"1801.02805","repositories_listed":6,"syntology":null},{"url":"/paper/virtual-to-real-reinforcement-learning-for","title":"Virtual to Real Reinforcement Learning for Autonomous Driving","date":"2017-04-13","arxiv_id":"1704.03952","repositories_listed":6,"syntology":null},{"url":"/paper/awesome-multi-modal-object-tracking","title":"Awesome Multi-modal Object Tracking","date":"2024-05-23","arxiv_id":"2405.14200","repositories_listed":5,"syntology":null},{"url":"/paper/yolop-you-only-look-once-for-panoptic-driving","title":"YOLOP: You Only Look Once for Panoptic Driving Perception","date":"2021-08-25","arxiv_id":"2108.11250","repositories_listed":5,"syntology":{"n":18,"n_ran":1,"n_unverified":17,"n_pointer_only":0}},{"url":"/paper/smarts-scalable-multi-agent-reinforcement","title":"SMARTS: Scalable Multi-Agent Reinforcement Learning Training School for Autonomous Driving","date":"2020-10-19","arxiv_id":"2010.09776","repositories_listed":5,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":5}},{"url":"/paper/carrada-dataset-camera-and-automotive-radar","title":"CARRADA Dataset: Camera and Automotive Radar with Range-Angle-Doppler Annotations","date":"2020-05-04","arxiv_id":"2005.01456","repositories_listed":5,"syntology":null},{"url":"/paper/salsanext-fast-semantic-segmentation-of-lidar","title":"SalsaNext: Fast, Uncertainty-aware Semantic Segmentation of LiDAR Point Clouds for Autonomous Driving","date":"2020-03-07","arxiv_id":"2003.03653","repositories_listed":5,"syntology":{"n":11,"n_ran":4,"n_unverified":7,"n_pointer_only":1}},{"url":"/paper/guiding-deep-learning-system-testing-using","title":"Guiding Deep Learning System Testing using Surprise Adequacy","date":"2018-08-25","arxiv_id":"1808.08444","repositories_listed":5,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/squeezeseg-convolutional-neural-nets-with","title":"SqueezeSeg: Convolutional Neural Nets with Recurrent CRF for Real-Time Road-Object Segmentation from 3D LiDAR Point Cloud","date":"2017-10-19","arxiv_id":"1710.07368","repositories_listed":5,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/zoo-zeroth-order-optimization-based-black-box","title":"ZOO: Zeroth Order Optimization based Black-box Attacks to Deep Neural Networks without Training Substitute Models","date":"2017-08-14","arxiv_id":"1708.03999","repositories_listed":5,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/bench2drive-towards-multi-ability","title":"Bench2Drive: Towards Multi-Ability Benchmarking of Closed-Loop End-To-End Autonomous Driving","date":"2024-06-06","arxiv_id":"2406.03877","repositories_listed":4,"syntology":{"n":17,"n_ran":13,"n_unverified":4,"n_pointer_only":17}},{"url":"/paper/deflow-decoder-of-scene-flow-network-in","title":"DeFlow: Decoder of Scene Flow Network in Autonomous Driving","date":"2024-01-29","arxiv_id":"2401.16122","repositories_listed":4,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}}],"syntology_records":21,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":2,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}