{"url":"/task/vision-language-action","name":"Vision-Language-Action","slug":"vision-language-action","description_markdown":null,"categories":[{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Robots","url":"/area/robots"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":157,"papers_with_code":49,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":1},"benchmarks":[],"datasets":[{"url":"/dataset/automotiveui-bench-4k","name":"AutomotiveUI-Bench-4K","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/vision-language-navigation","name":"Vision-Language Navigation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":49,"tagged_in_all":157,"items":[{"url":"/paper/openvla-an-open-source-vision-language-action","title":"OpenVLA: An Open-Source Vision-Language-Action Model","date":"2024-06-13","arxiv_id":"2406.09246","repositories_listed":3,"syntology":{"n":10,"n_ran":6,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/vision-language-action-models-in-robotic","title":"Vision Language Action Models in Robotic Manipulation: A Systematic Review","date":"2025-07-14","arxiv_id":"2507.10672","repositories_listed":1,"syntology":null},{"url":"/paper/vote-vision-language-action-optimization-with","title":"VOTE: Vision-Language-Action Optimization with Trajectory Ensemble Voting","date":"2025-07-07","arxiv_id":"2507.05116","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":5}},{"url":"/paper/dreamvla-a-vision-language-action-model-1","title":"DreamVLA: A Vision-Language-Action Model Dreamed with Comprehensive World Knowledge","date":"2025-07-06","arxiv_id":"2507.04447","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/a-survey-on-vision-language-action-models-for-1","title":"A Survey on Vision-Language-Action Models for Autonomous Driving","date":"2025-06-30","arxiv_id":"2506.24044","repositories_listed":1,"syntology":null},{"url":"/paper/worldvla-towards-autoregressive-action-world","title":"WorldVLA: Towards Autoregressive Action World Model","date":"2025-06-26","arxiv_id":"2506.21539","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":4}},{"url":"/paper/parallels-between-vla-model-post-training-and","title":"Parallels Between VLA Model Post-Training and Human Motor Learning: Progress, Challenges, and Trends","date":"2025-06-26","arxiv_id":"2506.20966","repositories_listed":1,"syntology":null},{"url":"/paper/autovla-a-vision-language-action-model-for","title":"AutoVLA: A Vision-Language-Action Model for End-to-End Autonomous Driving with Adaptive Reasoning and Reinforcement Fine-Tuning","date":"2025-06-16","arxiv_id":"2506.13757","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/a-comprehensive-survey-on-continual-learning","title":"A Comprehensive Survey on Continual Learning in Generative Models","date":"2025-06-16","arxiv_id":"2506.13045","repositories_listed":1,"syntology":null},{"url":"/paper/surgeon-style-fingerprinting-and-privacy-risk","title":"Surgeon Style Fingerprinting and Privacy Risk Quantification via Discrete Diffusion Models in a Vision-Language-Action Framework","date":"2025-06-09","arxiv_id":"2506.08185","repositories_listed":1,"syntology":null},{"url":"/paper/bitvla-1-bit-vision-language-action-models","title":"BitVLA: 1-bit Vision-Language-Action Models for Robotics Manipulation","date":"2025-06-09","arxiv_id":"2506.07530","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-execution-of-action-chunking-flow","title":"Real-Time Execution of Action Chunking Flow Policies","date":"2025-06-09","arxiv_id":"2506.07339","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/adversarial-attacks-on-robotic-vision","title":"Adversarial Attacks on Robotic Vision Language Action Models","date":"2025-06-03","arxiv_id":"2506.03350","repositories_listed":1,"syntology":null},{"url":"/paper/smolvla-a-vision-language-action-model-for","title":"SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics","date":"2025-06-02","arxiv_id":"2506.01844","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":5}},{"url":"/paper/impromptu-vla-open-weights-and-open-data-for","title":"Impromptu VLA: Open Weights and Open Data for Driving Vision-Language-Action Models","date":"2025-05-29","arxiv_id":"2505.23757","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-action-model-with-open-world","title":"ChatVLA-2: Vision-Language-Action Model with Open-World Embodied Reasoning from Pretrained Knowledge","date":"2025-05-28","arxiv_id":"2505.21906","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/vla-rl-towards-masterful-and-general-robotic","title":"VLA-RL: Towards Masterful and General Robotic Manipulation with Scalable Reinforcement Learning","date":"2025-05-24","arxiv_id":"2505.18719","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/perceptual-quality-assessment-for-embodied-ai","title":"Perceptual Quality Assessment for Embodied AI","date":"2025-05-22","arxiv_id":"2505.16815","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-limits-of-vision-language","title":"Exploring the Limits of Vision-Language-Action Manipulations in Cross-task Generalization","date":"2025-05-21","arxiv_id":"2505.15660","repositories_listed":1,"syntology":null},{"url":"/paper/robofac-a-comprehensive-framework-for-robotic","title":"RoboFAC: A Comprehensive Framework for Robotic Failure Analysis and Correction","date":"2025-05-18","arxiv_id":"2505.12224","repositories_listed":1,"syntology":null},{"url":"/paper/from-seeing-to-doing-bridging-reasoning-and","title":"From Seeing to Doing: Bridging Reasoning and Decision for Robotic Manipulation","date":"2025-05-13","arxiv_id":"2505.08548","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/univla-learning-to-act-anywhere-with-task","title":"UniVLA: Learning to Act Anywhere with Task-centric Latent Actions","date":"2025-05-09","arxiv_id":"2505.06111","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/benchmarking-vision-language-action-models-in","title":"Benchmarking Vision, Language, & Action Models in Procedurally Generated, Open Ended Action Environments","date":"2025-05-08","arxiv_id":"2505.05540","repositories_listed":1,"syntology":null},{"url":"/paper/openhelix-a-short-survey-empirical-analysis","title":"OpenHelix: A Short Survey, Empirical Analysis, and Open-Source Dual-System VLA Model for Robotic Manipulation","date":"2025-05-06","arxiv_id":"2505.03912","repositories_listed":1,"syntology":{"n":12,"n_ran":5,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/gui-r1-a-generalist-r1-style-vision-language","title":"GUI-R1 : A Generalist R1-Style Vision-Language Action Model For GUI Agents","date":"2025-04-14","arxiv_id":"2504.10458","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/opendrivevla-towards-end-to-end-autonomous","title":"OpenDriveVLA: Towards End-to-end Autonomous Driving with Large Vision Language Action Model","date":"2025-03-30","arxiv_id":"2503.23463","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/dita-scaling-diffusion-transformer-for","title":"Dita: Scaling Diffusion Transformer for Generalist Vision-Language-Action Policy","date":"2025-03-25","arxiv_id":"2503.19757","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/combatvla-an-efficient-vision-language-action","title":"CombatVLA: An Efficient Vision-Language-Action Model for Combat Tasks in 3D Action Role-Playing Games","date":"2025-03-12","arxiv_id":"2503.09527","repositories_listed":1,"syntology":null},{"url":"/paper/pointvla-injecting-the-3d-world-into-vision","title":"PointVLA: Injecting the 3D World into Vision-Language-Action Models","date":"2025-03-10","arxiv_id":"2503.07511","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/fine-tuning-vision-language-action-models","title":"Fine-Tuning Vision-Language-Action Models: Optimizing Speed and Success","date":"2025-02-27","arxiv_id":"2502.19645","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}}],"syntology_records":17,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-25T09:33:49+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}