{"url":"/task/3d-visual-grounding","name":"3D visual grounding","slug":"3d-visual-grounding","description_markdown":null,"categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":82,"papers_with_code":39,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":2,"subtasks":0,"parent_tasks":1},"benchmarks":[],"datasets":[{"url":"/dataset/mono3drefer","name":"Mono3DRefer","full_name":"","num_papers_in_archive":3},{"url":"/dataset/beacon3d","name":"Beacon3D","full_name":"","num_papers_in_archive":2}],"subtasks":[],"parent_tasks":[{"url":"/task/visual-grounding","name":"Visual Grounding"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":39,"tagged_in_all":82,"items":[{"url":"/paper/viewrefer-grasp-the-multi-view-knowledge-for","title":"ViewRefer: Grasp the Multi-view Knowledge for 3D Visual Grounding with GPT and Prototype Guidance","date":"2023-03-29","arxiv_id":"2303.16894","repositories_listed":7,"syntology":{"n":12,"n_ran":8,"n_unverified":4,"n_pointer_only":12}},{"url":"/paper/eda-explicit-text-decoupling-and-dense","title":"EDA: Explicit Text-Decoupling and Dense Alignment for 3D Visual Grounding","date":"2022-09-29","arxiv_id":"2209.14941","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/refer-it-in-rgbd-a-bottom-up-approach-for-3d","title":"Refer-it-in-RGBD: A Bottom-up Approach for 3D Visual Grounding in RGBD Images","date":"2021-03-14","arxiv_id":"2103.07894","repositories_listed":2,"syntology":{"n":16,"n_ran":11,"n_unverified":5,"n_pointer_only":6}},{"url":"/paper/extending-large-vision-language-model-for","title":"Extending Large Vision-Language Model for Diverse Interactive Tasks in Autonomous Driving","date":"2025-05-13","arxiv_id":"2505.08725","repositories_listed":1,"syntology":null},{"url":"/paper/as3d-2d-assisted-cross-modal-understanding","title":"AS3D: 2D-Assisted Cross-Modal Understanding with Semantic-Spatial Scene Graphs for 3D Visual Grounding","date":"2025-05-07","arxiv_id":"2505.04058","repositories_listed":1,"syntology":null},{"url":"/paper/ges3vig-incorporating-pointing-gestures-into","title":"Ges3ViG: Incorporating Pointing Gestures into Language-Based 3D Visual Grounding for Embodied Reference Understanding","date":"2025-04-13","arxiv_id":"2504.09623","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-the-mist-over-3d-vision-language","title":"Unveiling the Mist over 3D Vision-Language Understanding: Object-centric Evaluation with Chain-of-Analysis","date":"2025-03-28","arxiv_id":"2503.22420","repositories_listed":1,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/text-guided-sparse-voxel-pruning-for","title":"Text-guided Sparse Voxel Pruning for Efficient 3D Visual Grounding","date":"2025-02-14","arxiv_id":"2502.10392","repositories_listed":1,"syntology":null},{"url":"/paper/evolving-symbolic-3d-visual-grounder-with","title":"Evolving Symbolic 3D Visual Grounder with Weakly Supervised Reflection","date":"2025-02-03","arxiv_id":"2502.01401","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/ges3vig-incorporating-pointing-gestures-into-1","title":"Ges3ViG : Incorporating Pointing Gestures into Language-Based 3D Visual Grounding for Embodied Reference Understanding","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/bip3d-bridging-2d-images-and-3d-perception","title":"BIP3D: Bridging 2D Images and 3D Perception for Embodied Intelligence","date":"2024-11-22","arxiv_id":"2411.14869","repositories_listed":1,"syntology":null},{"url":"/paper/solving-zero-shot-3d-visual-grounding-as","title":"Solving Zero-Shot 3D Visual Grounding as Constraint Satisfaction Problems","date":"2024-11-21","arxiv_id":"2411.14594","repositories_listed":1,"syntology":null},{"url":"/paper/vlm-grounder-a-vlm-agent-for-zero-shot-3d","title":"VLM-Grounder: A VLM Agent for Zero-Shot 3D Visual Grounding","date":"2024-10-17","arxiv_id":"2410.13860","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_unverified":1,"n_pointer_only":11}},{"url":"/paper/refmask3d-language-guided-transformer-for-3d","title":"RefMask3D: Language-Guided Transformer for 3D Referring Segmentation","date":"2024-07-25","arxiv_id":"2407.18244","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/multi-branch-collaborative-learning-network","title":"Multi-branch Collaborative Learning Network for 3D Visual Grounding","date":"2024-07-07","arxiv_id":"2407.05363","repositories_listed":1,"syntology":null},{"url":"/paper/mmscan-a-multi-modal-3d-scene-dataset-with","title":"MMScan: A Multi-Modal 3D Scene Dataset with Hierarchical Grounded Language Annotations","date":"2024-06-13","arxiv_id":"2406.09401","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-text-guided-3d-visual-grounding","title":"A Survey on Text-guided 3D Visual Grounding: Elements, Recent Advances, and Future Directions","date":"2024-06-09","arxiv_id":"2406.05785","repositories_listed":1,"syntology":null},{"url":"/paper/talk2radar-bridging-natural-language-with-4d","title":"Talk2Radar: Bridging Natural Language with 4D mmWave Radar for 3D Referring Expression Comprehension","date":"2024-05-21","arxiv_id":"2405.12821","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-3d-dense-caption-and-visual","title":"Rethinking 3D Dense Caption and Visual Grounding in A Unified Framework through Prompt-based Localization","date":"2024-04-17","arxiv_id":"2404.11064","repositories_listed":1,"syntology":null},{"url":"/paper/secg-semantic-enhanced-3d-visual-grounding","title":"SeCG: Semantic-Enhanced 3D Visual Grounding via Cross-modal Graph Attention","date":"2024-03-13","arxiv_id":"2403.08182","repositories_listed":1,"syntology":null},{"url":"/paper/mikasa-multi-key-anchor-scene-aware","title":"MiKASA: Multi-Key-Anchor & Scene-Aware Transformer for 3D Visual Grounding","date":"2024-03-05","arxiv_id":"2403.03077","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_unverified":5,"n_pointer_only":16}},{"url":"/paper/multi-attribute-interactions-matter-for-3d","title":"Multi-Attribute Interactions Matter for 3D Visual Grounding","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mono3dvg-3d-visual-grounding-in-monocular","title":"Mono3DVG: 3D Visual Grounding in Monocular Images","date":"2023-12-13","arxiv_id":"2312.08022","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":7}},{"url":"/paper/visual-programming-for-zero-shot-open","title":"Visual Programming for Zero-shot Open-Vocabulary 3D Visual Grounding","date":"2023-11-26","arxiv_id":"2311.15383","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/cityrefer-geography-aware-3d-visual-grounding","title":"CityRefer: Geography-aware 3D Visual Grounding Dataset on City-scale Point Cloud Data","date":"2023-10-28","arxiv_id":"2310.18773","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/cot3dref-chain-of-thoughts-data-efficient-3d","title":"CoT3DRef: Chain-of-Thoughts Data-Efficient 3D Visual Grounding","date":"2023-10-10","arxiv_id":"2310.06214","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_unverified":1,"n_pointer_only":11}},{"url":"/paper/llm-grounder-open-vocabulary-3d-visual","title":"LLM-Grounder: Open-Vocabulary 3D Visual Grounding with Large Language Model as an Agent","date":"2023-09-21","arxiv_id":"2309.12311","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/multi3drefer-grounding-text-description-to","title":"Multi3DRefer: Grounding Text Description to Multiple 3D Objects","date":"2023-09-11","arxiv_id":"2309.05251","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/distilling-coarse-to-fine-semantic-matching","title":"Distilling Coarse-to-Fine Semantic Matching Knowledge for Weakly Supervised 3D Visual Grounding","date":"2023-07-18","arxiv_id":"2307.09267","repositories_listed":1,"syntology":null},{"url":"/paper/cross3dvg-baseline-and-dataset-for-cross","title":"Cross3DVG: Cross-Dataset 3D Visual Grounding on Different RGB-D Scans","date":"2023-05-23","arxiv_id":"2305.13876","repositories_listed":1,"syntology":null}],"syntology_records":14,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}