{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/scene-understanding/papers/9","list_of":"/task/scene-understanding","task":"Scene Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":18,"rows_per_page":100,"rows":[801,900],"of":1723,"counts":{"archive_papers_tagged":1723,"with_a_code_link":720,"where_syntology_ran_a_sample":208,"not_listed_spam_title":0,"listed":1723,"listed_where_code_ran":208,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":26,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":26,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/scene-understanding","prev":"/task/scene-understanding/papers/8","next":"/task/scene-understanding/papers/10","papers":[{"url":null,"slug":"single-input-multi-output-model-merging","title":"Single-Input Multi-Output Model Merging: Leveraging Foundation Models for Dense Multi-Task Learning","date":"2025-04-15","arxiv_id":"2504.11268","repositories_listed":0,"syntology":null},{"url":null,"slug":"foundation-models-for-remote-sensing-an","title":"Foundation Models for Remote Sensing: An Analysis of MLLMs for Object Localization","date":"2025-04-14","arxiv_id":"2504.10727","repositories_listed":0,"syntology":null},{"url":null,"slug":"dsm-building-a-diverse-semantic-map-for-3d","title":"DSM: Building A Diverse Semantic Map for 3D Visual Grounding","date":"2025-04-11","arxiv_id":"2504.08307","repositories_listed":0,"syntology":null},{"url":null,"slug":"findanything-open-vocabulary-and-object","title":"FindAnything: Open-Vocabulary and Object-Centric Mapping for Robot Exploration in Any Environment","date":"2025-04-11","arxiv_id":"2504.08603","repositories_listed":0,"syntology":null},{"url":null,"slug":"fmlgs-fast-multilevel-language-embedded","title":"FMLGS: Fast Multilevel Language Embedded Gaussians for Part-level Interactive Agents","date":"2025-04-11","arxiv_id":"2504.08581","repositories_listed":0,"syntology":null},{"url":null,"slug":"dgocc-depth-aware-global-query-based-network","title":"DGOcc: Depth-aware Global Query-based Network for Monocular 3D Occupancy Prediction","date":"2025-04-10","arxiv_id":"2504.07524","repositories_listed":0,"syntology":null},{"url":null,"slug":"attributes-aware-visual-emotion","title":"Attributes-aware Visual Emotion Representation Learning","date":"2025-04-09","arxiv_id":"2504.06578","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-event-localization-on-portrait","title":"Audio-visual Event Localization on Portrait Mode Short Videos","date":"2025-04-09","arxiv_id":"2504.06884","repositories_listed":0,"syntology":null},{"url":null,"slug":"movsam-a-single-image-moving-object","title":"MovSAM: A Single-image Moving Object Segmentation Framework Based on Deep Thinking","date":"2025-04-09","arxiv_id":"2504.06863","repositories_listed":0,"syntology":null},{"url":null,"slug":"rayfronts-open-set-semantic-ray-frontiers-for","title":"RayFronts: Open-Set Semantic Ray Frontiers for Online Scene Understanding and Exploration","date":"2025-04-09","arxiv_id":"2504.06994","repositories_listed":0,"syntology":null},{"url":null,"slug":"primedrive-cot-a-precognitive-chain-of","title":"PRIMEDrive-CoT: A Precognitive Chain-of-Thought Framework for Uncertainty-Aware Object Interaction in Driving Scene Scenario","date":"2025-04-08","arxiv_id":"2504.05908","repositories_listed":0,"syntology":null},{"url":null,"slug":"rs-rag-bridging-remote-sensing-imagery-and","title":"RS-RAG: Bridging Remote Sensing Imagery and Comprehensive Knowledge with a Multi-Modal Dataset and Retrieval-Augmented Generation Model","date":"2025-04-07","arxiv_id":"2504.04988","repositories_listed":0,"syntology":null},{"url":null,"slug":"comatcher-multi-view-collaborative-feature","title":"CoMatcher: Multi-View Collaborative Feature Matching","date":"2025-04-02","arxiv_id":"2504.01872","repositories_listed":0,"syntology":null},{"url":null,"slug":"overlap-aware-feature-learning-for-robust","title":"Overlap-Aware Feature Learning for Robust Unsupervised Domain Adaptation for 3D Semantic Segmentation","date":"2025-04-02","arxiv_id":"2504.01668","repositories_listed":0,"syntology":null},{"url":null,"slug":"ross3d-reconstructive-visual-instruction","title":"Ross3D: Reconstructive Visual Instruction Tuning with 3D-Awareness","date":"2025-04-02","arxiv_id":"2504.01901","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformerger-transformer-based-voice","title":"TransforMerger: Transformer-based Voice-Gesture Fusion for Robust Human-Robot Communication","date":"2025-04-02","arxiv_id":"2504.01708","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-aware-human-behavior-prediction-using","title":"Context-Aware Human Behavior Prediction Using Multimodal Large Language Models: Challenges and Insights","date":"2025-04-01","arxiv_id":"2504.00839","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-4d-lidar-panoptic-segmentation","title":"Zero-Shot 4D Lidar Panoptic Segmentation","date":"2025-04-01","arxiv_id":"2504.00848","repositories_listed":0,"syntology":null},{"url":null,"slug":"physpose-refining-6d-object-poses-with","title":"PhysPose: Refining 6D Object Poses with Physical Constraints","date":"2025-03-30","arxiv_id":"2503.23587","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-deepseek-v3-reason-like-a-surgeon-an","title":"Can DeepSeek Reason Like a Surgeon? An Empirical Evaluation for Vision-Language Understanding in Robotic-Assisted Surgery","date":"2025-03-29","arxiv_id":"2503.23130","repositories_listed":0,"syntology":null},{"url":null,"slug":"empowering-large-language-models-with-3d","title":"Empowering Large Language Models with 3D Situation Awareness","date":"2025-03-29","arxiv_id":"2503.23024","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-vocabulary-semantic-segmentation-with-3","title":"Open-Vocabulary Semantic Segmentation with Uncertainty Alignment for Robotic Scene Understanding in Indoor Building Environments","date":"2025-03-29","arxiv_id":"2503.23105","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dataset-for-semantic-segmentation-in-the","title":"A Dataset for Semantic Segmentation in the Presence of Unknowns","date":"2025-03-28","arxiv_id":"2503.22309","repositories_listed":0,"syntology":null},{"url":null,"slug":"endo-ttap-robust-endoscopic-tissue-tracking","title":"Endo-TTAP: Robust Endoscopic Tissue Tracking via Multi-Facet Guided Attention and Hybrid Flow-point Supervision","date":"2025-03-28","arxiv_id":"2503.22394","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-multimodal-language-models-as","title":"Evaluating Multimodal Language Models as Visual Assistants for Visually Impaired Users","date":"2025-03-28","arxiv_id":"2503.22610","repositories_listed":0,"syntology":null},{"url":null,"slug":"next-best-trajectory-planning-of-robot","title":"Next-Best-Trajectory Planning of Robot Manipulators for Effective Observation and Exploration","date":"2025-03-28","arxiv_id":"2503.22588","repositories_listed":0,"syntology":null},{"url":null,"slug":"nugrounding-a-multi-view-3d-visual-grounding","title":"NuGrounding: A Multi-View 3D Visual Grounding Framework in Autonomous Driving","date":"2025-03-28","arxiv_id":"2503.22436","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-jenga-discovering-object-dependencies","title":"Visual Jenga: Discovering Object Dependencies via Counterfactual Inpainting","date":"2025-03-27","arxiv_id":"2503.21770","repositories_listed":0,"syntology":null},{"url":null,"slug":"dinemo-learning-neural-mesh-models-with-no-3d","title":"DINeMo: Learning Neural Mesh Models with no 3D Annotations","date":"2025-03-26","arxiv_id":"2503.20220","repositories_listed":0,"syntology":null},{"url":null,"slug":"openlex3d-a-new-evaluation-benchmark-for-open","title":"OpenLex3D: A New Evaluation Benchmark for Open-Vocabulary 3D Scene Representations","date":"2025-03-25","arxiv_id":"2503.19764","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-the-road-ahead-a-knowledge-graph","title":"Predicting the Road Ahead: A Knowledge Graph based Foundation Model for Scene Understanding in Autonomous Driving","date":"2025-03-24","arxiv_id":"2503.18730","repositories_listed":0,"syntology":null},{"url":null,"slug":"geometric-constrained-non-line-of-sight","title":"Geometric Constrained Non-Line-of-Sight Imaging","date":"2025-03-23","arxiv_id":"2503.17992","repositories_listed":0,"syntology":null},{"url":null,"slug":"mllm-for3d-adapting-multimodal-large-language","title":"MLLM-For3D: Adapting Multimodal Large Language Model for 3D Reasoning Segmentation","date":"2025-03-23","arxiv_id":"2503.18135","repositories_listed":0,"syntology":null},{"url":null,"slug":"panogs-gaussian-based-panoptic-segmentation","title":"PanoGS: Gaussian-based Panoptic Segmentation for 3D Open Vocabulary Scene Understanding","date":"2025-03-23","arxiv_id":"2503.18107","repositories_listed":0,"syntology":null},{"url":null,"slug":"panopticsplatting-end-to-end-panoptic","title":"PanopticSplatting: End-to-End Panoptic Gaussian Splatting","date":"2025-03-23","arxiv_id":"2503.18073","repositories_listed":0,"syntology":null},{"url":null,"slug":"claravid-a-holistic-scene-reconstruction","title":"ClaraVid: A Holistic Scene Reconstruction Benchmark From Aerial Perspective With Delentropy-Based Complexity Profiling","date":"2025-03-22","arxiv_id":"2503.17856","repositories_listed":0,"syntology":null},{"url":null,"slug":"excap3d-expressive-3d-scene-understanding-via","title":"ExCap3D: Expressive 3D Scene Understanding via Object Captioning with Varying Detail","date":"2025-03-21","arxiv_id":"2503.17044","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-monocular-vision-to-autonomous-action","title":"From Monocular Vision to Autonomous Action: Guiding Tumor Resection via 3D Reconstruction","date":"2025-03-20","arxiv_id":"2503.16263","repositories_listed":0,"syntology":null},{"url":null,"slug":"semanticflow-a-self-supervised-framework-for","title":"SemanticFlow: A Self-Supervised Framework for Joint Scene Flow Prediction and Instance Segmentation in Dynamic Environments","date":"2025-03-19","arxiv_id":"2503.14837","repositories_listed":0,"syntology":null},{"url":null,"slug":"chatbev-a-visual-language-model-that","title":"ChatBEV: A Visual Language Model that Understands BEV Maps","date":"2025-03-18","arxiv_id":"2503.13938","repositories_listed":0,"syntology":null},{"url":"/paper/psa-ssl-pose-and-size-aware-self-supervised","slug":"psa-ssl-pose-and-size-aware-self-supervised","title":"PSA-SSL: Pose and Size-aware Self-Supervised Learning on LiDAR Point Clouds","date":"2025-03-18","arxiv_id":"2503.13914","repositories_listed":0,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":5,"n_pointer_only":4,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 2 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/psa-ssl-pose-and-size-aware-self-supervised#ran","syntology_url":"https://syntology.ai/paper/2503.13914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13914"}},"official":null}},{"url":null,"slug":"these-magic-moments-differentiable","title":"These Magic Moments: Differentiable Uncertainty Quantification of Radiance Field Models","date":"2025-03-18","arxiv_id":"2503.14665","repositories_listed":0,"syntology":null},{"url":null,"slug":"his-gpt-towards-3d-human-in-scene-multimodal","title":"HIS-GPT: Towards 3D Human-In-Scene Multimodal Understanding","date":"2025-03-17","arxiv_id":"2503.12955","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-3d-reconstruction-in","title":"Learning-based 3D Reconstruction in Autonomous Driving: A Comprehensive Survey","date":"2025-03-17","arxiv_id":"2503.14537","repositories_listed":0,"syntology":null},{"url":null,"slug":"egosplat-open-vocabulary-egocentric-scene","title":"EgoSplat: Open-Vocabulary Egocentric Scene Understanding with Language Embedded 3D Gaussian Splatting","date":"2025-03-14","arxiv_id":"2503.11345","repositories_listed":0,"syntology":null},{"url":null,"slug":"road-rage-reasoning-with-vision-language","title":"Road Rage Reasoning with Vision-language Models (VLMs): Task Definition and Evaluation Dataset","date":"2025-03-14","arxiv_id":"2503.11342","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-grounded-llms-leveraging-graphical","title":"Graph-Grounded LLMs: Leveraging Graphical Function Calling to Minimize LLM Hallucinations","date":"2025-03-13","arxiv_id":"2503.10941","repositories_listed":0,"syntology":null},{"url":null,"slug":"tars-traffic-aware-radar-scene-flow","title":"TARS: Traffic-Aware Radar Scene Flow Estimation","date":"2025-03-13","arxiv_id":"2503.10210","repositories_listed":0,"syntology":null},{"url":null,"slug":"tgp-two-modal-occupancy-prediction-with-3d","title":"TGP: Two-modal occupancy prediction with 3D Gaussian and sparse points for 3D Environment Awareness","date":"2025-03-13","arxiv_id":"2503.09941","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-aware-dino-oh-a-dino-enhancing-self","title":"Object-Aware DINO (Oh-A-Dino): Enhancing Self-Supervised Representations for Multi-Object Instance Retrieval","date":"2025-03-12","arxiv_id":"2503.09867","repositories_listed":0,"syntology":null},{"url":null,"slug":"div-ff-dynamic-image-video-feature-fields-for","title":"DIV-FF: Dynamic Image-Video Feature Fields For Environment Understanding in Egocentric Videos","date":"2025-03-11","arxiv_id":"2503.08344","repositories_listed":0,"syntology":null},{"url":null,"slug":"general-purpose-aerial-intelligent-agents","title":"General-Purpose Aerial Intelligent Agents Empowered by Large Language Models","date":"2025-03-11","arxiv_id":"2503.08302","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-robot-constitutions-benchmarks-for","title":"Generating Robot Constitutions & Benchmarks for Semantic Safety","date":"2025-03-11","arxiv_id":"2503.08663","repositories_listed":0,"syntology":null},{"url":null,"slug":"maskattn-unet-a-mask-attention-driven","title":"MaskAttn-UNet: A Mask Attention-Driven Framework for Universal Low-Resolution Image Segmentation","date":"2025-03-11","arxiv_id":"2503.10686","repositories_listed":0,"syntology":null},{"url":null,"slug":"cot-drive-efficient-motion-forecasting-for","title":"CoT-Drive: Efficient Motion Forecasting for Autonomous Driving with LLMs and Chain-of-Thought Prompting","date":"2025-03-10","arxiv_id":"2503.07234","repositories_listed":0,"syntology":null},{"url":null,"slug":"llafea-frame-event-complementary-fusion-for","title":"LLaFEA: Frame-Event Complementary Fusion for Fine-Grained Spatiotemporal Understanding in LMMs","date":"2025-03-10","arxiv_id":"2503.06934","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-endogaussian-feature-distilled","title":"Feature-EndoGaussian: Feature Distilled Gaussian Splatting in Surgical Deformable Scene Reconstruction","date":"2025-03-08","arxiv_id":"2503.06161","repositories_listed":0,"syntology":null},{"url":null,"slug":"segment-anything-even-occluded","title":"Segment Anything, Even Occluded","date":"2025-03-08","arxiv_id":"2503.06261","repositories_listed":0,"syntology":null},{"url":null,"slug":"splattalk-3d-vqa-with-gaussian-splatting","title":"SplatTalk: 3D VQA with Gaussian Splatting","date":"2025-03-08","arxiv_id":"2503.06271","repositories_listed":0,"syntology":null},{"url":null,"slug":"evidmtl-evidential-multi-task-learning-for","title":"EvidMTL: Evidential Multi-Task Learning for Uncertainty-Aware Semantic Surface Mapping from Monocular RGB Images","date":"2025-03-06","arxiv_id":"2503.04441","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-6d-object-pose-estimation-of","title":"Improving 6D Object Pose Estimation of metallic Household and Industry Objects","date":"2025-03-05","arxiv_id":"2503.03655","repositories_listed":0,"syntology":null},{"url":null,"slug":"surgisam2-fine-tuning-a-foundational-model","title":"SurgiSAM2: Fine-tuning a foundational model for surgical video anatomy segmentation and detection","date":"2025-03-05","arxiv_id":"2503.03942","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-language-models-struggle-to-align","title":"Vision-Language Models Struggle to Align Entities across Modalities","date":"2025-03-05","arxiv_id":"2503.03854","repositories_listed":0,"syntology":null},{"url":null,"slug":"label-efficient-lidar-panoptic-segmentation","title":"Label-Efficient LiDAR Panoptic Segmentation","date":"2025-03-04","arxiv_id":"2503.02372","repositories_listed":0,"syntology":null},{"url":null,"slug":"ssnet-saliency-prior-and-state-space-model","title":"SSNet: Saliency Prior and State Space Model-based Network for Salient Object Detection in RGB-D Images","date":"2025-03-04","arxiv_id":"2503.02270","repositories_listed":0,"syntology":null},{"url":null,"slug":"every-sam-drop-counts-embracing-semantic","title":"Every SAM Drop Counts: Embracing Semantic Priors for Multi-Modality Image Fusion and Beyond","date":"2025-03-03","arxiv_id":"2503.01210","repositories_listed":0,"syntology":null},{"url":null,"slug":"opengs-slam-open-set-dense-semantic-slam-with","title":"OpenGS-SLAM: Open-Set Dense Semantic SLAM with 3D Gaussian Splatting for Object-Level Scene Understanding","date":"2025-03-03","arxiv_id":"2503.01646","repositories_listed":0,"syntology":null},{"url":null,"slug":"vs-graphs-integrating-visual-slam-and","title":"vS-Graphs: Integrating Visual SLAM and Situational Graphs through Multi-level Scene Understanding","date":"2025-03-03","arxiv_id":"2503.01783","repositories_listed":0,"syntology":null},{"url":null,"slug":"floorplan-slam-a-real-time-high-accuracy-and","title":"Floorplan-SLAM: A Real-Time, High-Accuracy, and Long-Term Multi-Session Point-Plane SLAM for Efficient Floorplan Reconstruction","date":"2025-03-01","arxiv_id":"2503.00397","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlm-e2e-enhancing-end-to-end-autonomous","title":"VLM-E2E: Enhancing End-to-End Autonomous Driving with Multimodal Driver Attention Fusion","date":"2025-02-25","arxiv_id":"2502.18042","repositories_listed":0,"syntology":null},{"url":null,"slug":"aad-llm-neural-attention-driven-auditory","title":"AAD-LLM: Neural Attention-Driven Auditory Scene Understanding","date":"2025-02-24","arxiv_id":"2502.16794","repositories_listed":0,"syntology":null},{"url":null,"slug":"dr-splat-directly-referring-3d-gaussian","title":"Dr. Splat: Directly Referring 3D Gaussian Splatting via Direct Language Embedding Registration","date":"2025-02-23","arxiv_id":"2502.16652","repositories_listed":0,"syntology":null},{"url":null,"slug":"avd2-accident-video-diffusion-for-accident","title":"AVD2: Accident Video Diffusion for Accident Video Description","date":"2025-02-20","arxiv_id":"2502.14801","repositories_listed":0,"syntology":null},{"url":null,"slug":"sce2drivex-a-generalized-mllm-framework-for","title":"Sce2DriveX: A Generalized MLLM Framework for Scene-to-Drive Learning","date":"2025-02-19","arxiv_id":"2502.14917","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-and-evaluating-hallucinations","title":"Understanding and Evaluating Hallucinations in 3D Visual Language Models","date":"2025-02-18","arxiv_id":"2502.15888","repositories_listed":0,"syntology":null},{"url":null,"slug":"surgical-scene-understanding-in-the-era-of","title":"Surgical Scene Understanding in the Era of Foundation AI Models: A Comprehensive Review","date":"2025-02-16","arxiv_id":"2502.14886","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-grounded-vision-language-framework-for","title":"3D-Grounded Vision-Language Framework for Robotic Task Planning: Automated Prompt Synthesis and Supervised Reasoning","date":"2025-02-13","arxiv_id":"2502.08903","repositories_listed":0,"syntology":null},{"url":null,"slug":"flares-fast-and-accurate-lidar-multi-range","title":"FLARES: Fast and Accurate LiDAR Multi-Range Semantic Segmentation","date":"2025-02-13","arxiv_id":"2502.09274","repositories_listed":0,"syntology":null},{"url":null,"slug":"sshelf-single-shot-hierarchical-extrapolation","title":"sshELF: Single-Shot Hierarchical Extrapolation of Latent Features for 3D Reconstruction from Sparse-Views","date":"2025-02-06","arxiv_id":"2502.04318","repositories_listed":0,"syntology":null},{"url":null,"slug":"mosaic3d-foundation-dataset-and-model-for","title":"Mosaic3D: Foundation Dataset and Model for Open-Vocabulary 3D Segmentation","date":"2025-02-04","arxiv_id":"2502.02548","repositories_listed":0,"syntology":null},{"url":null,"slug":"aquaticclip-a-vision-language-foundation","title":"AquaticCLIP: A Vision-Language Foundation Model for Underwater Scene Analysis","date":"2025-02-03","arxiv_id":"2502.01785","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-lmm-planners-and-3d-skill","title":"Integrating LMM Planners and 3D Skill Policies for Generalizable Manipulation","date":"2025-01-30","arxiv_id":"2501.18733","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-interactive-3d-multi-object-removal","title":"Efficient Interactive 3D Multi-Object Removal","date":"2025-01-29","arxiv_id":"2501.17636","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-self-paced-learning-for-weakly","title":"Contextual Self-paced Learning for Weakly Supervised Spatio-Temporal Video Grounding","date":"2025-01-28","arxiv_id":"2501.17053","repositories_listed":0,"syntology":null},{"url":null,"slug":"physbench-benchmarking-and-enhancing-vision","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","date":"2025-01-27","arxiv_id":"2501.16411","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveiling-the-potential-of-imarkers-invisible","title":"Unveiling the Potential of iMarkers: Invisible Fiducial Markers for Advanced Robotics","date":"2025-01-26","arxiv_id":"2501.15505","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-understanding-enabled-semantic","title":"Scene Understanding Enabled Semantic Communication with Open Channel Coding","date":"2025-01-24","arxiv_id":"2501.14520","repositories_listed":0,"syntology":null},{"url":null,"slug":"geomgs-lidar-guided-geometry-aware-gaussian","title":"GeomGS: LiDAR-Guided Geometry-Aware Gaussian Splatting for Robot Localization","date":"2025-01-23","arxiv_id":"2501.13417","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-radiance-fields-for-the-real-world-a","title":"Neural Radiance Fields for the Real World: A Survey","date":"2025-01-22","arxiv_id":"2501.13104","repositories_listed":0,"syntology":null},{"url":null,"slug":"separated-inter-intra-modal-fusion-prompts","title":"Separated Inter/Intra-Modal Fusion Prompts for Compositional Zero-Shot Learning","date":"2025-01-22","arxiv_id":"2501.17171","repositories_listed":0,"syntology":null},{"url":"/paper/dynamic-scene-understanding-from-vision","slug":"dynamic-scene-understanding-from-vision","title":"Dynamic Scene Understanding from Vision-Language Representations","date":"2025-01-20","arxiv_id":"2501.11653","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-vision-language-framework-for-multispectral","title":"A Vision-Language Framework for Multispectral Scene Representation Using Language-Grounded Features","date":"2025-01-17","arxiv_id":"2501.10144","repositories_listed":0,"syntology":null},{"url":null,"slug":"yeti-yet-to-intervene-proactive-interventions","title":"YETI (YET to Intervene) Proactive Interventions by Multimodal AI Agents in Augmented Reality Tasks","date":"2025-01-16","arxiv_id":"2501.09355","repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-scene-understanding-for-vision","title":"Embodied Scene Understanding for Vision Language Models via MetaVQA","date":"2025-01-15","arxiv_id":"2501.09167","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-scene-understanding-for-automatic","title":"Zero-Shot Scene Understanding for Automatic Target Recognition Using Large Vision-Language Models","date":"2025-01-13","arxiv_id":"2501.07396","repositories_listed":0,"syntology":null},{"url":null,"slug":"application-of-vision-language-model-to","title":"Application of Vision-Language Model to Pedestrians Behavior and Scene Understanding in Autonomous Driving","date":"2025-01-12","arxiv_id":"2501.06680","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniq-unified-decoder-with-task-specific","title":"UniQ: Unified Decoder with Task-specific Queries for Efficient Scene Graph Generation","date":"2025-01-10","arxiv_id":"2501.05687","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-systematic-literature-review-on-deep","title":"A Systematic Literature Review on Deep Learning-based Depth Estimation in Computer Vision","date":"2025-01-09","arxiv_id":"2501.05147","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-language-models-for-autonomous-driving","title":"Vision-Language Models for Autonomous Driving: CLIP-Based Dynamic Scene Understanding","date":"2025-01-09","arxiv_id":"2501.05566","repositories_listed":0,"syntology":null},{"url":null,"slug":"nextstop-an-improved-tracker-for-panoptic","title":"NextStop: An Improved Tracker For Panoptic LIDAR Segmentation Data","date":"2025-01-08","arxiv_id":"2501.06235","repositories_listed":0,"syntology":null}],"record_sha256":"0dbc0c0cf9249564c6bde6a142599183dc54d4b693ce7e1266caed09e7b837a2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}