{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/scene-understanding/papers/8","list_of":"/task/scene-understanding","task":"Scene Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":18,"rows_per_page":100,"rows":[701,800],"of":1723,"counts":{"archive_papers_tagged":1723,"with_a_code_link":720,"where_syntology_ran_a_sample":208,"not_listed_spam_title":0,"listed":1723,"listed_where_code_ran":208,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":26,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":26,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/scene-understanding","prev":"/task/scene-understanding/papers/7","next":"/task/scene-understanding/papers/9","papers":[{"url":"/paper/lost-appearance-invariant-place-recognition","slug":"lost-appearance-invariant-place-recognition","title":"LoST? Appearance-Invariant Place Recognition for Opposite Viewpoints using Visual Semantics","date":"2018-04-16","arxiv_id":"1804.05526","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lost-appearance-invariant-place-recognition#ran","syntology_url":"https://syntology.ai/paper/1804.05526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.05526"}},"official":{"repos":["oravus/lostX"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-rigidity-in-dynamic-scenes-with-a","slug":"learning-rigidity-in-dynamic-scenes-with-a","title":"Learning Rigidity in Dynamic Scenes with a Moving Camera for 3D Motion Field Estimation","date":"2018-04-12","arxiv_id":"1804.04259","repositories_listed":1,"syntology":null},{"url":"/paper/general-purpose-deep-point-cloud-feature","slug":"general-purpose-deep-point-cloud-feature","title":"General-Purpose Deep Point Cloud Feature Extractor","date":"2018-03-12","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/structured-label-inference-for-visual","slug":"structured-label-inference-for-visual","title":"Structured Label Inference for Visual Understanding","date":"2018-02-18","arxiv_id":"1802.06459","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-line-detection-and-its-applications","slug":"semantic-line-detection-and-its-applications","title":"Semantic Line Detection and Its Applications","date":"2017-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/matterport3d-learning-from-rgb-d-data-in","slug":"matterport3d-learning-from-rgb-d-data-in","title":"Matterport3D: Learning from RGB-D Data in Indoor Environments","date":"2017-09-18","arxiv_id":"1709.06158","repositories_listed":1,"syntology":null},{"url":"/paper/fast-scene-understanding-for-autonomous","slug":"fast-scene-understanding-for-autonomous","title":"Fast Scene Understanding for Autonomous Driving","date":"2017-08-08","arxiv_id":"1708.02550","repositories_listed":1,"syntology":null},{"url":"/paper/dual-glance-model-for-deciphering-social","slug":"dual-glance-model-for-deciphering-social","title":"Dual-Glance Model for Deciphering Social Relationships","date":"2017-08-02","arxiv_id":"1708.00634","repositories_listed":1,"syntology":null},{"url":"/paper/scene-graph-generation-from-objects-phrases","slug":"scene-graph-generation-from-objects-phrases","title":"Scene Graph Generation from Objects, Phrases and Region Captions","date":"2017-07-31","arxiv_id":"1707.09700","repositories_listed":1,"syntology":null},{"url":"/paper/deep-video-deblurring-for-hand-held-cameras","slug":"deep-video-deblurring-for-hand-held-cameras","title":"Deep Video Deblurring for Hand-Held Cameras","date":"2017-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-affordance-detection","slug":"weakly-supervised-affordance-detection","title":"Weakly Supervised Affordance Detection","date":"2017-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/3d-semantic-segmentation-of-modular-furniture","slug":"3d-semantic-segmentation-of-modular-furniture","title":"3D Semantic Segmentation of Modular Furniture using rjMCMC","date":"2017-05-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/segan-segmenting-and-generating-the-invisible","slug":"segan-segmenting-and-generating-the-invisible","title":"SeGAN: Segmenting and Generating the Invisible","date":"2017-03-29","arxiv_id":"1703.10239","repositories_listed":1,"syntology":null},{"url":"/paper/da-rnn-semantic-mapping-with-data-associated","slug":"da-rnn-semantic-mapping-with-data-associated","title":"DA-RNN: Semantic Mapping with Data Associated Recurrent Neural Networks","date":"2017-03-09","arxiv_id":"1703.03098","repositories_listed":1,"syntology":null},{"url":"/paper/scannet-richly-annotated-3d-reconstructions","slug":"scannet-richly-annotated-3d-reconstructions","title":"ScanNet: Richly-annotated 3D Reconstructions of Indoor Scenes","date":"2017-02-14","arxiv_id":"1702.04405","repositories_listed":1,"syntology":null},{"url":"/paper/dirty-pixels-optimizing-image-classification","slug":"dirty-pixels-optimizing-image-classification","title":"Dirty Pixels: Towards End-to-End Image Processing and Perception","date":"2017-01-23","arxiv_id":"1701.06487","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/dirty-pixels-optimizing-image-classification#ran","syntology_url":"https://syntology.ai/paper/1701.06487","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1701.06487"}},"official":{"repos":["princeton-computational-imaging/DirtyPixels"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/scenenet-rgb-d-5m-photorealistic-images-of","slug":"scenenet-rgb-d-5m-photorealistic-images-of","title":"SceneNet RGB-D: 5M Photorealistic Images of Synthetic Indoor Trajectories with Ground Truth","date":"2016-12-15","arxiv_id":"1612.05079","repositories_listed":1,"syntology":null},{"url":"/paper/deep-video-deblurring","slug":"deep-video-deblurring","title":"Deep Video Deblurring","date":"2016-11-25","arxiv_id":"1611.08387","repositories_listed":1,"syntology":null},{"url":"/paper/scenenet-understanding-real-world-indoor","slug":"scenenet-understanding-real-world-indoor","title":"SceneNet: Understanding Real World Indoor Scenes With Synthetic Data","date":"2015-11-22","arxiv_id":"1511.07041","repositories_listed":1,"syntology":null},{"url":"/paper/parsing-natural-scenes-and-natural-language","slug":"parsing-natural-scenes-and-natural-language","title":"Parsing Natural Scenes and Natural Language with Recursive Neural Networks","date":"2011-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":null,"slug":"advancing-complex-wide-area-scene","title":"Advancing Complex Wide-Area Scene Understanding with Hierarchical Coresets Selection","date":"2025-07-17","arxiv_id":"2507.13061","repositories_listed":0,"syntology":null},{"url":null,"slug":"argus-leveraging-multiview-images-for","title":"Argus: Leveraging Multiview Images for Improved 3-D Scene Understanding With Large Language Models","date":"2025-07-17","arxiv_id":"2507.12916","repositories_listed":0,"syntology":null},{"url":null,"slug":"city-vlm-towards-multidomain-perception-scene","title":"City-VLM: Towards Multidomain Perception Scene Understanding via Multimodal Incomplete Learning","date":"2025-07-17","arxiv_id":"2507.12795","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-the-signs-a-survey-of-edge-deployable","title":"Seeing the Signs: A Survey of Edge-Deployable OCR Models for Billboard Visibility Analysis","date":"2025-07-15","arxiv_id":"2507.11730","repositories_listed":0,"syntology":null},{"url":null,"slug":"tactical-decision-for-multi-ugv-confrontation","title":"Tactical Decision for Multi-UGV Confrontation with a Vision-Language Model-Based Commander","date":"2025-07-15","arxiv_id":"2507.11079","repositories_listed":0,"syntology":null},{"url":null,"slug":"embrace-3k-embodied-reasoning-and-action-in","title":"EmbRACE-3K: Embodied Reasoning and Action in Complex Environments","date":"2025-07-14","arxiv_id":"2507.10548","repositories_listed":0,"syntology":null},{"url":null,"slug":"muvod-a-novel-multi-view-video-object","title":"MUVOD: A Novel Multi-view Video Object Segmentation Dataset and A Benchmark for 3D Segmentation","date":"2025-07-10","arxiv_id":"2507.07519","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-demands-attention-in-urban-street-scenes","title":"What Demands Attention in Urban Street Scenes? From Scene Understanding towards Road Safety: A Survey of Vision-driven Datasets and Studies","date":"2025-07-09","arxiv_id":"2507.06513","repositories_listed":0,"syntology":null},{"url":null,"slug":"votesplat-hough-voting-gaussian-splatting-for","title":"VoteSplat: Hough Voting Gaussian Splatting for 3D Scene Understanding","date":"2025-06-28","arxiv_id":"2506.22799","repositories_listed":0,"syntology":null},{"url":null,"slug":"copa-sg-dense-scene-graphs-with-parametric","title":"CoPa-SG: Dense Scene Graphs with Parametric and Proto-Relations","date":"2025-06-26","arxiv_id":"2506.21357","repositories_listed":0,"syntology":null},{"url":null,"slug":"case-based-reasoning-augmented-large-language","title":"Case-based Reasoning Augmented Large Language Model Framework for Decision Making in Realistic Safety-Critical Driving Scenarios","date":"2025-06-25","arxiv_id":"2506.20531","repositories_listed":0,"syntology":null},{"url":null,"slug":"dreamanywhere-object-centric-panoramic-3d","title":"DreamAnywhere: Object-Centric Panoramic 3D Scene Generation","date":"2025-06-25","arxiv_id":"2506.20367","repositories_listed":0,"syntology":null},{"url":null,"slug":"ipformer-visual-3d-panoptic-scene-completion","title":"IPFormer: Visual 3D Panoptic Scene Completion with Context-Adaptive Instance Proposals","date":"2025-06-25","arxiv_id":"2506.20671","repositories_listed":0,"syntology":null},{"url":null,"slug":"hoiverse-a-synthetic-scene-graph-dataset-with","title":"HOIverse: A Synthetic Scene Graph Dataset With Human Object Interactions","date":"2025-06-24","arxiv_id":"2506.19639","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-r1-video-grounded-large-language-models","title":"Scene-R1: Video-Grounded Large Language Models for 3D Scene Reasoning without 3D Annotations","date":"2025-06-21","arxiv_id":"2506.17545","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-segmentation-with-large-language-models","title":"Image Segmentation with Large Language Models: A Survey with Perspectives for Intelligent Transportation Systems","date":"2025-06-17","arxiv_id":"2506.14096","repositories_listed":0,"syntology":null},{"url":null,"slug":"leader360v-the-large-scale-real-world-360","title":"Leader360V: The Large-scale, Real-world 360 Video Dataset for Multi-task Learning in Diverse Environment","date":"2025-06-17","arxiv_id":"2506.14271","repositories_listed":0,"syntology":null},{"url":null,"slug":"sceneaware-scene-constrained-pedestrian","title":"SceneAware: Scene-Constrained Pedestrian Trajectory Prediction with LLM-Guided Walkability","date":"2025-06-17","arxiv_id":"2506.14144","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-representation-space-for-3d-visual","title":"Unified Representation Space for 3D Visual Grounding","date":"2025-06-17","arxiv_id":"2506.14238","repositories_listed":0,"syntology":null},{"url":null,"slug":"freeq-graph-free-form-querying-with-semantic","title":"FreeQ-Graph: Free-form Querying with Semantic Consistent Scene Graph for 3D Scene Understanding","date":"2025-06-16","arxiv_id":"2506.13629","repositories_listed":0,"syntology":null},{"url":null,"slug":"scenecompleter-dense-3d-scene-completion-for","title":"SceneCompleter: Dense 3D Scene Completion for Generative Novel View Synthesis","date":"2025-06-12","arxiv_id":"2506.10981","repositories_listed":0,"syntology":null},{"url":null,"slug":"semanticsplat-feed-forward-3d-scene","title":"SemanticSplat: Feed-Forward 3D Scene Understanding with Language-Aware Gaussian Fields","date":"2025-06-11","arxiv_id":"2506.09565","repositories_listed":0,"syntology":null},{"url":null,"slug":"phyblock-a-progressive-benchmark-for-physical","title":"PhyBlock: A Progressive Benchmark for Physical Understanding and Planning via 3D Block Assembly","date":"2025-06-10","arxiv_id":"2506.08708","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-visual-localization-via-semantic","title":"Robust Visual Localization via Semantic-Guided Multi-Scale Transformer","date":"2025-06-10","arxiv_id":"2506.08526","repositories_listed":0,"syntology":null},{"url":null,"slug":"scenesplat-a-large-dataset-and-comprehensive","title":"SceneSplat++: A Large Dataset and Comprehensive Benchmark for Language Gaussian Splatting","date":"2025-06-10","arxiv_id":"2506.08710","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-and-evaluation-of-deep-learning-based","title":"Design and Evaluation of Deep Learning-Based Dual-Spectrum Image Fusion Methods","date":"2025-06-09","arxiv_id":"2506.07779","repositories_listed":0,"syntology":null},{"url":null,"slug":"opensplat3d-open-vocabulary-3d-instance","title":"OpenSplat3D: Open-Vocabulary 3D Instance Segmentation using Gaussian Splatting","date":"2025-06-09","arxiv_id":"2506.07697","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatiallm-training-large-language-models-for","title":"SpatialLM: Training Large Language Models for Structured Indoor Modeling","date":"2025-06-09","arxiv_id":"2506.07491","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-your-3d-encoder-really-work-when","title":"Does Your 3D Encoder Really Work? When Pretrain-SFT from 2D VLMs Meets 3D VLMs","date":"2025-06-05","arxiv_id":"2506.05318","repositories_listed":0,"syntology":null},{"url":null,"slug":"projo4d-progressive-joint-optimization-for","title":"ProJo4D: Progressive Joint Optimization for Sparse-View Inverse Physics Estimation","date":"2025-06-05","arxiv_id":"2506.05317","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-based-transformer-models-for-image","title":"Attention-based transformer models for image captioning across languages: An in-depth survey and evaluation","date":"2025-06-03","arxiv_id":"2506.05399","repositories_listed":0,"syntology":null},{"url":null,"slug":"tactile-mnist-benchmarking-active-tactile","title":"Tactile MNIST: Benchmarking Active Tactile Perception","date":"2025-06-03","arxiv_id":"2506.06361","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-sparsity-for-effective-and-efficient","title":"Learning Sparsity for Effective and Efficient Music Performance Question Answering","date":"2025-06-02","arxiv_id":"2506.01319","repositories_listed":0,"syntology":null},{"url":null,"slug":"sam2-love-segment-anything-model-2-in-1","title":"SAM2-LOVE: Segment Anything Model 2 in Language-aided Audio-Visual Scenes","date":"2025-06-02","arxiv_id":"2506.01558","repositories_listed":0,"syntology":null},{"url":null,"slug":"seg-sr-integrating-semantic-knowledge-into","title":"SeG-SR: Integrating Semantic Knowledge into Remote Sensing Image Super-Resolution via Vision-Language Model","date":"2025-05-29","arxiv_id":"2505.23010","repositories_listed":0,"syntology":null},{"url":null,"slug":"doraemon-decentralized-ontology-aware","title":"DORAEMON: Decentralized Ontology-aware Reliable Agent with Enhanced Memory Oriented Navigation","date":"2025-05-28","arxiv_id":"2505.21969","repositories_listed":0,"syntology":null},{"url":null,"slug":"lidar-based-semantic-perception-for-forklifts","title":"LiDAR Based Semantic Perception for Forklifts in Outdoor Environments","date":"2025-05-28","arxiv_id":"2505.22258","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-graph-completion-method-that-jointly","title":"A Graph Completion Method that Jointly Predicts Geometry and Topology Enables Effective Molecule Assembly","date":"2025-05-27","arxiv_id":"2505.21833","repositories_listed":0,"syntology":null},{"url":null,"slug":"compositional-scene-understanding-through","title":"Compositional Scene Understanding through Inverse Generative Modeling","date":"2025-05-27","arxiv_id":"2505.21780","repositories_listed":0,"syntology":null},{"url":null,"slug":"occle-label-efficient-3d-semantic-occupancy","title":"OccLE: Label-Efficient 3D Semantic Occupancy Prediction","date":"2025-05-27","arxiv_id":"2505.20617","repositories_listed":0,"syntology":null},{"url":null,"slug":"omniindoor3d-comprehensive-indoor-3d","title":"OmniIndoor3D: Comprehensive Indoor 3D Reconstruction","date":"2025-05-27","arxiv_id":"2505.20610","repositories_listed":0,"syntology":null},{"url":null,"slug":"right-side-up-disentangling-orientation","title":"Right Side Up? Disentangling Orientation Understanding in MLLMs with Fine-grained Multi-axis Perception Tasks","date":"2025-05-27","arxiv_id":"2505.21649","repositories_listed":0,"syntology":null},{"url":null,"slug":"underwater-diffusion-attention-network-with","title":"Underwater Diffusion Attention Network with Contrastive Language-Image Joint Learning for Underwater Image Enhancement","date":"2025-05-26","arxiv_id":"2505.19895","repositories_listed":0,"syntology":null},{"url":null,"slug":"co-attendwg-co-attentive-dimension-wise","title":"Co-AttenDWG: Co-Attentive Dimension-Wise Gating and Expert Fusion for Multi-Modal Offensive Content Detection","date":"2025-05-25","arxiv_id":"2505.19010","repositories_listed":0,"syntology":null},{"url":null,"slug":"fhgs-feature-homogenized-gaussian-splatting","title":"FHGS: Feature-Homogenized Gaussian Splatting","date":"2025-05-25","arxiv_id":"2505.19154","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-mllms-guide-me-home-a-benchmark-study-on","title":"Can MLLMs Guide Me Home? A Benchmark Study on Fine-Grained Visual Reasoning from Transit Maps","date":"2025-05-24","arxiv_id":"2505.18675","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-and-generalizable","title":"Self-Supervised and Generalizable Tokenization for CLIP-Based 3D Understanding","date":"2025-05-24","arxiv_id":"2505.18819","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-flight-to-insight-semantic-3d","title":"From Flight to Insight: Semantic 3D Reconstruction for Aerial Inspection via Gaussian Splatting and Language-Guided Segmentation","date":"2025-05-23","arxiv_id":"2505.17402","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-the-generalization-performance-of","title":"Assessing the generalization performance of SAM for ureteroscopy scene understanding","date":"2025-05-22","arxiv_id":"2505.17210","repositories_listed":0,"syntology":null},{"url":null,"slug":"hamf-a-hybrid-attention-mamba-framework-for","title":"HAMF: A Hybrid Attention-Mamba Framework for Joint Scene Context Understanding and Future Motion Representation Learning","date":"2025-05-21","arxiv_id":"2505.15703","repositories_listed":0,"syntology":null},{"url":null,"slug":"razer-robust-accelerated-zero-shot-3d-open","title":"RAZER: Robust Accelerated Zero-Shot 3D Open-Vocabulary Panoptic Reconstruction with Spatio-Temporal Aggregation","date":"2025-05-21","arxiv_id":"2505.15373","repositories_listed":0,"syntology":null},{"url":null,"slug":"robo2vlm-visual-question-answering-from-large","title":"Robo2VLM: Visual Question Answering from Large-Scale In-the-Wild Robot Manipulation Datasets","date":"2025-05-21","arxiv_id":"2505.15517","repositories_listed":0,"syntology":null},{"url":null,"slug":"adatoken-3d-dynamic-spatial-gating-for","title":"AdaToken-3D: Dynamic Spatial Gating for Efficient 3D Large Multimodal-Models Reasoning","date":"2025-05-19","arxiv_id":"2505.12782","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-reaction-time-to-comprehend-scenes","title":"Predicting Reaction Time to Comprehend Scenes with Foveated Scene Understanding Maps","date":"2025-05-19","arxiv_id":"2505.12660","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-large-multimodal-models-understand","title":"Can Large Multimodal Models Understand Agricultural Scenes? Benchmarking with AgroMind","date":"2025-05-18","arxiv_id":"2505.12207","repositories_listed":0,"syntology":null},{"url":null,"slug":"llava-4d-embedding-spatiotemporal-prompt-into","title":"LLaVA-4D: Embedding SpatioTemporal Prompt into LMMs for 4D Scene Understanding","date":"2025-05-18","arxiv_id":"2505.12253","repositories_listed":0,"syntology":null},{"url":null,"slug":"sept-standard-definition-map-enhanced-scene","title":"SEPT: Standard-Definition Map Enhanced Scene Perception and Topology Reasoning for Autonomous Driving","date":"2025-05-18","arxiv_id":"2505.12246","repositories_listed":0,"syntology":null},{"url":null,"slug":"tinyrs-r1-compact-multimodal-language-model","title":"TinyRS-R1: Compact Multimodal Language Model for Remote Sensing","date":"2025-05-17","arxiv_id":"2505.12099","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-beyond-the-scene-enhancing-vision","title":"Seeing Beyond the Scene: Enhancing Vision-Language Models with Interactional Reasoning","date":"2025-05-14","arxiv_id":"2505.09118","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-advances-in-vision-based","title":"Deep Learning Advances in Vision-Based Traffic Accident Anticipation: A Comprehensive Review of Methods,Datasets,and Future Directions","date":"2025-05-12","arxiv_id":"2505.07611","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-cross-spectral-unsupervised-domain","title":"Boosting Cross-spectral Unsupervised Domain Adaptation for Thermal Semantic Segmentation","date":"2025-05-11","arxiv_id":"2505.06951","repositories_listed":0,"syntology":null},{"url":null,"slug":"technical-report-for-icra-2025-goose-2d","title":"Technical Report for ICRA 2025 GOOSE 2D Semantic Segmentation Challenge: Leveraging Color Shift Correction, RoPE-Swin Backbone, and Quantile-based Label Denoising Strategy for Robust Outdoor Scene Understanding","date":"2025-05-11","arxiv_id":"2505.06991","repositories_listed":0,"syntology":null},{"url":null,"slug":"camera-control-at-the-edge-with-language","title":"Camera Control at the Edge with Language Models for Scene Understanding","date":"2025-05-09","arxiv_id":"2505.06402","repositories_listed":0,"syntology":null},{"url":null,"slug":"camera-only-bird-s-eye-view-perception-a","title":"Camera-Only Bird's Eye View Perception: A Neural Approach to LiDAR-Free Environmental Mapping for Autonomous Vehicles","date":"2025-05-09","arxiv_id":"2505.06113","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-clip-perceive-art-the-same-way-we-do","title":"Does CLIP perceive art the same way we do?","date":"2025-05-08","arxiv_id":"2505.05229","repositories_listed":0,"syntology":null},{"url":null,"slug":"padriver-towards-personalized-autonomous","title":"PADriver: Towards Personalized Autonomous Driving","date":"2025-05-08","arxiv_id":"2505.05240","repositories_listed":0,"syntology":null},{"url":null,"slug":"raft-robust-augmentation-of-features-for","title":"RAFT: Robust Augmentation of FeaTures for Image Segmentation","date":"2025-05-07","arxiv_id":"2505.04529","repositories_listed":0,"syntology":null},{"url":null,"slug":"segment-any-rgb-thermal-model-with-language","title":"Segment Any RGB-Thermal Model with Language-aided Distillation","date":"2025-05-04","arxiv_id":"2505.01950","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-recognition-evaluating-visual","title":"Beyond Recognition: Evaluating Visual Perspective Taking in Vision Language Models","date":"2025-05-03","arxiv_id":"2505.03821","repositories_listed":0,"syntology":null},{"url":null,"slug":"embracing-diffraction-a-paradigm-shift-in","title":"Embracing Diffraction: A Paradigm Shift in Wireless Sensing and Communication","date":"2025-05-02","arxiv_id":"2505.01625","repositories_listed":0,"syntology":null},{"url":null,"slug":"v3lma-visual-3d-enhanced-language-model-for","title":"V3LMA: Visual 3D-enhanced Language Model for Autonomous Driving","date":"2025-04-30","arxiv_id":"2505.00156","repositories_listed":0,"syntology":null},{"url":null,"slug":"category-level-and-open-set-object-pose","title":"Category-Level and Open-Set Object Pose Estimation for Robotics","date":"2025-04-28","arxiv_id":"2504.19572","repositories_listed":0,"syntology":null},{"url":null,"slug":"masked-point-entity-contrast-for-open","title":"Masked Point-Entity Contrast for Open-Vocabulary 3D Scene Understanding","date":"2025-04-28","arxiv_id":"2504.19500","repositories_listed":0,"syntology":null},{"url":null,"slug":"travellama-facilitating-multi-modal-large","title":"TraveLLaMA: Facilitating Multi-modal Large Language Models to Understand Urban Scenes and Provide Travel Assistance","date":"2025-04-23","arxiv_id":"2504.16505","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-large-language-models-for-enhanced","title":"Multimodal Large Language Models for Enhanced Traffic Safety: A Comprehensive Review and Future Trends","date":"2025-04-21","arxiv_id":"2504.16134","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-centric-representation-efficient-fine","title":"Vision-Centric Representation-Efficient Fine-Tuning for Robust Universal Foreground Segmentation","date":"2025-04-20","arxiv_id":"2504.14481","repositories_listed":0,"syntology":null},{"url":null,"slug":"haeccity-open-vocabulary-scene-understanding","title":"HAECcity: Open-Vocabulary Scene Understanding of City-Scale Point Clouds with Superpoint Graph Clustering","date":"2025-04-18","arxiv_id":"2504.13590","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-propagation-of-asymmetric-feature","title":"Temporal Propagation of Asymmetric Feature Pyramid for Surgical Scene Segmentation","date":"2025-04-18","arxiv_id":"2504.13440","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-scene-understanding-with","title":"Explainable Scene Understanding with Qualitative Representations and Graph Neural Networks","date":"2025-04-17","arxiv_id":"2504.12817","repositories_listed":0,"syntology":null},{"url":null,"slug":"cags-open-vocabulary-3d-scene-understanding","title":"CAGS: Open-Vocabulary 3D Scene Understanding with Context-Aware Gaussian Splatting","date":"2025-04-16","arxiv_id":"2504.11893","repositories_listed":0,"syntology":null}],"record_sha256":"824b89cc13363d0a805ca0a02cc9e5e4477bd89cee5fbe0a2d48dbf0663769b0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}