{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/scene-understanding/papers/11","list_of":"/task/scene-understanding","task":"Scene Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":18,"rows_per_page":100,"rows":[1001,1100],"of":1723,"counts":{"archive_papers_tagged":1723,"with_a_code_link":720,"where_syntology_ran_a_sample":208,"not_listed_spam_title":0,"listed":1723,"listed_where_code_ran":208,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":26,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":26,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/scene-understanding","prev":"/task/scene-understanding/papers/10","next":"/task/scene-understanding/papers/12","papers":[{"url":null,"slug":"box3d-lightweight-camera-lidar-fusion-for-3d","title":"BOX3D: Lightweight Camera-LiDAR Fusion for 3D Object Detection and Localization","date":"2024-08-27","arxiv_id":"2408.14941","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-occlusion-boundary-estimation","title":"Interactive Occlusion Boundary Estimation through Exploitation of Synthetic Data","date":"2024-08-27","arxiv_id":"2408.15038","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusionsam-latent-space-driven-segment","title":"FusionSAM: Latent Space driven Segment Anything Model for Multimodal Fusion and Segmentation","date":"2024-08-26","arxiv_id":"2408.13980","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-virtfusion-synthetic-3d-data-augmentation","title":"3D-VirtFusion: Synthetic 3D Data Augmentation through Generative Diffusion Models and Controllable Editing","date":"2024-08-25","arxiv_id":"2408.13788","repositories_listed":0,"syntology":null},{"url":null,"slug":"making-large-language-models-better-planners","title":"Making Large Language Models Better Planners with Reasoning-Decision Alignment","date":"2024-08-25","arxiv_id":"2408.13890","repositories_listed":0,"syntology":null},{"url":null,"slug":"neco-improving-dinov2-s-spatial","title":"Near, far: Patch-ordering enhances vision foundation models' scene understanding","date":"2024-08-20","arxiv_id":"2408.11054","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-aware-instance-segmentation-and-tracking","title":"3D-Aware Instance Segmentation and Tracking in Egocentric Videos","date":"2024-08-19","arxiv_id":"2408.09860","repositories_listed":0,"syntology":null},{"url":null,"slug":"scenegpt-a-language-model-for-3d-scene","title":"SceneGPT: A Language Model for 3D Scene Understanding","date":"2024-08-13","arxiv_id":"2408.06926","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectralgaussians-semantic-spectral-3d","title":"SpectralGaussians: Semantic, spectral 3D Gaussian splatting for multi-spectral scene representation, visualization and analysis","date":"2024-08-13","arxiv_id":"2408.06975","repositories_listed":0,"syntology":null},{"url":null,"slug":"helimos-a-dataset-for-moving-object","title":"HeLiMOS: A Dataset for Moving Object Segmentation in 3D Point Clouds From Heterogeneous LiDAR Sensors","date":"2024-08-12","arxiv_id":"2408.06328","repositories_listed":0,"syntology":null},{"url":null,"slug":"spherical-world-locking-for-audio-visual","title":"Spherical World-Locking for Audio-Visual Localization in Egocentric Videos","date":"2024-08-09","arxiv_id":"2408.05364","repositories_listed":0,"syntology":null},{"url":"/paper/complete-3d-relationships-extraction-modality","slug":"complete-3d-relationships-extraction-modality","title":"Complete 3d relationships extraction modality alignment network for 3d dense captioning","date":"2024-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"2407-21267","title":"DEF-oriCORN: efficient 3D scene understanding for robust language-directed manipulation without demonstrations","date":"2024-07-31","arxiv_id":"2407.21267","repositories_listed":0,"syntology":null},{"url":null,"slug":"nis-slam-neural-implicit-semantic-rgb-d-slam","title":"NIS-SLAM: Neural Implicit Semantic RGB-D SLAM for 3D Consistent Scene Understanding","date":"2024-07-30","arxiv_id":"2407.20853","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-rgb-d-fusion-for-semantic","title":"Rethinking RGB-D Fusion for Semantic Segmentation in Surgical Datasets","date":"2024-07-29","arxiv_id":"2407.19714","repositories_listed":0,"syntology":null},{"url":null,"slug":"gp-vls-a-general-purpose-vision-language","title":"GP-VLS: A general-purpose vision language model for surgery","date":"2024-07-27","arxiv_id":"2407.19305","repositories_listed":0,"syntology":null},{"url":null,"slug":"answerability-fields-answerable-location","title":"Answerability Fields: Answerable Location Estimation via Diffusion Models","date":"2024-07-26","arxiv_id":"2407.18497","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-question-answering-for-city-scene","title":"3D Question Answering for City Scene Understanding","date":"2024-07-24","arxiv_id":"2407.17398","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmented-efficiency-reducing-memory","title":"Augmented Efficiency: Reducing Memory Footprint and Accelerating Inference for 3D Semantic Segmentation through Hybrid Vision","date":"2024-07-23","arxiv_id":"2407.16102","repositories_listed":0,"syntology":null},{"url":null,"slug":"inlut3d-challenging-real-indoor-dataset-for","title":"InLUT3D: Challenging real indoor dataset for point cloud analysis","date":"2024-07-22","arxiv_id":"2408.03338","repositories_listed":0,"syntology":null},{"url":null,"slug":"videogamebunny-towards-vision-assistants-for","title":"VideoGameBunny: Towards vision assistants for video games","date":"2024-07-21","arxiv_id":"2407.15295","repositories_listed":0,"syntology":null},{"url":null,"slug":"gaussianbev-3d-gaussian-representation-meets","title":"GaussianBeV: 3D Gaussian Representation meets Perception Models for BeV Segmentation","date":"2024-07-19","arxiv_id":"2407.14108","repositories_listed":0,"syntology":null},{"url":null,"slug":"opensu3d-open-world-3d-scene-understanding","title":"OpenSU3D: Open World 3D Scene Understanding using Foundation Models","date":"2024-07-19","arxiv_id":"2407.14279","repositories_listed":0,"syntology":null},{"url":"/paper/open-vocabulary-3d-scene-understanding-via","slug":"open-vocabulary-3d-scene-understanding-via","title":"Open Vocabulary 3D Scene Understanding via Geometry Guided Self-Distillation","date":"2024-07-18","arxiv_id":"2407.13362","repositories_listed":0,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/open-vocabulary-3d-scene-understanding-via#ran","syntology_url":"https://syntology.ai/paper/2407.13362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13362"}},"official":null}},{"url":null,"slug":"training-free-model-merging-for-multi-target","title":"Training-Free Model Merging for Multi-target Domain Adaptation","date":"2024-07-18","arxiv_id":"2407.13771","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-vision-language-models-for","title":"Benchmarking Vision Language Models for Cultural Understanding","date":"2024-07-15","arxiv_id":"2407.10920","repositories_listed":0,"syntology":null},{"url":null,"slug":"dense-multimodal-alignment-for-open","title":"Dense Multimodal Alignment for Open-Vocabulary 3D Scene Understanding","date":"2024-07-13","arxiv_id":"2407.09781","repositories_listed":0,"syntology":null},{"url":null,"slug":"blos-bev-navigation-map-enhanced-lane","title":"BLOS-BEV: Navigation Map Enhanced Lane Segmentation Network, Beyond Line of Sight","date":"2024-07-11","arxiv_id":"2407.08526","repositories_listed":0,"syntology":null},{"url":"/paper/pareto-low-rank-adapters-efficient-multi-task","slug":"pareto-low-rank-adapters-efficient-multi-task","title":"Pareto Low-Rank Adapters: Efficient Multi-Task Learning with Preferences","date":"2024-07-10","arxiv_id":"2407.08056","repositories_listed":0,"syntology":{"n":11,"n_ran":7,"n_constructed":7,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 7 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 7 samples that ran constructed an object rather than computing a result","sample_list":"/paper/pareto-low-rank-adapters-efficient-multi-task#ran","syntology_url":"https://syntology.ai/paper/2407.08056","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08056"}},"official":null}},{"url":null,"slug":"joint-prototype-and-coefficient-prediction","title":"Joint prototype and coefficient prediction for 3D instance segmentation","date":"2024-07-09","arxiv_id":"2407.06958","repositories_listed":0,"syntology":null},{"url":null,"slug":"lvlm-empowered-multi-modal-representation","title":"LVLM-empowered Multi-modal Representation Learning for Visual Place Recognition","date":"2024-07-09","arxiv_id":"2407.06730","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-learning-via-cluster-distance","title":"Self-supervised Learning via Cluster Distance Prediction for Operating Room Context Awareness","date":"2024-07-07","arxiv_id":"2407.05448","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-primal-sketch-combining-analogy","title":"Hybrid Primal Sketch: Combining Analogy, Qualitative Representations, and Computer Vision for Scene Understanding","date":"2024-07-05","arxiv_id":"2407.04859","repositories_listed":0,"syntology":null},{"url":null,"slug":"panopticrecon-leverage-open-vocabulary","title":"PanopticRecon: Leverage Open-vocabulary Instance Segmentation for Zero-shot Panoptic Reconstruction","date":"2024-07-01","arxiv_id":"2407.01349","repositories_listed":0,"syntology":null},{"url":null,"slug":"esgnn-towards-equivariant-scene-graph-neural","title":"ESGNN: Towards Equivariant Scene Graph Neural Network for 3D Scene Understanding","date":"2024-06-30","arxiv_id":"2407.00609","repositories_listed":0,"syntology":null},{"url":null,"slug":"egogaussian-dynamic-scene-understanding-from","title":"EgoGaussian: Dynamic Scene Understanding from Egocentric Video with 3D Gaussian Splatting","date":"2024-06-28","arxiv_id":"2406.19811","repositories_listed":0,"syntology":null},{"url":null,"slug":"pptformer-pseudo-multi-perspective","title":"PPTFormer: Pseudo Multi-Perspective Transformer for UAV Segmentation","date":"2024-06-28","arxiv_id":"2406.19632","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-mvp-3d-multiview-pretraining-for-robotic","title":"3D-MVP: 3D Multiview Pretraining for Robotic Manipulation","date":"2024-06-26","arxiv_id":"2406.18158","repositories_listed":0,"syntology":null},{"url":null,"slug":"gpt-4v-explorations-mining-autonomous-driving","title":"GPT-4V Explorations: Mining Autonomous Driving","date":"2024-06-24","arxiv_id":"2406.16817","repositories_listed":0,"syntology":null},{"url":null,"slug":"evsegsnn-neuromorphic-semantic-segmentation","title":"EvSegSNN: Neuromorphic Semantic Segmentation for Event Data","date":"2024-06-20","arxiv_id":"2406.14178","repositories_listed":0,"syntology":null},{"url":null,"slug":"distillnerf-perceiving-3d-scenes-from-single","title":"DistillNeRF: Perceiving 3D Scenes from Single-Glance Images by Distilling Neural Fields and Foundation Model Features","date":"2024-06-17","arxiv_id":"2406.12095","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-generalizability-of-representation","title":"Enhancing Generalizability of Representation Learning for Data-Efficient 3D Scene Understanding","date":"2024-06-17","arxiv_id":"2406.11283","repositories_listed":0,"syntology":null},{"url":null,"slug":"mapvision-cvpr-2024-autonomous-grand","title":"MapVision: CVPR 2024 Autonomous Grand Challenge Mapless Driving Tech Report","date":"2024-06-14","arxiv_id":"2406.10125","repositories_listed":0,"syntology":null},{"url":null,"slug":"fastlgs-speeding-up-language-embedded","title":"FastLGS: Speeding up Language Embedded Gaussians with Feature Grid Mapping","date":"2024-06-04","arxiv_id":"2406.01916","repositories_listed":0,"syntology":null},{"url":null,"slug":"cyclo-cyclic-graph-transformer-approach-to","title":"CYCLO: Cyclic Graph Transformer Approach to Multi-Object Relationship Modeling in Aerial Videos","date":"2024-06-03","arxiv_id":"2406.01029","repositories_listed":0,"syntology":null},{"url":null,"slug":"eagle-efficient-adaptive-geometry-based","title":"EAGLE: Efficient Adaptive Geometry-based Learning in Cross-view Understanding","date":"2024-06-03","arxiv_id":"2406.01429","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-aware-egocentric-online-action","title":"Object Aware Egocentric Online Action Detection","date":"2024-06-03","arxiv_id":"2406.01079","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-video-semantic-segmentation-1","title":"Semi-supervised Video Semantic Segmentation Using Unreliable Pseudo Labels for PVUW2024","date":"2024-06-02","arxiv_id":"2406.00587","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-3d-robotics-perception-using","title":"Learning 3D Robotics Perception using Inductive Priors","date":"2024-05-30","arxiv_id":"2405.20364","repositories_listed":0,"syntology":null},{"url":"/paper/sam-e-leveraging-visual-foundation-model-with","slug":"sam-e-leveraging-visual-foundation-model-with","title":"SAM-E: Leveraging Visual Foundation Model with Sequence Imitation for Embodied Manipulation","date":"2024-05-30","arxiv_id":"2405.19586","repositories_listed":0,"syntology":null},{"url":null,"slug":"kestrel-point-grounding-multimodal-llm-for","title":"Kestrel: Point Grounding Multimodal LLM for Part-Aware 3D Vision-Language Understanding","date":"2024-05-29","arxiv_id":"2405.18937","repositories_listed":0,"syntology":null},{"url":null,"slug":"goi-find-3d-gaussians-of-interest-with-an","title":"GOI: Find 3D Gaussians of Interest with an Optimizable Open-vocabulary Semantic-space Hyperplane","date":"2024-05-27","arxiv_id":"2405.17596","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-vocabulary-sam3d-understand-any-3d-scene","title":"Open-Vocabulary SAM3D: Towards Training-free Open-Vocabulary 3D Scene Understanding","date":"2024-05-24","arxiv_id":"2405.15580","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-camera-dolly-extreme-monocular","title":"Generative Camera Dolly: Extreme Monocular Dynamic Novel View Synthesis","date":"2024-05-23","arxiv_id":"2405.14868","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformers-for-image-goal-navigation","title":"Transformers for Image-Goal Navigation","date":"2024-05-23","arxiv_id":"2405.14128","repositories_listed":0,"syntology":null},{"url":null,"slug":"gamevlm-a-decision-making-framework-for","title":"GameVLM: A Decision-making Framework for Robotic Task Planning Based on Visual Language Models and Zero-sum Games","date":"2024-05-22","arxiv_id":"2405.13751","repositories_listed":0,"syntology":null},{"url":null,"slug":"ts40k-a-3d-point-cloud-dataset-of-rural","title":"TS40K: a 3D Point Cloud Dataset of Rural Terrain and Electrical Transmission System","date":"2024-05-22","arxiv_id":"2405.13989","repositories_listed":0,"syntology":null},{"url":null,"slug":"anticipating-object-state-changes","title":"Anticipating Object State Changes in Long Procedural Videos","date":"2024-05-21","arxiv_id":"2405.12789","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-preprocessing-and-postprocessing-voxel","title":"A Preprocessing and Postprocessing Voxel-based Method for LiDAR Semantic Segmentation Improvement in Long Distance","date":"2024-05-16","arxiv_id":"2405.10046","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-shape-augmentation-with-content-aware","title":"3D Shape Augmentation with Content-Aware Shape Resizing","date":"2024-05-15","arxiv_id":"2405.09050","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavior-vision-suite-customizable-dataset","title":"BEHAVIOR Vision Suite: Customizable Dataset Generation via Simulation","date":"2024-05-15","arxiv_id":"2405.09546","repositories_listed":0,"syntology":null},{"url":null,"slug":"driveworld-4d-pre-trained-scene-understanding","title":"DriveWorld: 4D Pre-trained Scene Understanding via World Models for Autonomous Driving","date":"2024-05-07","arxiv_id":"2405.04390","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-groundcam-quantifying-grounding-in-vision","title":"Q-GroundCAM: Quantifying Grounding in Vision Language Models via GradCAM","date":"2024-04-29","arxiv_id":"2404.19128","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-beyond-classes-zero-shot-grounded","title":"Seeing Beyond Classes: Zero-Shot Grounded Situation Recognition via Language Explainer","date":"2024-04-24","arxiv_id":"2404.15785","repositories_listed":0,"syntology":null},{"url":null,"slug":"cloudfort-enhancing-robustness-of-3d-point","title":"CloudFort: Enhancing Robustness of 3D Point Cloud Classification Against Backdoor Attacks via Spatial Partitioning and Ensemble Prediction","date":"2024-04-22","arxiv_id":"2404.14042","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-support-relations-inference-and-scene","title":"On Support Relations Inference and Scene Hierarchy Graph Construction from Point Cloud in Clustered Environments","date":"2024-04-22","arxiv_id":"2404.13842","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-scene-representation-and","title":"Unified Scene Representation and Reconstruction for 3D Large Language Models","date":"2024-04-19","arxiv_id":"2404.13044","repositories_listed":0,"syntology":null},{"url":null,"slug":"accidentblip2-accident-detection-with-multi","title":"AccidentBlip: Agent of Accident Warning based on MA-former","date":"2024-04-18","arxiv_id":"2404.12149","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-3d-object-detection-on-unseen","title":"Multimodal 3D Object Detection on Unseen Domains","date":"2024-04-17","arxiv_id":"2404.11764","repositories_listed":0,"syntology":null},{"url":null,"slug":"pregsu-a-generalized-traffic-scene","title":"PreGSU-A Generalized Traffic Scene Understanding Model for Autonomous Driving based on Pre-trained Graph Attention Network","date":"2024-04-16","arxiv_id":"2404.10263","repositories_listed":0,"syntology":null},{"url":null,"slug":"depth-estimation-using-weighted-loss-and","title":"Depth Estimation using Weighted-loss and Transfer Learning","date":"2024-04-11","arxiv_id":"2404.07686","repositories_listed":0,"syntology":null},{"url":null,"slug":"gaga-group-any-gaussians-via-3d-aware-memory","title":"Gaga: Group Any Gaussians via 3D-aware Memory Bank","date":"2024-04-11","arxiv_id":"2404.07977","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-explanations-into-human-machine","title":"Incorporating Explanations into Human-Machine Interfaces for Trust and Situation Awareness in Autonomous Vehicles","date":"2024-04-10","arxiv_id":"2404.07383","repositories_listed":0,"syntology":null},{"url":null,"slug":"o2v-mapping-online-open-vocabulary-mapping","title":"O2V-Mapping: Online Open-Vocabulary Mapping with Neural Implicit Representation","date":"2024-04-10","arxiv_id":"2404.06836","repositories_listed":0,"syntology":null},{"url":null,"slug":"daf-bevseg-distortion-aware-fisheye-camera","title":"DaF-BEVSeg: Distortion-aware Fisheye Camera based Bird's Eye View Segmentation with Occlusion Reasoning","date":"2024-04-09","arxiv_id":"2404.06352","repositories_listed":0,"syntology":null},{"url":null,"slug":"questmaps-queryable-semantic-topological-maps","title":"QueSTMaps: Queryable Semantic Topological Maps for 3D Scene Understanding","date":"2024-04-09","arxiv_id":"2404.06442","repositories_listed":0,"syntology":null},{"url":null,"slug":"panoptic-perception-a-novel-task-and-fine","title":"Panoptic Perception: A Novel Task and Fine-grained Dataset for Universal Remote Sensing Image Interpretation","date":"2024-04-06","arxiv_id":"2404.04608","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-only-scan-once-a-dynamic-scene","title":"You Only Scan Once: A Dynamic Scene Reconstruction Pipeline for 6-DoF Robotic Grasping of Novel Objects","date":"2024-04-04","arxiv_id":"2404.03462","repositories_listed":0,"syntology":null},{"url":"/paper/360-x-a-panoptic-multi-modal-scene","slug":"360-x-a-panoptic-multi-modal-scene","title":"360+x: A Panoptic Multi-modal Scene Understanding Dataset","date":"2024-04-01","arxiv_id":"2404.00989","repositories_listed":0,"syntology":null},{"url":null,"slug":"mm3dgs-slam-multi-modal-3d-gaussian-splatting","title":"MM3DGS SLAM: Multi-modal 3D Gaussian Splatting for SLAM Using Vision, Depth, and Inertial Measurements","date":"2024-04-01","arxiv_id":"2404.00923","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-to-length-shift-flexilength-network","title":"Adapting to Length Shift: FlexiLength Network for Trajectory Prediction","date":"2024-03-31","arxiv_id":"2404.00742","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-radiance-field-based-visual-rendering","title":"Neural Radiance Field-based Visual Rendering: A Comprehensive Review","date":"2024-03-31","arxiv_id":"2404.00714","repositories_listed":0,"syntology":null},{"url":null,"slug":"hgs-mapping-online-dense-mapping-using-hybrid","title":"HGS-Mapping: Online Dense Mapping Using Hybrid Gaussian Representation in Urban Scenes","date":"2024-03-29","arxiv_id":"2403.20159","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-3d-instance-mapping-and","title":"Efficient 3D Instance Mapping and Localization with Neural Fields","date":"2024-03-28","arxiv_id":"2403.19797","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-trustworthy-automated-driving-through","title":"Towards Trustworthy Automated Driving through Qualitative Scene Understanding and Explanations","date":"2024-03-25","arxiv_id":"2403.16908","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-learning-with-multi-task","title":"Multi-Task Learning with Multi-Task Optimization","date":"2024-03-24","arxiv_id":"2403.16162","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-is-enough-only-semantic-information","title":"Semantic Is Enough: Only Semantic Information For NeRF Reconstruction","date":"2024-03-24","arxiv_id":"2403.16043","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusionmtl-learning-multi-task-denoising","title":"DiffusionMTL: Learning Multi-Task Denoising Diffusion Model from Partially Annotated Data","date":"2024-03-22","arxiv_id":"2403.15389","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-gaussians-open-vocabulary-scene","title":"Semantic Gaussians: Open-Vocabulary Scene Understanding with 3D Gaussian Splatting","date":"2024-03-22","arxiv_id":"2403.15624","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-object-detection-from-point-cloud-via","title":"3D Object Detection from Point Cloud via Voting Step Diffusion","date":"2024-03-21","arxiv_id":"2403.14133","repositories_listed":0,"syntology":null},{"url":null,"slug":"exosense-a-vision-centric-scene-understanding","title":"Exosense: A Vision-Based Scene Understanding System For Exoskeletons","date":"2024-03-21","arxiv_id":"2403.14320","repositories_listed":0,"syntology":null},{"url":null,"slug":"surroundsdf-implicit-3d-scene-understanding","title":"SurroundSDF: Implicit 3D Scene Understanding Based on Signed Distance Field","date":"2024-03-21","arxiv_id":"2403.14366","repositories_listed":0,"syntology":null},{"url":null,"slug":"geometric-constraints-in-deep-learning","title":"Geometric Constraints in Deep Learning Frameworks: A Survey","date":"2024-03-19","arxiv_id":"2403.12431","repositories_listed":0,"syntology":null},{"url":null,"slug":"hugs-holistic-urban-3d-scene-understanding","title":"HUGS: Holistic Urban 3D Scene Understanding via Gaussian Splatting","date":"2024-03-19","arxiv_id":"2403.12722","repositories_listed":0,"syntology":null},{"url":null,"slug":"m2da-multi-modal-fusion-transformer","title":"M2DA: Multi-Modal Fusion Transformer Incorporating Driver Attention for Autonomous Driving","date":"2024-03-19","arxiv_id":"2403.12552","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent3d-zero-an-agent-for-zero-shot-3d","title":"Agent3D-Zero: An Agent for Zero-shot 3D Understanding","date":"2024-03-18","arxiv_id":"2403.11835","repositories_listed":0,"syntology":null},{"url":null,"slug":"r3ds-reality-linked-3d-scenes-for-panoramic","title":"R3DS: Reality-linked 3D Scenes for Panoramic Scene Understanding","date":"2024-03-18","arxiv_id":"2403.12301","repositories_listed":0,"syntology":null},{"url":null,"slug":"urban-scene-diffusion-through-semantic","title":"Urban Scene Diffusion through Semantic Occupancy Map","date":"2024-03-18","arxiv_id":"2403.11697","repositories_listed":0,"syntology":null},{"url":null,"slug":"n2f2-hierarchical-scene-understanding-with","title":"N2F2: Hierarchical Scene Understanding with Nested Neural Feature Fields","date":"2024-03-16","arxiv_id":"2403.10997","repositories_listed":0,"syntology":null},{"url":null,"slug":"segment-any-object-model-saom-real-to","title":"Segment Any Object Model (SAOM): Real-to-Simulation Fine-Tuning Strategy for Multi-Class Multi-Instance Segmentation","date":"2024-03-16","arxiv_id":"2403.10780","repositories_listed":0,"syntology":null}],"record_sha256":"33d5080ba7ab8aacd53c53efe7e1b7ef94fff42bb3399ce8cc681f6774aea18d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}