{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/scene-understanding/papers/10","list_of":"/task/scene-understanding","task":"Scene Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":10,"pages_in_order":18,"rows_per_page":100,"rows":[901,1000],"of":1723,"counts":{"archive_papers_tagged":1723,"with_a_code_link":720,"where_syntology_ran_a_sample":208,"not_listed_spam_title":0,"listed":1723,"listed_where_code_ran":208,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":26,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":26,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/scene-understanding","prev":"/task/scene-understanding/papers/9","next":"/task/scene-understanding/papers/11","papers":[{"url":null,"slug":"tadformer-task-adaptive-dynamic-transformer","title":"TADFormer : Task-Adaptive Dynamic Transformer for Efficient Multi-Task Learning","date":"2025-01-08","arxiv_id":"2501.04293","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-the-understanding-of-fine-grained","title":"Advancing the Understanding of Fine-Grained 3D Forest Structures using Digital Cousins and Simulation-to-Reality: Methods and Datasets","date":"2025-01-07","arxiv_id":"2501.03637","repositories_listed":0,"syntology":null},{"url":null,"slug":"cl3dor-contrastive-learning-for-3d-large","title":"CL3DOR: Contrastive Learning for 3D Large Multimodal Models via Odds Ratio on High-Resolution Point Clouds","date":"2025-01-07","arxiv_id":"2501.03879","repositories_listed":0,"syntology":null},{"url":null,"slug":"largead-large-scale-cross-sensor-data","title":"LargeAD: Large-Scale Cross-Sensor Data Pretraining for Autonomous Driving","date":"2025-01-07","arxiv_id":"2501.04005","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-llava-towards-generalist-3d-lmms-with-omni","title":"3D-LLaVA: Towards Generalist 3D LMMs with Omni Superpoint Transformer","date":"2025-01-02","arxiv_id":"2501.01163","repositories_listed":0,"syntology":null},{"url":null,"slug":"leverage-cross-attention-for-end-to-end-open","title":"Leverage Cross-Attention for End-to-End Open-Vocabulary Panoptic Reconstruction","date":"2025-01-02","arxiv_id":"2501.01119","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-mvp-3d-multiview-pretraining-for","title":"3D-MVP: 3D Multiview Pretraining for Manipulation","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-human-perception-understanding-multi","title":"Beyond Human Perception: Understanding Multi-Object World from Monocular View","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"groundingface-fine-grained-face-understanding","title":"GroundingFace: Fine-grained Face Understanding via Pixel Grounding Multimodal Large Language Model","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hush-holistic-panoramic-3d-scene","title":"HUSH: Holistic Panoramic 3D Scene Understanding using Spherical Harmonics","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-map-based-prompt-tuning-for-navigation","title":"Scene Map-based Prompt Tuning for Navigation Instruction Generation","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tadformer-task-adaptive-dynamic-transformer-1","title":"TADFormer: Task-Adaptive Dynamic TransFormer for Efficient Multi-Task Learning","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-language-embodiment-for-monocular","title":"Vision-Language Embodiment for Monocular Depth Estimation","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-videoagent-persistent-memory-from","title":"Embodied VideoAgent: Persistent Memory from Egocentric Videos and Embodied Sensors Enables Dynamic Scene Understanding","date":"2024-12-31","arxiv_id":"2501.00358","repositories_listed":0,"syntology":null},{"url":null,"slug":"ovgaussian-generalizable-3d-gaussian","title":"OVGaussian: Generalizable 3D Gaussian Segmentation with Open Vocabularies","date":"2024-12-31","arxiv_id":"2501.00326","repositories_listed":0,"syntology":null},{"url":null,"slug":"4d-gaussian-splatting-modeling-dynamic-scenes","title":"4D Gaussian Splatting: Modeling Dynamic Scenes with Native 4D Primitives","date":"2024-12-30","arxiv_id":"2412.20720","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-image-gan-with-pretrained","title":"Text-to-Image GAN with Pretrained Representations","date":"2024-12-30","arxiv_id":"2501.00116","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniplv-towards-label-efficient-open-world-3d","title":"UniPLV: Towards Label-Efficient Open-World 3D Scene Understanding by Regional Visual Language Supervision","date":"2024-12-24","arxiv_id":"2412.18131","repositories_listed":0,"syntology":null},{"url":null,"slug":"langsurf-language-embedded-surface-gaussians","title":"LangSurf: Language-Embedded Surface Gaussians for 3D Scene Understanding","date":"2024-12-23","arxiv_id":"2412.17635","repositories_listed":0,"syntology":null},{"url":null,"slug":"application-of-multimodal-large-language","title":"Application of Multimodal Large Language Models in Autonomous Driving","date":"2024-12-21","arxiv_id":"2412.16410","repositories_listed":0,"syntology":null},{"url":null,"slug":"objvariantensemble-advancing-point-cloud-llm","title":"ObjVariantEnsemble: Advancing Point Cloud LLM Evaluation in Challenging Scenes with Subtly Distinguished Objects","date":"2024-12-19","arxiv_id":"2412.14837","repositories_listed":0,"syntology":null},{"url":null,"slug":"gags-granularity-aware-feature-distillation","title":"GAGS: Granularity-Aware Feature Distillation for Language Gaussian Splatting","date":"2024-12-18","arxiv_id":"2412.13654","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-view-pedestrian-occupancy-prediction","title":"Multi-View Pedestrian Occupancy Prediction with a Novel Synthetic Dataset","date":"2024-12-18","arxiv_id":"2412.13569","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-enhanced-classification-method-based-on","title":"An Enhanced Classification Method Based on Adaptive Multi-Scale Fusion for Long-tailed Multispectral Point Clouds","date":"2024-12-16","arxiv_id":"2412.11407","repositories_listed":0,"syntology":null},{"url":null,"slug":"supergseg-open-vocabulary-3d-segmentation","title":"SuperGSeg: Open-Vocabulary 3D Segmentation with Structured Super-Gaussians","date":"2024-12-13","arxiv_id":"2412.10231","repositories_listed":0,"syntology":null},{"url":null,"slug":"magic-mastering-physical-adversarial","title":"MAGIC: Mastering Physical Adversarial Generation in Context through Collaborative LLM Agents","date":"2024-12-11","arxiv_id":"2412.08014","repositories_listed":0,"syntology":null},{"url":null,"slug":"slgaussian-fast-language-gaussian-splatting","title":"SLGaussian: Fast Language Gaussian Splatting in Sparse Views","date":"2024-12-11","arxiv_id":"2412.08331","repositories_listed":0,"syntology":null},{"url":null,"slug":"tgospa-metric-parameters-selection-and","title":"TGOSPA Metric Parameters Selection and Evaluation for Visual Multi-object Tracking","date":"2024-12-11","arxiv_id":"2412.08321","repositories_listed":0,"syntology":null},{"url":null,"slug":"event-fields-capturing-light-fields-at-high","title":"Event fields: Capturing light fields at high speed, resolution, and dynamic range","date":"2024-12-09","arxiv_id":"2412.06191","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-lexicon-rich-image-features-in","title":"Visual Lexicon: Rich Image Features in Language Space","date":"2024-12-09","arxiv_id":"2412.06774","repositories_listed":0,"syntology":null},{"url":null,"slug":"tb-hsu-hierarchical-3d-scene-understanding","title":"TB-HSU: Hierarchical 3D Scene Understanding with Contextual Affordances","date":"2024-12-07","arxiv_id":"2412.05596","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-the-performance-of-ct-image","title":"Assessing the performance of CT image denoisers using Laguerre-Gauss Channelized Hotelling Observer for lesion detection","date":"2024-12-04","arxiv_id":"2412.02920","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-dnns-for-a-trade-off-between","title":"Designing DNNs for a trade-off between robustness and processing performance in embedded devices","date":"2024-12-04","arxiv_id":"2412.03682","repositories_listed":0,"syntology":null},{"url":null,"slug":"bye-build-your-encoder-with-one-sequence-of","title":"BYE: Build Your Encoder with One Sequence of Exploration Data for Long-Term Dynamic Scene Understanding","date":"2024-12-03","arxiv_id":"2412.02449","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparselgs-sparse-view-language-embedded","title":"SparseLGS: Sparse View Language Embedded Gaussian Splatting","date":"2024-12-03","arxiv_id":"2412.02245","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-semantic-communication-system-for-real-time","title":"A Semantic Communication System for Real-time 3D Reconstruction Tasks","date":"2024-12-02","arxiv_id":"2412.01191","repositories_listed":0,"syntology":null},{"url":null,"slug":"holistic-understanding-of-3d-scenes-as","title":"Holistic Understanding of 3D Scenes as Universal Scene Description","date":"2024-12-02","arxiv_id":"2412.01398","repositories_listed":0,"syntology":null},{"url":null,"slug":"occam-s-lgs-a-simple-approach-for-language","title":"Occam's LGS: A Simple Approach for Language Gaussian Splatting","date":"2024-12-02","arxiv_id":"2412.01807","repositories_listed":0,"syntology":null},{"url":null,"slug":"chatsplat-3d-conversational-gaussian","title":"ChatSplat: 3D Conversational Gaussian Splatting","date":"2024-12-01","arxiv_id":"2412.00734","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantifying-the-synthetic-and-real-domain-gap","title":"Quantifying the synthetic and real domain gap in aerial scene understanding","date":"2024-11-29","arxiv_id":"2411.19913","repositories_listed":0,"syntology":null},{"url":null,"slug":"sims-simulating-human-scene-interactions-with","title":"SIMS: Simulating Stylized Human-Scene Interactions with Retrieval-Augmented Script Generation","date":"2024-11-29","arxiv_id":"2411.19921","repositories_listed":0,"syntology":null},{"url":null,"slug":"instancegaussian-appearance-semantic-joint","title":"InstanceGaussian: Appearance-Semantic Joint Gaussian Representation for 3D Instance-Level Perception","date":"2024-11-28","arxiv_id":"2411.19235","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-chip-hyperspectral-image-segmentation-with","title":"On-chip Hyperspectral Image Segmentation with Fully Convolutional Networks for Scene Understanding in Autonomous Driving","date":"2024-11-28","arxiv_id":"2411.19274","repositories_listed":0,"syntology":null},{"url":null,"slug":"scenetap-scene-coherent-typographic","title":"SceneTAP: Scene-Coherent Typographic Adversarial Planner against Vision-Language Models in Real-World Environments","date":"2024-11-28","arxiv_id":"2412.00114","repositories_listed":0,"syntology":null},{"url":null,"slug":"reconstructing-animals-and-the-wild","title":"Reconstructing Animals and the Wild","date":"2024-11-27","arxiv_id":"2411.18807","repositories_listed":0,"syntology":null},{"url":"/paper/hsi-drive-v2-0-more-data-for-new-challenges","slug":"hsi-drive-v2-0-more-data-for-new-challenges","title":"HSI-Drive v2.0: More Data for New Challenges in Scene Understanding for Autonomous Driving","date":"2024-11-26","arxiv_id":"2411.17530","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-vocabulary-octree-graph-for-3d-scene","title":"Open-Vocabulary Octree-Graph for 3D Scene Understanding","date":"2024-11-25","arxiv_id":"2411.16253","repositories_listed":0,"syntology":null},{"url":null,"slug":"robospatial-teaching-spatial-understanding-to","title":"RoboSpatial: Teaching Spatial Understanding to 2D and 3D Vision-Language Models for Robotics","date":"2024-11-25","arxiv_id":"2411.16537","repositories_listed":0,"syntology":null},{"url":null,"slug":"unigaussian-driving-scene-reconstruction-from","title":"UniGaussian: Driving Scene Reconstruction from Multiple Camera Models via Unified Gaussian Representations","date":"2024-11-22","arxiv_id":"2411.15355","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-3d-reasoning-segmentation-with","title":"Multimodal 3D Reasoning Segmentation with Complex Scenes","date":"2024-11-21","arxiv_id":"2411.13927","repositories_listed":0,"syntology":null},{"url":null,"slug":"classification-of-geographical-land-structure","title":"Classification of Geographical Land Structure Using Convolution Neural Network and Transfer Learning","date":"2024-11-19","arxiv_id":"2411.12415","repositories_listed":0,"syntology":null},{"url":null,"slug":"calibrated-and-efficient-sampling-free","title":"Calibrated and Efficient Sampling-Free Confidence Estimation for LiDAR Scene Semantic Segmentation","date":"2024-11-18","arxiv_id":"2411.11935","repositories_listed":0,"syntology":null},{"url":null,"slug":"mgnicenet-unified-monocular-geometric-scene","title":"MGNiceNet: Unified Monocular Geometric Scene Understanding","date":"2024-11-18","arxiv_id":"2411.11466","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-label-dependency-for-underwater","title":"Reducing Label Dependency for Underwater Scene Understanding: A Survey of Datasets, Techniques and Applications","date":"2024-11-18","arxiv_id":"2411.11287","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-aduulm-360-dataset-a-multi-modal-dataset","title":"The ADUULM-360 Dataset -- A Multi-Modal Dataset for Depth Estimation in Adverse Weather","date":"2024-11-18","arxiv_id":"2411.11455","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-augmented-multimodal-llms-for-surgical","title":"Memory-Augmented Multimodal LLMs for Surgical VQA via Self-Contained Inquiry","date":"2024-11-17","arxiv_id":"2411.10937","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-llms-as-traffic-control","title":"Large Language Models (LLMs) as Traffic Control Systems at Urban Intersections: A New Paradigm","date":"2024-11-16","arxiv_id":"2411.10869","repositories_listed":0,"syntology":null},{"url":null,"slug":"content-aware-preserving-image-generation","title":"Content-Aware Preserving Image Generation","date":"2024-11-15","arxiv_id":"2411.09871","repositories_listed":0,"syntology":null},{"url":null,"slug":"se-3-equivariant-ray-embeddings-for-implicit","title":"$SE(3)$ Equivariant Ray Embeddings for Implicit Multi-View Depth Estimation","date":"2024-11-11","arxiv_id":"2411.07326","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-based-multi-modal-sensor-fusion-for","title":"Graph-Based Multi-Modal Sensor Fusion for Autonomous Driving","date":"2024-11-06","arxiv_id":"2411.03702","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-uncertainty-in-3d-gaussian-splatting","title":"Modeling Uncertainty in 3D Gaussian Splatting through Continuous Semantic Splatting","date":"2024-11-04","arxiv_id":"2411.02547","repositories_listed":0,"syntology":null},{"url":null,"slug":"symbolic-graph-inference-for-compound-scene","title":"Symbolic Graph Inference for Compound Scene Understanding","date":"2024-10-30","arxiv_id":"2410.22626","repositories_listed":0,"syntology":null},{"url":null,"slug":"unirit-towards-few-shot-non-rigid-point-cloud","title":"UniRiT: Towards Few-Shot Non-Rigid Point Cloud Registration","date":"2024-10-30","arxiv_id":"2410.22909","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-algorithms-for-surgical-phase","title":"Towards Robust Algorithms for Surgical Phase Recognition via Digital Twin-based Scene Representation","date":"2024-10-26","arxiv_id":"2410.20026","repositories_listed":0,"syntology":null},{"url":null,"slug":"perspectivenet-multi-view-perception-for","title":"PerspectiveNet: Multi-View Perception for Dynamic Scene Understanding","date":"2024-10-22","arxiv_id":"2410.16824","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-for-autonomous-driving-1","title":"Large Language Models for Autonomous Driving (LLM4AD): Concept, Benchmark, Experiments, and Challenges","date":"2024-10-20","arxiv_id":"2410.15281","repositories_listed":0,"syntology":null},{"url":null,"slug":"sam-guided-masked-token-prediction-for-3d","title":"SAM-Guided Masked Token Prediction for 3D Scene Understanding","date":"2024-10-16","arxiv_id":"2410.12158","repositories_listed":0,"syntology":null},{"url":null,"slug":"3darticcyclists-generating-simulated-dynamic","title":"3DArticCyclists: Generating Synthetic Articulated 8D Pose-Controllable Cyclist Data for Computer Vision Applications","date":"2024-10-14","arxiv_id":"2410.10782","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-single-image-to-3d-generation-using","title":"Enhancing Single Image to 3D Generation using Gaussian Splatting and Hybrid Diffusion Priors","date":"2024-10-12","arxiv_id":"2410.09467","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-vision-language-gaussian-splatting","title":"3D Vision-Language Gaussian Splatting","date":"2024-10-10","arxiv_id":"2410.07577","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-transition-towards-virtual-representations","title":"A transition towards virtual representations of visual scenes","date":"2024-10-10","arxiv_id":"2410.07987","repositories_listed":0,"syntology":null},{"url":null,"slug":"test-time-intensity-consistency-adaptation","title":"Test-Time Intensity Consistency Adaptation for Shadow Detection","date":"2024-10-10","arxiv_id":"2410.07695","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-the-impact-of-point-cloud","title":"Evaluating the Impact of Point Cloud Colorization on Semantic Segmentation Accuracy","date":"2024-10-09","arxiv_id":"2410.06725","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-rgbt-open-vocabulary-rgb-t-zero-shot","title":"Open-RGBT: Open-vocabulary RGB-T Zero-shot Semantic Segmentation in Open-world Environments","date":"2024-10-09","arxiv_id":"2410.06626","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-models-in-3d-vision-a-survey","title":"Diffusion Models in 3D Vision: A Survey","date":"2024-10-07","arxiv_id":"2410.04738","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-efficient-multiview-perception","title":"Resource-Efficient Multiview Perception: Integrating Semantic Masking with Masked Autoencoders","date":"2024-10-07","arxiv_id":"2410.04817","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-place-panoptic-radiance-field-segmentation","title":"In-Place Panoptic Radiance Field Segmentation with Perceptual Prior for 3D Scene Understanding","date":"2024-10-06","arxiv_id":"2410.04529","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-object-detection-with-a-machine-learning","title":"Fast Object Detection with a Machine Learning Edge Device","date":"2024-10-05","arxiv_id":"2410.04173","repositories_listed":0,"syntology":null},{"url":null,"slug":"spartun3d-situated-spatial-understanding-of","title":"SPARTUN3D: Situated Spatial Understanding of 3D World in Large Language Models","date":"2024-10-04","arxiv_id":"2410.03878","repositories_listed":0,"syntology":null},{"url":null,"slug":"class-agnostic-visio-temporal-scene-sketch","title":"Class-Agnostic Visio-Temporal Scene Sketch Semantic Segmentation","date":"2024-09-30","arxiv_id":"2410.00266","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-only-speak-once-to-see","title":"You Only Speak Once to See","date":"2024-09-27","arxiv_id":"2409.18372","repositories_listed":0,"syntology":null},{"url":"/paper/llava-3d-a-simple-yet-effective-pathway-to","slug":"llava-3d-a-simple-yet-effective-pathway-to","title":"LLaVA-3D: A Simple yet Effective Pathway to Empowering LMMs with 3D-awareness","date":"2024-09-26","arxiv_id":"2409.18125","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-understanding-in-pick-and-place-tasks","title":"Scene Understanding in Pick-and-Place Tasks: Analyzing Transformations Between Initial and Final Scenes","date":"2024-09-26","arxiv_id":"2409.17720","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-world-object-detection-with-instance","title":"OW-Rep: Open World Object Detection with Instance Representation Learning","date":"2024-09-24","arxiv_id":"2409.16073","repositories_listed":0,"syntology":null},{"url":"/paper/diffusion-based-rgb-d-semantic-segmentation","slug":"diffusion-based-rgb-d-semantic-segmentation","title":"Diffusion-based RGB-D Semantic Segmentation with Deformable Attention Transformer","date":"2024-09-23","arxiv_id":"2409.15117","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-semantic-segmentation-for-large","title":"Multilateral Cascading Network for Semantic Segmentation of Large-Scale Outdoor Point Clouds","date":"2024-09-21","arxiv_id":"2409.13983","repositories_listed":0,"syntology":null},{"url":null,"slug":"mose-monocular-semantic-reconstruction-using","title":"MOSE: Monocular Semantic Reconstruction Using NeRF-Lifted Noisy Priors","date":"2024-09-21","arxiv_id":"2409.14019","repositories_listed":0,"syntology":null},{"url":null,"slug":"relevance-driven-decision-making-for-safer","title":"Relevance-driven Decision Making for Safer and More Efficient Human Robot Collaboration","date":"2024-09-21","arxiv_id":"2409.13998","repositories_listed":0,"syntology":null},{"url":null,"slug":"dae-fuse-an-adaptive-discriminative","title":"DAE-Fuse: An Adaptive Discriminative Autoencoder for Multi-Modality Image Fusion","date":"2024-09-16","arxiv_id":"2409.10080","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-token-sparsification-for-efficient","title":"Video Token Sparsification for Efficient Multimodal LLMs in Autonomous Driving","date":"2024-09-16","arxiv_id":"2409.11182","repositories_listed":0,"syntology":null},{"url":null,"slug":"relevance-for-human-robot-collaboration","title":"Relevance for Human Robot Collaboration","date":"2024-09-12","arxiv_id":"2409.07753","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-localizing-structural-elements","title":"Towards Localizing Structural Elements: Merging Geometrical Detection with Semantic Verification in RGB-D Data","date":"2024-09-10","arxiv_id":"2409.06625","repositories_listed":0,"syntology":null},{"url":null,"slug":"tandepth-leveraging-global-dems-for-metric","title":"TanDepth: Leveraging Global DEMs for Metric Monocular Depth Estimation in UAVs","date":"2024-09-08","arxiv_id":"2409.05142","repositories_listed":0,"syntology":null},{"url":null,"slug":"future-does-matter-boosting-3d-object","title":"Future Does Matter: Boosting 3D Object Detection with Temporal Motion Estimation in Point Cloud Sequences","date":"2024-09-06","arxiv_id":"2409.04390","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-3d-gaussian-splatting-for-sparse","title":"Optimizing 3D Gaussian Splatting for Sparse Viewpoint Scene Reconstruction","date":"2024-09-05","arxiv_id":"2409.03213","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-lvlms-obtain-a-driver-s-license-a","title":"Can LVLMs Obtain a Driver's License? A Benchmark Towards Reliable AGI for Autonomous Driving","date":"2024-09-04","arxiv_id":"2409.02914","repositories_listed":0,"syntology":null},{"url":null,"slug":"gaussianpu-a-hybrid-2d-3d-upsampling","title":"GaussianPU: A Hybrid 2D-3D Upsampling Framework for Enhancing Color Point Clouds via 3D Gaussian Splatting","date":"2024-09-03","arxiv_id":"2409.01581","repositories_listed":0,"syntology":null},{"url":null,"slug":"leaky-wave-antenna-equipped-rf-chipless-tags","title":"Leaky Wave Antenna-Equipped RF Chipless Tags for Orientation Estimation","date":"2024-08-31","arxiv_id":"2409.00501","repositories_listed":0,"syntology":null},{"url":null,"slug":"drivegenvlm-real-world-video-generation-for","title":"DriveGenVLM: Real-world Video Generation for Vision Language Model based Autonomous Driving","date":"2024-08-29","arxiv_id":"2408.16647","repositories_listed":0,"syntology":null},{"url":null,"slug":"str-l-pose-integrating-point-and-structured","title":"Str-L Pose: Integrating Point and Structured Line for Relative Pose Estimation in Dual-Graph","date":"2024-08-28","arxiv_id":"2408.15750","repositories_listed":0,"syntology":null}],"record_sha256":"5d8f58dba8972e460de727573f79c8a9de6ce2a67c79e8c0d0cc7840b0930c37","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}