{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/scene-understanding/papers/12","list_of":"/task/scene-understanding","task":"Scene Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":12,"pages_in_order":18,"rows_per_page":100,"rows":[1101,1200],"of":1723,"counts":{"archive_papers_tagged":1723,"with_a_code_link":720,"where_syntology_ran_a_sample":208,"not_listed_spam_title":0,"listed":1723,"listed_where_code_ran":208,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":26,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":26,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/scene-understanding","prev":"/task/scene-understanding/papers/11","next":"/task/scene-understanding/papers/13","papers":[{"url":null,"slug":"enhancing-human-centered-dynamic-scene","title":"Enhancing Human-Centered Dynamic Scene Understanding via Multiple LLMs Collaborated Reasoning","date":"2024-03-15","arxiv_id":"2403.10107","repositories_listed":0,"syntology":null},{"url":null,"slug":"mapping-high-level-semantic-regions-in-indoor","title":"Mapping High-level Semantic Regions in Indoor Environments without Object Recognition","date":"2024-03-11","arxiv_id":"2403.07076","repositories_listed":0,"syntology":null},{"url":null,"slug":"out-of-the-room-generalizing-event-based","title":"Out of the Room: Generalizing Event-Based Dynamic Motion Segmentation for Complex Scenes","date":"2024-03-07","arxiv_id":"2403.04562","repositories_listed":0,"syntology":null},{"url":null,"slug":"gsnerf-generalizable-semantic-neural-radiance","title":"GSNeRF: Generalizable Semantic Neural Radiance Fields with Enhanced 3D Scene Understanding","date":"2024-03-06","arxiv_id":"2403.03608","repositories_listed":0,"syntology":null},{"url":null,"slug":"hunter-unsupervised-human-centric-3d","title":"HUNTER: Unsupervised Human-centric 3D Detection via Transferring Knowledge from Synthetic Instances to Real Scenes","date":"2024-03-05","arxiv_id":"2403.02769","repositories_listed":0,"syntology":null},{"url":null,"slug":"pcdepth-pattern-based-complementary-learning","title":"PCDepth: Pattern-based Complementary Learning for Monocular Depth Estimation by Best of Both Worlds","date":"2024-02-29","arxiv_id":"2402.18925","repositories_listed":0,"syntology":null},{"url":null,"slug":"livehps-lidar-based-scene-level-human-pose","title":"LiveHPS: LiDAR-based Scene-level Human Pose and Shape Estimation in Free Environment","date":"2024-02-27","arxiv_id":"2402.17171","repositories_listed":0,"syntology":null},{"url":null,"slug":"opensun3d-1st-workshop-challenge-on-open","title":"OpenSUN3D: 1st Workshop Challenge on Open-Vocabulary 3D Scene Understanding","date":"2024-02-23","arxiv_id":"2402.15321","repositories_listed":0,"syntology":null},{"url":null,"slug":"drivevlm-the-convergence-of-autonomous","title":"DriveVLM: The Convergence of Autonomous Driving and Large Vision-Language Models","date":"2024-02-19","arxiv_id":"2402.12289","repositories_listed":0,"syntology":null},{"url":null,"slug":"moving-object-proposals-with-deep-learned","title":"Moving Object Proposals with Deep Learned Optical Flow for Video Object Segmentation","date":"2024-02-14","arxiv_id":"2402.08882","repositories_listed":0,"syntology":null},{"url":null,"slug":"incoro-in-context-learning-for-robotics","title":"InCoRo: In-Context Learning for Robotics Control with Feedback Loops","date":"2024-02-07","arxiv_id":"2402.05188","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-world-modeling-via-semantic-vector","title":"Neural Language of Thought Models","date":"2024-02-02","arxiv_id":"2402.01203","repositories_listed":0,"syntology":null},{"url":null,"slug":"autort-embodied-foundation-models-for-large","title":"AutoRT: Embodied Foundation Models for Large Scale Orchestration of Robotic Agents","date":"2024-01-23","arxiv_id":"2401.12963","repositories_listed":0,"syntology":null},{"url":null,"slug":"digital-divides-in-scene-recognition","title":"Digital Divides in Scene Recognition: Uncovering Socioeconomic Biases in Deep Learning Systems","date":"2024-01-23","arxiv_id":"2401.13097","repositories_listed":0,"syntology":null},{"url":null,"slug":"s-3-m-net-joint-learning-of-semantic","title":"S$^3$M-Net: Joint Learning of Semantic Segmentation and Stereo Matching for Autonomous Driving","date":"2024-01-21","arxiv_id":"2401.11414","repositories_listed":0,"syntology":null},{"url":null,"slug":"bpdo-boundary-points-dynamic-optimization-for","title":"BPDO:Boundary Points Dynamic Optimization for Arbitrary Shape Scene Text Detection","date":"2024-01-18","arxiv_id":"2401.09997","repositories_listed":0,"syntology":null},{"url":null,"slug":"sceneverse-scaling-3d-vision-language","title":"SceneVerse: Scaling 3D Vision-Language Learning for Grounded Scene Understanding","date":"2024-01-17","arxiv_id":"2401.09340","repositories_listed":0,"syntology":null},{"url":null,"slug":"class-imbalanced-semi-supervised-learning-for","title":"Class-Imbalanced Semi-Supervised Learning for Large-Scale Point Cloud Semantic Segmentation via Decoupling Optimization","date":"2024-01-13","arxiv_id":"2401.06975","repositories_listed":0,"syntology":null},{"url":null,"slug":"cosseggaussians-compact-and-swift-scene","title":"Learning Segmented 3D Gaussians via Efficient Feature Unprojection for Zero-shot Neural Scene Segmentation","date":"2024-01-11","arxiv_id":"2401.05925","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-self-and-cross-triplet-correlations","title":"Exploring Self- and Cross-Triplet Correlations for Human-Object Interaction Detection","date":"2024-01-11","arxiv_id":"2401.05676","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlp-vision-language-planning-for-autonomous","title":"VLP: Vision Language Planning for Autonomous Driving","date":"2024-01-10","arxiv_id":"2401.05577","repositories_listed":0,"syntology":null},{"url":null,"slug":"fmgs-foundation-model-embedded-3d-gaussian","title":"FMGS: Foundation Model Embedded 3D Gaussian Splatting for Holistic 3D Scene Understanding","date":"2024-01-03","arxiv_id":"2401.01970","repositories_listed":0,"syntology":null},{"url":null,"slug":"bilateral-adaptation-for-human-object","title":"Bilateral Adaptation for Human-Object Interaction Detection with Occlusion-Robustness","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"going-beyond-multi-task-dense-prediction-with","title":"Going Beyond Multi-Task Dense Prediction with Synergy Embedding Models","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"omni-q-omni-directional-scene-understanding","title":"Omni-Q: Omni-Directional Scene Understanding for Unsupervised Visual Grounding","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"panorecon-real-time-panoptic-3d","title":"PanoRecon: Real-Time Panoptic 3D Reconstruction from Monocular Video","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"scenefun3d-fine-grained-functionality-and","title":"SceneFun3D: Fine-Grained Functionality and Affordance Understanding in 3D Scenes","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-clip-driven-language-free-3d-visual","title":"Towards CLIP-driven Language-free 3D Visual Grounding via 2D-3D Relational Enhancement and Consistency","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-3d-structure-inference-from","title":"Unsupervised 3D Structure Inference from Category-Specific Image Collections","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"when-visual-grounding-meets-gigapixel-level","title":"When Visual Grounding Meets Gigapixel-level Large-scale Scenes: Benchmark and Approach","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-multi-modal-image-stitching-for","title":"Robust Multi-Modal Image Stitching for Improved Scene Understanding","date":"2023-12-28","arxiv_id":"2312.17010","repositories_listed":0,"syntology":null},{"url":null,"slug":"cloud-device-collaborative-learning-for","title":"Cloud-Device Collaborative Learning for Multimodal Large Language Models","date":"2023-12-26","arxiv_id":"2312.16279","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-of-feature-interaction-for-multi","title":"BridgeNet: Comprehensive and Effective Feature Interactions via Bridge Feature for Multi-task Dense Predictions","date":"2023-12-21","arxiv_id":"2312.13514","repositories_listed":0,"syntology":null},{"url":null,"slug":"accidentgpt-accident-analysis-and-prevention","title":"AccidentGPT: Accident Analysis and Prevention from V2X Environmental Perception with Multi-modal Large Model","date":"2023-12-20","arxiv_id":"2312.13156","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-assisted-3d-scene-understanding","title":"Language-Assisted 3D Scene Understanding","date":"2023-12-18","arxiv_id":"2312.11451","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-3d-visual-grounding-based","title":"Weakly-Supervised 3D Visual Grounding based on Visual Linguistic Alignment","date":"2023-12-15","arxiv_id":"2312.09625","repositories_listed":0,"syntology":null},{"url":null,"slug":"dietary-assessment-with-multimodal-chatgpt-a","title":"Dietary Assessment with Multimodal ChatGPT: A Systematic Analysis","date":"2023-12-14","arxiv_id":"2312.08592","repositories_listed":0,"syntology":null},{"url":null,"slug":"vmt-adapter-parameter-efficient-transfer","title":"VMT-Adapter: Parameter-Efficient Transfer Learning for Multi-Task Dense Scene Understanding","date":"2023-12-14","arxiv_id":"2312.08733","repositories_listed":0,"syntology":null},{"url":null,"slug":"cataract-1k-cataract-surgery-dataset-for","title":"Cataract-1K: Cataract Surgery Dataset for Scene Segmentation, Phase Recognition, and Irregularity Detection","date":"2023-12-11","arxiv_id":"2312.06295","repositories_listed":0,"syntology":null},{"url":null,"slug":"skyscenes-a-synthetic-dataset-for-aerial","title":"SkyScenes: A Synthetic Dataset for Aerial Scene Understanding","date":"2023-12-11","arxiv_id":"2312.06719","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatiotemporal-event-graphs-for-dynamic-scene","title":"Spatiotemporal Event Graphs for Dynamic Scene Understanding","date":"2023-12-11","arxiv_id":"2312.07621","repositories_listed":0,"syntology":null},{"url":null,"slug":"prospective-role-of-foundation-models-in","title":"Prospective Role of Foundation Models in Advancing Autonomous Vehicles","date":"2023-12-08","arxiv_id":"2405.02288","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-and-a-robust-framework-of-data","title":"A Review and A Robust Framework of Data-Efficient 3D Scene Parsing with Traditional/Learned 3D Descriptors","date":"2023-12-03","arxiv_id":"2312.01262","repositories_listed":0,"syntology":null},{"url":null,"slug":"segment-any-3d-gaussians","title":"Segment Any 3D Gaussians","date":"2023-12-01","arxiv_id":"2312.00860","repositories_listed":0,"syntology":null},{"url":null,"slug":"hatt-flow-hierarchical-attention-flow","title":"HAtt-Flow: Hierarchical Attention-Flow Mechanism for Group Activity Scene Graph Generation in Videos","date":"2023-11-28","arxiv_id":"2312.07740","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-summarization-clustering-scene-videos","title":"Scene Summarization: Clustering Scene Videos into Spatially Diverse Frames","date":"2023-11-28","arxiv_id":"2311.17940","repositories_listed":0,"syntology":null},{"url":null,"slug":"falcon-fairness-learning-via-contrastive","title":"FALCON: Fairness Learning via Contrastive Attention Approach to Continual Semantic Scene Understanding","date":"2023-11-27","arxiv_id":"2311.15965","repositories_listed":0,"syntology":null},{"url":null,"slug":"react-recognize-every-action-everywhere-all","title":"REACT: Recognize Every Action Everywhere All At Once","date":"2023-11-27","arxiv_id":"2312.00188","repositories_listed":0,"syntology":null},{"url":null,"slug":"gpt-4v-takes-the-wheel-evaluating-promise-and","title":"GPT-4V Takes the Wheel: Promises and Challenges for Pedestrian Behavior Prediction","date":"2023-11-24","arxiv_id":"2311.14786","repositories_listed":0,"syntology":null},{"url":null,"slug":"gp-nerf-generalized-perception-nerf-for","title":"GP-NeRF: Generalized Perception NeRF for Context-Aware 3D Scene Understanding","date":"2023-11-20","arxiv_id":"2311.11863","repositories_listed":0,"syntology":null},{"url":null,"slug":"seadsc-a-video-based-unsupervised-method-for","title":"SeaDSC: A video-based unsupervised method for dynamic scene change detection in unmanned surface vehicles","date":"2023-11-20","arxiv_id":"2311.11580","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-stream-scene-understanding-on-graph","title":"Two Stream Scene Understanding on Graph Embedding","date":"2023-11-12","arxiv_id":"2311.06746","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-large-scale-pretrained-vision","title":"Leveraging Large-Scale Pretrained Vision Foundation Models for Label-Efficient 3D Point Cloud Segmentation","date":"2023-11-03","arxiv_id":"2311.01989","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-view-3d-scene-reconstruction-with-high","title":"Single-view 3D Scene Reconstruction with High-fidelity Shape and Texture","date":"2023-11-01","arxiv_id":"2311.00457","repositories_listed":0,"syntology":null},{"url":null,"slug":"recent-advances-in-multi-modal-3d-scene","title":"Recent Advances in Multi-modal 3D Scene Understanding: A Comprehensive Survey and Evaluation","date":"2023-10-24","arxiv_id":"2310.15676","repositories_listed":0,"syntology":null},{"url":null,"slug":"panoptic-out-of-distribution-segmentation","title":"Panoptic Out-of-Distribution Segmentation","date":"2023-10-18","arxiv_id":"2310.11797","repositories_listed":0,"syntology":null},{"url":null,"slug":"s4c-self-supervised-semantic-scene-completion","title":"S4C: Self-Supervised Semantic Scene Completion with Neural Fields","date":"2023-10-11","arxiv_id":"2310.07522","repositories_listed":0,"syntology":null},{"url":null,"slug":"textpsg-panoptic-scene-graph-generation-from-1","title":"TextPSG: Panoptic Scene Graph Generation from Textual Descriptions","date":"2023-10-10","arxiv_id":"2310.07056","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-open-vocabulary-tracking-with-large","title":"Zero-Shot Open-Vocabulary Tracking with Large Pre-Trained Models","date":"2023-10-10","arxiv_id":"2310.06992","repositories_listed":0,"syntology":null},{"url":null,"slug":"elastic-interaction-energy-loss-for-traffic","title":"Elastic Interaction Energy-Informed Real-Time Traffic Scene Perception","date":"2023-10-02","arxiv_id":"2310.01449","repositories_listed":0,"syntology":null},{"url":null,"slug":"logical-bias-learning-for-object-relation","title":"Logical Bias Learning for Object Relation Prediction","date":"2023-10-01","arxiv_id":"2310.00712","repositories_listed":0,"syntology":null},{"url":null,"slug":"sgrec3d-self-supervised-3d-scene-graph","title":"SGRec3D: Self-Supervised 3D Scene Graph Learning via Object-Level Scene Reconstruction","date":"2023-09-27","arxiv_id":"2309.15702","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-extended-indoor-slam-lexis-a","title":"Language-EXtended Indoor SLAM (LEXIS): A Versatile System for Real-time Visual Scene Understanding","date":"2023-09-26","arxiv_id":"2309.15065","repositories_listed":0,"syntology":null},{"url":null,"slug":"llmr-real-time-prompting-of-interactive","title":"LLMR: Real-time Prompting of Interactive Worlds using Large Language Models","date":"2023-09-21","arxiv_id":"2309.12276","repositories_listed":0,"syntology":null},{"url":null,"slug":"sanpo-a-scene-understanding-accessibility","title":"SANPO: A Scene Understanding, Accessibility and Human Navigation Dataset","date":"2023-09-21","arxiv_id":"2309.12172","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-of-action-recognition-spotting-and","title":"Survey of Action Recognition, Spotting and Spatio-Temporal Localization in Soccer -- Current Trends and Research Perspectives","date":"2023-09-21","arxiv_id":"2309.12067","repositories_listed":0,"syntology":null},{"url":null,"slug":"panomixswap-panorama-mixing-via-structural","title":"PanoMixSwap Panorama Mixing via Structural Swapping for Indoor Scene Understanding","date":"2023-09-18","arxiv_id":"2309.09514","repositories_listed":0,"syntology":null},{"url":null,"slug":"so-you-think-you-can-track","title":"So you think you can track?","date":"2023-09-13","arxiv_id":"2309.07268","repositories_listed":0,"syntology":null},{"url":null,"slug":"amodalsynthdrive-a-synthetic-amodal","title":"AmodalSynthDrive: A Synthetic Amodal Perception Dataset for Autonomous Driving","date":"2023-09-12","arxiv_id":"2309.06547","repositories_listed":0,"syntology":null},{"url":null,"slug":"rank2tell-a-multimodal-driving-dataset-for","title":"Rank2Tell: A Multimodal Driving Dataset for Joint Importance Ranking and Reasoning","date":"2023-09-12","arxiv_id":"2309.06597","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-you-text-what-is-happening-integrating","title":"Can you text what is happening? Integrating pre-trained language encoders into trajectory prediction models for autonomous driving","date":"2023-09-11","arxiv_id":"2309.05282","repositories_listed":0,"syntology":null},{"url":null,"slug":"pag-nerf-towards-fast-and-efficient-end-to","title":"PAg-NeRF: Towards fast and efficient end-to-end panoptic 3D representations for agricultural robotics","date":"2023-09-11","arxiv_id":"2309.05339","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-point-clouds-transformer","title":"Weakly Supervised Point Clouds Transformer for 3D Object Detection","date":"2023-09-08","arxiv_id":"2309.04105","repositories_listed":0,"syntology":null},{"url":null,"slug":"structural-concept-learning-via-graph","title":"Structural Concept Learning via Graph Attention for Multi-Level Rearrangement Planning","date":"2023-09-05","arxiv_id":"2309.02547","repositories_listed":0,"syntology":null},{"url":null,"slug":"expanding-frozen-vision-language-models","title":"Expanding Frozen Vision-Language Models without Retraining: Towards Improved Robot Perception","date":"2023-08-31","arxiv_id":"2308.16493","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-semantic-depth-estimation-1","title":"Semi-Supervised Semantic Depth Estimation using Symbiotic Transformer and NearFarMix Augmentation","date":"2023-08-28","arxiv_id":"2308.14400","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-autonomous-driving-using-deep","title":"End-to-end Autonomous Driving using Deep Learning: A Systematic Review","date":"2023-08-27","arxiv_id":"2311.18636","repositories_listed":0,"syntology":null},{"url":null,"slug":"synergizing-contrastive-learning-and-optimal","title":"Synergizing Contrastive Learning and Optimal Transport for 3D Point Cloud Domain Adaptation","date":"2023-08-27","arxiv_id":"2308.14126","repositories_listed":0,"syntology":null},{"url":null,"slug":"surgnn-explainable-visual-scene-understanding","title":"SurGNN: Explainable visual scene understanding and assessment of surgical skill using graph neural networks","date":"2023-08-24","arxiv_id":"2308.13073","repositories_listed":0,"syntology":null},{"url":null,"slug":"novel-view-synthesis-and-pose-estimation-for","title":"Novel-view Synthesis and Pose Estimation for Hand-Object Interaction from Sparse Views","date":"2023-08-22","arxiv_id":"2308.11198","repositories_listed":0,"syntology":null},{"url":null,"slug":"explore-and-tell-embodied-visual-captioning","title":"Explore and Tell: Embodied Visual Captioning in 3D Environments","date":"2023-08-21","arxiv_id":"2308.10447","repositories_listed":0,"syntology":null},{"url":"/paper/caspnet-joint-multi-agent-motion-prediction","slug":"caspnet-joint-multi-agent-motion-prediction","title":"CASPNet++: Joint Multi-Agent Motion Prediction","date":"2023-08-15","arxiv_id":"2308.07751","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-dino-a-self-supervised-video","title":"Temporal DINO: A Self-supervised Video Strategy to Enhance Action Prediction","date":"2023-08-08","arxiv_id":"2308.04589","repositories_listed":0,"syntology":null},{"url":null,"slug":"syn-mediverse-a-multimodal-synthetic-dataset","title":"Syn-Mediverse: A Multimodal Synthetic Dataset for Intelligent Scene Understanding of Healthcare Facilities","date":"2023-08-06","arxiv_id":"2308.03193","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-aware-human-pose-generation-using","title":"Scene-aware Human Pose Generation using Transformer","date":"2023-08-04","arxiv_id":"2308.02177","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-3d-instance-segmentation","title":"Weakly Supervised 3D Instance Segmentation without Instance-level Annotations","date":"2023-08-03","arxiv_id":"2308.01721","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-end-to-end-driving-model-for","title":"Interpretable End-to-End Driving Model for Implicit Scene Understanding","date":"2023-08-02","arxiv_id":"2308.01180","repositories_listed":0,"syntology":null},{"url":"/paper/lowis3d-language-driven-open-world-instance","slug":"lowis3d-language-driven-open-world-instance","title":"Lowis3D: Language-Driven Open-World Instance-Level 3D Scene Understanding","date":"2023-08-01","arxiv_id":"2308.00353","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-image-captioning-with-depth","title":"Enhancing image captioning with depth information using a Transformer-based framework","date":"2023-07-24","arxiv_id":"2308.03767","repositories_listed":0,"syntology":null},{"url":null,"slug":"challenges-for-monocular-6d-object-pose","title":"Challenges for Monocular 6D Object Pose Estimation in Robotics","date":"2023-07-22","arxiv_id":"2307.12172","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-online-lane-graph-extraction-by","title":"Improving Online Lane Graph Extraction by Object-Lane Clustering","date":"2023-07-20","arxiv_id":"2307.10947","repositories_listed":0,"syntology":null},{"url":null,"slug":"mining-conditional-part-semantics-with","title":"Mining Conditional Part Semantics with Occluded Extrapolation for Human-Object Interaction Detection","date":"2023-07-19","arxiv_id":"2307.10499","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-action-recognition-in-still-images","title":"Human Action Recognition in Still Images Using ConViT","date":"2023-07-18","arxiv_id":"2307.08994","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-unified-agent-with-foundation","title":"Towards A Unified Agent with Foundation Models","date":"2023-07-18","arxiv_id":"2307.09668","repositories_listed":0,"syntology":null},{"url":null,"slug":"smart-infrastructure-a-research-junction","title":"Smart Infrastructure: A Research Junction","date":"2023-07-12","arxiv_id":"2307.06177","repositories_listed":0,"syntology":null},{"url":null,"slug":"test-time-adaptation-for-nighttime-color","title":"Test-Time Adaptation for Nighttime Color-Thermal Semantic Segmentation","date":"2023-07-10","arxiv_id":"2307.04470","repositories_listed":0,"syntology":null},{"url":null,"slug":"psdr-room-single-photo-to-scene-using","title":"PSDR-Room: Single Photo to Scene using Differentiable Rendering","date":"2023-07-06","arxiv_id":"2307.03244","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-recognition-system-on-a-tactile-device","title":"Object Recognition System on a Tactile Device for Visually Impaired","date":"2023-07-05","arxiv_id":"2307.02211","repositories_listed":0,"syntology":null},{"url":null,"slug":"artifacts-mapping-multi-modal-semantic","title":"Artifacts Mapping: Multi-Modal Semantic Mapping for Object Detection and 3D Localization","date":"2023-07-03","arxiv_id":"2307.01121","repositories_listed":0,"syntology":null},{"url":"/paper/physion-evaluating-physical-scene","slug":"physion-evaluating-physical-scene","title":"Physion++: Evaluating Physical Scene Understanding that Requires Online Inference of Different Physical Properties","date":"2023-06-27","arxiv_id":"2306.15668","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/physion-evaluating-physical-scene#ran","syntology_url":"https://syntology.ai/paper/2306.15668","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.15668"}},"official":null}}],"record_sha256":"2a6312d559120e01f7763b56cd1a3cf3fcb9cebed41b549bff66932e9e7fc283","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}