{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/neural-multisensory-scene-inference","title":"Neural Multisensory Scene Inference","arxiv_id":"1910.02344","date":"2019-10-06","proceeding":"NeurIPS 2019 12","authors":["Jae Hyun Lim","Pedro O. Pinheiro","Negar Rostamzadeh","Christopher Pal","Sungjin Ahn"],"abstract":"For embodied agents to infer representations of the underlying 3D physical world they inhabit, they should efficiently combine multisensory cues from numerous trials, e.g., by looking at and touching objects. Despite its importance, multisensory 3D scene representation learning has received less attention compared to the unimodal setting. In this paper, we propose the Generative Multisensory Network (GMN) for learning latent representations of 3D scenes which are partially observable through multiple sensory modalities. We also introduce a novel method, called the Amortized Product-of-Experts, to improve the computational efficiency and the robustness to unseen combinations of modalities at test time. Experimental results demonstrate that the proposed model can efficiently infer robust modality-invariant 3D-scene representations from arbitrary combinations of modalities and perform accurate cross-modal generation. To perform this exploration, we also develop the Multisensory Embodied 3D-Scene Environment (MESE).","url_abs":"https://arxiv.org/abs/1910.02344v2","url_pdf":"https://arxiv.org/pdf/1910.02344v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"neural-multisensory-scene-inference","repo_url":"https://github.com/lim0606/pytorch-generative-multisensory-network","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"neural-multisensory-scene-inference","repo_url":"https://github.com/lim0606/multisensory-embodied-3D-scene-environment","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":null}],"tasks":[{"task_slug":"computational-efficiency","task_name":"Computational Efficiency"},{"task_slug":"representation-learning","task_name":"Representation Learning"}],"methods":[{"method_slug":"test","method_name":"Test"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1910.02344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.02344"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/lim0606/multisensory-embodied-3D-scene-environment","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/lim0606/pytorch-generative-multisensory-network","reach":{"status":"ok","spdx":"MIT"}}],"summary":{"ran_fixture":1,"unverified":1},"by_repo_kind":{"official":{"samples":1,"ran":0,"repositories":1},"listed":{"samples":1,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"bb6046f340ccaba0","entry":"quat_from_angle_and_axis","repo":"lim0606/multisensory-embodied-3D-scene-environment","repo_kind":"listed","path":"envs/haptix/basic/manipulate.py","file_url":"https://github.com/lim0606/multisensory-embodied-3D-scene-environment/blob/HEAD/envs/haptix/basic/manipulate.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"bb6046f340ccaba0"}},{"code_sha256_prefix":"b74bf005a15bf9c7","entry":"combine_reps","repo":"lim0606/pytorch-generative-multisensory-network","repo_kind":"official","path":"models/amortized_poe_multimodal_cgqn.py","file_url":"https://github.com/lim0606/pytorch-generative-multisensory-network/blob/HEAD/models/amortized_poe_multimodal_cgqn.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"b74bf005a15bf9c7"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}