{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/pix2gestalt-amodal-segmentation-by","title":"pix2gestalt: Amodal Segmentation by Synthesizing Wholes","arxiv_id":"2401.14398","date":"2024-01-25","proceeding":"CVPR 2024 1","authors":["Ege Ozguroglu","Ruoshi Liu","Dídac Surís","Dian Chen","Achal Dave","Pavel Tokmakov","Carl Vondrick"],"abstract":"We introduce pix2gestalt, a framework for zero-shot amodal segmentation, which learns to estimate the shape and appearance of whole objects that are only partially visible behind occlusions. By capitalizing on large-scale diffusion models and transferring their representations to this task, we learn a conditional diffusion model for reconstructing whole objects in challenging zero-shot cases, including examples that break natural and physical priors, such as art. As training data, we use a synthetically curated dataset containing occluded objects paired with their whole counterparts. Experiments show that our approach outperforms supervised baselines on established benchmarks. Our model can furthermore be used to significantly improve the performance of existing object recognition and 3D reconstruction methods in the presence of occlusions.","url_abs":"https://arxiv.org/abs/2401.14398v1","url_pdf":"https://arxiv.org/pdf/2401.14398v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"pix2gestalt-amodal-segmentation-by","repo_url":"https://github.com/cvlab-columbia/pix2gestalt","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"NOASSERTION"}}],"tasks":[{"task_slug":"3d-reconstruction","task_name":"3D Reconstruction"},{"task_slug":"object-recognition","task_name":"Object Recognition"},{"task_slug":"segmentation","task_name":"Segmentation"}],"methods":[{"method_slug":"diffusion","method_name":"Diffusion"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":null,"atlas_url":"https://app.syntology.ai/?focus=2401.14398","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.14398"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/cvlab-columbia/pix2gestalt","reach":{"status":"ok","spdx":"NOASSERTION"}}],"summary":{"ran":2,"ran_draft_wrong":3,"ran_violates":3,"unverified":2},"by_repo_kind":{"official":{"samples":10,"ran":8,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":10,"samples":[{"code_sha256_prefix":"c95e88aa409419d4","entry":"add_margin","repo":"cvlab-columbia/pix2gestalt","repo_kind":"official","path":"pix2gestalt/ldm/util.py","file_url":"https://github.com/cvlab-columbia/pix2gestalt/blob/HEAD/pix2gestalt/ldm/util.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"c95e88aa409419d4"}},{"code_sha256_prefix":"fe5dd5258046898c","entry":"always","repo":"cvlab-columbia/pix2gestalt","repo_kind":"official","path":"pix2gestalt/ldm/modules/x_transformer.py","file_url":"https://github.com/cvlab-columbia/pix2gestalt/blob/HEAD/pix2gestalt/ldm/modules/x_transformer.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"fe5dd5258046898c"}},{"code_sha256_prefix":"424012cb37b31172","entry":"default","repo":"cvlab-columbia/pix2gestalt","repo_kind":"official","path":"pix2gestalt/ldm/modules/attention.py","file_url":"https://github.com/cvlab-columbia/pix2gestalt/blob/HEAD/pix2gestalt/ldm/modules/attention.py","link_basis":"harvester_set","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"424012cb37b31172"}},{"code_sha256_prefix":"4cb732f513d69dfd","entry":"disabled_train","repo":"cvlab-columbia/pix2gestalt","repo_kind":"official","path":"pix2gestalt/ldm/models/diffusion/classifier.py","file_url":"https://github.com/cvlab-columbia/pix2gestalt/blob/HEAD/pix2gestalt/ldm/models/diffusion/classifier.py","link_basis":"harvester_set","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"code_sha256_prefix":"aa5486a3650902d8","entry":"exists","repo":"cvlab-columbia/pix2gestalt","repo_kind":"official","path":"pix2gestalt/ldm/modules/attention.py","file_url":"https://github.com/cvlab-columbia/pix2gestalt/blob/HEAD/pix2gestalt/ldm/modules/attention.py","link_basis":"harvester_set","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"aa5486a3650902d8"}},{"code_sha256_prefix":"bf1758f345c1c4d5","entry":"sample_model","repo":"cvlab-columbia/pix2gestalt","repo_kind":"official","path":"pix2gestalt/inference.py","file_url":"https://github.com/cvlab-columbia/pix2gestalt/blob/HEAD/pix2gestalt/inference.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"bf1758f345c1c4d5"}},{"code_sha256_prefix":"d48d8354986e3b0e","entry":"uniform_on_device","repo":"cvlab-columbia/pix2gestalt","repo_kind":"official","path":"pix2gestalt/ldm/models/diffusion/ddpm.py","file_url":"https://github.com/cvlab-columbia/pix2gestalt/blob/HEAD/pix2gestalt/ldm/models/diffusion/ddpm.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"d48d8354986e3b0e"}},{"code_sha256_prefix":"9a299fe5ae09e407","entry":"uniq","repo":"cvlab-columbia/pix2gestalt","repo_kind":"official","path":"pix2gestalt/ldm/modules/attention.py","file_url":"https://github.com/cvlab-columbia/pix2gestalt/blob/HEAD/pix2gestalt/ldm/modules/attention.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"9a299fe5ae09e407"}},{"code_sha256_prefix":"a221ef33702f106c","entry":"log_txt_as_img","repo":"cvlab-columbia/pix2gestalt","repo_kind":"official","path":"pix2gestalt/ldm/util.py","file_url":"https://github.com/cvlab-columbia/pix2gestalt/blob/HEAD/pix2gestalt/ldm/util.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"a221ef33702f106c"}},{"code_sha256_prefix":"afe61ba472afc243","entry":"pil_rectangle_crop","repo":"cvlab-columbia/pix2gestalt","repo_kind":"official","path":"pix2gestalt/ldm/util.py","file_url":"https://github.com/cvlab-columbia/pix2gestalt/blob/HEAD/pix2gestalt/ldm/util.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"afe61ba472afc243"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}