{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/make-a-scene-scene-based-text-to-image","title":"Make-A-Scene: Scene-Based Text-to-Image Generation with Human Priors","arxiv_id":"2203.13131","date":"2022-03-24","proceeding":null,"authors":["Oran Gafni","Adam Polyak","Oron Ashual","Shelly Sheynin","Devi Parikh","Yaniv Taigman"],"abstract":"Recent text-to-image generation methods provide a simple yet exciting conversion capability between text and image domains. While these methods have incrementally improved the generated image fidelity and text relevancy, several pivotal gaps remain unanswered, limiting applicability and quality. We propose a novel text-to-image method that addresses these gaps by (i) enabling a simple control mechanism complementary to text in the form of a scene, (ii) introducing elements that substantially improve the tokenization process by employing domain-specific knowledge over key image regions (faces and salient objects), and (iii) adapting classifier-free guidance for the transformer use case. Our model achieves state-of-the-art FID and human evaluation results, unlocking the ability to generate high fidelity images in a resolution of 512x512 pixels, significantly improving visual quality. Through scene controllability, we introduce several new capabilities: (i) Scene editing, (ii) text editing with anchor scenes, (iii) overcoming out-of-distribution text prompts, and (iv) story illustration generation, as demonstrated in the story we wrote.","url_abs":"https://arxiv.org/abs/2203.13131v1","url_pdf":"https://arxiv.org/pdf/2203.13131v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"make-a-scene-scene-based-text-to-image","repo_url":"https://github.com/CasualGANPapers/Make-A-Scene","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"image-generation","task_name":"Image Generation"},{"task_slug":"semantic-segmentation","task_name":"Semantic Segmentation"},{"task_slug":"text-to-image-generation-1","task_name":"Text to Image Generation"},{"task_slug":"text-to-image-generation","task_name":"Text-to-Image Generation"}],"methods":[{"method_slug":"make-a-scene","method_name":"Make-A-Scene"}],"datasets_introduced":[],"methods_introduced":[{"slug":"make-a-scene","name":"Make-A-Scene","full_name":"Make-A-Scene"}],"results":[{"leaderboard":"/sota/text-to-image-generation-on-coco","task":"Text-to-Image Generation","dataset":"COCO (Common Objects in Context)","model":"Make-a-Scene (unfiltered)","rank_in_archive_order":18,"of":69,"metrics":{"FID":"7.55"},"uses_additional_data":true},{"leaderboard":"/sota/text-to-image-generation-on-coco","task":"Text-to-Image Generation","dataset":"COCO (Common Objects in Context)","model":"Make-a-Scene (unfiltered)","rank_in_archive_order":30,"of":69,"metrics":{"FID":"11.84"},"uses_additional_data":true}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2203.13131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13131"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/CasualGANPapers/Make-A-Scene","reach":null}],"summary":{"ran":3,"ran_fixture":1,"unverified":2},"by_repo_kind":{"listed":{"samples":6,"ran":4,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"98ea870a15f7b0e8","entry":"MLP","repo":"CasualGANPapers/Make-A-Scene","repo_kind":"listed","path":"models/transformer.py","file_url":"https://github.com/CasualGANPapers/Make-A-Scene/blob/HEAD/models/transformer.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"98ea870a15f7b0e8"}},{"code_sha256_prefix":"910b60593238b32f","entry":"SelfAttention","repo":"CasualGANPapers/Make-A-Scene","repo_kind":"listed","path":"models/transformer.py","file_url":"https://github.com/CasualGANPapers/Make-A-Scene/blob/HEAD/models/transformer.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"910b60593238b32f"}},{"code_sha256_prefix":"a73f4b74b84d9692","entry":"TransformerLayer","repo":"CasualGANPapers/Make-A-Scene","repo_kind":"listed","path":"models/transformer.py","file_url":"https://github.com/CasualGANPapers/Make-A-Scene/blob/HEAD/models/transformer.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"a73f4b74b84d9692"}},{"code_sha256_prefix":"d60ff4b7f07480e1","entry":"gelu","repo":"CasualGANPapers/Make-A-Scene","repo_kind":"listed","path":"models/transformer.py","file_url":"https://github.com/CasualGANPapers/Make-A-Scene/blob/HEAD/models/transformer.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"d60ff4b7f07480e1"}},{"code_sha256_prefix":"97975a90b96d4150","entry":"MakeAScene","repo":"CasualGANPapers/Make-A-Scene","repo_kind":"listed","path":"models/transformer.py","file_url":"https://github.com/CasualGANPapers/Make-A-Scene/blob/HEAD/models/transformer.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"97975a90b96d4150"}},{"code_sha256_prefix":"6a293ec4d1355238","entry":"Transformer","repo":"CasualGANPapers/Make-A-Scene","repo_kind":"listed","path":"models/transformer.py","file_url":"https://github.com/CasualGANPapers/Make-A-Scene/blob/HEAD/models/transformer.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"6a293ec4d1355238"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}