{"url":"/task/text-to-image-generation-1","name":"Text to Image Generation","slug":"text-to-image-generation-1","description_markdown":null,"categories":[{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":969,"papers_with_code":461,"benchmarks":0,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":null,"slug":"text-to-image-generation-on-diffusiondb","dataset":"DiffusionDB","dataset_url":"/dataset/diffusiondb","rows_in_archive":0,"metrics":["Fréchet Inception Distance","Inception Score","Similarity Score (CLIP)"],"first_row_in_archive_order":null}],"datasets":[{"url":"/dataset/diffusiondb","name":"DiffusionDB","full_name":"","num_papers_in_archive":70}],"subtasks":[{"url":"/task/text-to-3d","name":"Text to 3D"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":461,"tagged_in_all":969,"items":[{"url":"/paper/attngan-fine-grained-text-to-image-generation","title":"AttnGAN: Fine-Grained Text to Image Generation with Attentional Generative Adversarial Networks","date":"2017-11-28","arxiv_id":"1711.10485","repositories_listed":20,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/adding-conditional-control-to-text-to-image","title":"Adding Conditional Control to Text-to-Image Diffusion Models","date":"2023-02-10","arxiv_id":"2302.05543","repositories_listed":12,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/zero-shot-text-to-image-generation","title":"Zero-Shot Text-to-Image Generation","date":"2021-02-24","arxiv_id":"2102.12092","repositories_listed":12,"syntology":{"n":7,"n_ran":7,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/an-image-is-worth-one-word-personalizing-text","title":"An Image is Worth One Word: Personalizing Text-to-Image Generation using Textual Inversion","date":"2022-08-02","arxiv_id":"2208.01618","repositories_listed":9,"syntology":{"n":13,"n_ran":10,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/composer-creative-and-controllable-image","title":"Composer: Creative and Controllable Image Synthesis with Composable Conditions","date":"2023-02-20","arxiv_id":"2302.09778","repositories_listed":6,"syntology":null},{"url":"/paper/instructpix2pix-learning-to-follow-image","title":"InstructPix2Pix: Learning to Follow Image Editing Instructions","date":"2022-11-17","arxiv_id":"2211.09800","repositories_listed":6,"syntology":{"n":20,"n_ran":15,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/latent-consistency-models-synthesizing-high","title":"Latent Consistency Models: Synthesizing High-Resolution Images with Few-Step Inference","date":"2023-10-06","arxiv_id":"2310.04378","repositories_listed":5,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/muse-text-to-image-generation-via-masked","title":"Muse: Text-To-Image Generation via Masked Generative Transformers","date":"2023-01-02","arxiv_id":"2301.00704","repositories_listed":5,"syntology":{"n":21,"n_ran":18,"n_unverified":3,"n_pointer_only":9}},{"url":"/paper/styledrop-text-to-image-generation-in-any","title":"StyleDrop: Text-to-Image Generation in Any Style","date":"2023-06-01","arxiv_id":"2306.00983","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/masactrl-tuning-free-mutual-self-attention","title":"MasaCtrl: Tuning-Free Mutual Self-Attention Control for Consistent Image Synthesis and Editing","date":"2023-04-17","arxiv_id":"2304.08465","repositories_listed":4,"syntology":{"n":7,"n_ran":4,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/fast-text-conditional-discrete-denoising-on","title":"A Novel Sampling Scheme for Text- and Image-Conditional Image Synthesis in Quantized Latent Spaces","date":"2022-11-14","arxiv_id":"2211.07292","repositories_listed":4,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/cogview-mastering-text-to-image-generation","title":"CogView: Mastering Text-to-Image Generation via Transformers","date":"2021-05-26","arxiv_id":"2105.13290","repositories_listed":4,"syntology":{"n":6,"n_ran":4,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/navigating-the-synthetic-realm-harnessing","title":"Navigating the Synthetic Realm: Harnessing Diffusion-based Models for Laparoscopic Text-to-Image Generation","date":"2023-12-05","arxiv_id":"2312.03043","repositories_listed":3,"syntology":null},{"url":"/paper/imagereward-learning-and-evaluating-human-1","title":"ImageReward: Learning and Evaluating Human Preferences for Text-to-Image Generation","date":"2023-04-12","arxiv_id":"2304.05977","repositories_listed":3,"syntology":{"n":12,"n_ran":3,"n_unverified":9,"n_pointer_only":2}},{"url":"/paper/glyphdraw-learning-to-draw-chinese-characters","title":"GlyphDraw: Seamlessly Rendering Text with Intricate Spatial Structures in Text-to-Image Generation","date":"2023-03-31","arxiv_id":"2303.17870","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/one-transformer-fits-all-distributions-in","title":"One Transformer Fits All Distributions in Multi-Modal Diffusion at Scale","date":"2023-03-12","arxiv_id":"2303.06555","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/reduce-reuse-recycle-compositional-generation","title":"Reduce, Reuse, Recycle: Compositional Generation with Energy-Based Diffusion Models and MCMC","date":"2023-02-22","arxiv_id":"2302.11552","repositories_listed":3,"syntology":{"n":24,"n_ran":4,"n_unverified":20,"n_pointer_only":0}},{"url":"/paper/multidiffusion-fusing-diffusion-paths-for","title":"MultiDiffusion: Fusing Diffusion Paths for Controlled Image Generation","date":"2023-02-16","arxiv_id":"2302.08113","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/optimizing-prompts-for-text-to-image-1","title":"Optimizing Prompts for Text-to-Image Generation","date":"2022-12-19","arxiv_id":"2212.09611","repositories_listed":3,"syntology":{"n":9,"n_ran":1,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/dpm-solver-fast-solver-for-guided-sampling-of","title":"DPM-Solver++: Fast Solver for Guided Sampling of Diffusion Probabilistic Models","date":"2022-11-02","arxiv_id":"2211.01095","repositories_listed":3,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":6}},{"url":"/paper/all-are-worth-words-a-vit-backbone-for-score","title":"All are Worth Words: A ViT Backbone for Diffusion Models","date":"2022-09-25","arxiv_id":"2209.12152","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/lafite-towards-language-free-training-for","title":"LAFITE: Towards Language-Free Training for Text-to-Image Generation","date":"2021-11-27","arxiv_id":"2111.13792","repositories_listed":3,"syntology":{"n":18,"n_ran":4,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/conditional-image-generation-and-manipulation","title":"Conditional Image Generation and Manipulation for User-Specified Content","date":"2020-05-11","arxiv_id":"2005.04909","repositories_listed":3,"syntology":null},{"url":"/paper/keep-drawing-it-iterative-language-based","title":"Tell, Draw, and Repeat: Generating and Modifying Images Based on Continual Linguistic Instruction","date":"2018-11-24","arxiv_id":"1811.09845","repositories_listed":3,"syntology":null},{"url":"/paper/uniworld-v1-high-resolution-semantic-encoders","title":"UniWorld-V1: High-Resolution Semantic Encoders for Unified Visual Understanding and Generation","date":"2025-06-03","arxiv_id":"2506.03147","repositories_listed":2,"syntology":null},{"url":"/paper/hidream-i1-a-high-efficient-image-generative","title":"HiDream-I1: A High-Efficient Image Generative Foundation Model with Sparse Diffusion Transformer","date":"2025-05-28","arxiv_id":"2505.22705","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/imgedit-a-unified-image-editing-dataset-and","title":"ImgEdit: A Unified Image Editing Dataset and Benchmark","date":"2025-05-26","arxiv_id":"2505.20275","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/t2i-r1-reinforcing-image-generation-with","title":"T2I-R1: Reinforcing Image Generation with Collaborative Semantic-level and Token-level CoT","date":"2025-05-01","arxiv_id":"2505.00703","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/omni-dish-photorealistic-and-faithful-image","title":"Omni-Dish: Photorealistic and Faithful Image Generation and Editing for Arbitrary Chinese Dishes","date":"2025-04-14","arxiv_id":"2504.09948","repositories_listed":2,"syntology":null},{"url":"/paper/halton-scheduler-for-masked-generative-image","title":"Halton Scheduler For Masked Generative Image Transformer","date":"2025-03-21","arxiv_id":"2503.17076","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":0}}],"syntology_records":24,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}