{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-image-generation/papers/2","list_of":"/task/text-to-image-generation","task":"Text-to-Image Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":11,"rows_per_page":100,"rows":[101,200],"of":1085,"counts":{"archive_papers_tagged":1085,"with_a_code_link":546,"where_syntology_ran_a_sample":246,"not_listed_spam_title":0,"listed":1085,"listed_where_code_ran":246,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":215,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":215,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-image-generation","prev":"/task/text-to-image-generation","next":"/task/text-to-image-generation/papers/3","papers":[{"url":"/paper/glide-towards-photorealistic-image-generation","slug":"glide-towards-photorealistic-image-generation","title":"GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models","date":"2021-12-20","arxiv_id":"2112.10741","repositories_listed":2,"syntology":{"n":15,"n_ran":9,"n_constructed":7,"n_ran_checked":8,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 7 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/glide-towards-photorealistic-image-generation#ran","syntology_url":"https://syntology.ai/paper/2112.10741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.10741"}},"official":{"repos":["openai/glide-text2im"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":7,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/translation-equivariant-image-quantizer-for","slug":"translation-equivariant-image-quantizer-for","title":"Exploration into Translation-Equivariant Image Quantization","date":"2021-12-01","arxiv_id":"2112.00384","repositories_listed":2,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":4,"n_instrument":9,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":10,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 9 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/translation-equivariant-image-quantizer-for#ran","syntology_url":"https://syntology.ai/paper/2112.00384","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.00384"}},"official":{"repos":["wcshin-git/te-vqgan"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vector-quantized-diffusion-model-for-text-to","slug":"vector-quantized-diffusion-model-for-text-to","title":"Vector Quantized Diffusion Model for Text-to-Image Synthesis","date":"2021-11-29","arxiv_id":"2111.14822","repositories_listed":2,"syntology":null},{"url":"/paper/towards-open-world-text-guided-face-image","slug":"towards-open-world-text-guided-face-image","title":"Towards Open-World Text-Guided Face Image Generation and Manipulation","date":"2021-04-18","arxiv_id":"2104.08910","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-open-world-text-guided-face-image#ran","syntology_url":"https://syntology.ai/paper/2104.08910","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.08910"}},"official":{"repos":["weihaox/TediGAN"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/191013321","slug":"191013321","title":"Semantic Object Accuracy for Generative Text-to-Image Synthesis","date":"2019-10-29","arxiv_id":"1910.13321","repositories_listed":2,"syntology":{"n":23,"n_ran":18,"n_constructed":0,"n_ran_checked":14,"n_instrument":4,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":1,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/191013321#ran","syntology_url":"https://syntology.ai/paper/1910.13321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.13321"}},"official":{"repos":["tohinz/semantic-object-accuracy-for-generative-text-to-image-synthesis"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/controllable-text-to-image-generation","slug":"controllable-text-to-image-generation","title":"Controllable Text-to-Image Generation","date":"2019-09-16","arxiv_id":"1909.07083","repositories_listed":2,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/controllable-text-to-image-generation#ran","syntology_url":"https://syntology.ai/paper/1909.07083","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.07083"}},"official":{"repos":["mrlibw/ControlGAN"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/mirrorgan-learning-text-to-image-generation","slug":"mirrorgan-learning-text-to-image-generation","title":"MirrorGAN: Learning Text-to-image Generation by Redescription","date":"2019-03-14","arxiv_id":"1903.05854","repositories_listed":2,"syntology":null},{"url":"/paper/mc-gan-multi-conditional-generative","slug":"mc-gan-multi-conditional-generative","title":"MC-GAN: Multi-conditional Generative Adversarial Network for Image Synthesis","date":"2018-05-03","arxiv_id":"1805.01123","repositories_listed":2,"syntology":null},{"url":"/paper/generating-images-from-captions-with","slug":"generating-images-from-captions-with","title":"Generating Images from Captions with Attention","date":"2015-11-09","arxiv_id":"1511.02793","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/generating-images-from-captions-with#ran","syntology_url":"https://syntology.ai/paper/1511.02793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1511.02793"}},"official":{"repos":["emansim/text2image"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/characonsist-fine-grained-consistent","slug":"characonsist-fine-grained-consistent","title":"CharaConsist: Fine-Grained Consistent Character Generation","date":"2025-07-15","arxiv_id":"2507.11533","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/characonsist-fine-grained-consistent#ran","syntology_url":"https://syntology.ai/paper/2507.11533","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.11533"}},"official":{"repos":["murray-wang/characonsist"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/neobabel-a-multilingual-open-tower-for-visual","slug":"neobabel-a-multilingual-open-tower-for-visual","title":"NeoBabel: A Multilingual Open Tower for Visual Generation","date":"2025-07-08","arxiv_id":"2507.06137","repositories_listed":1,"syntology":null},{"url":"/paper/ovis-u1-technical-report","slug":"ovis-u1-technical-report","title":"Ovis-U1 Technical Report","date":"2025-06-29","arxiv_id":"2506.23044","repositories_listed":1,"syntology":null},{"url":"/paper/rethink-sparse-signals-for-pose-guided-text","slug":"rethink-sparse-signals-for-pose-guided-text","title":"Rethink Sparse Signals for Pose-guided Text-to-image Generation","date":"2025-06-26","arxiv_id":"2506.20983","repositories_listed":1,"syntology":null},{"url":"/paper/xverse-consistent-multi-subject-control-of","slug":"xverse-consistent-multi-subject-control-of","title":"XVerse: Consistent Multi-Subject Control of Identity and Semantic Attributes via DiT Modulation","date":"2025-06-26","arxiv_id":"2506.21416","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/xverse-consistent-multi-subject-control-of#ran","syntology_url":"https://syntology.ai/paper/2506.21416","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.21416"}},"official":{"repos":["bytedance/xverse"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sharegpt-4o-image-aligning-multimodal-models","slug":"sharegpt-4o-image-aligning-multimodal-models","title":"ShareGPT-4o-Image: Aligning Multimodal Models with GPT-4o-Level Image Generation","date":"2025-06-22","arxiv_id":"2506.18095","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sharegpt-4o-image-aligning-multimodal-models#ran","syntology_url":"https://syntology.ai/paper/2506.18095","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.18095"}},"official":{"repos":["freedomintelligence/sharegpt-4o-image"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/consistent-story-generation-with-asymmetry","slug":"consistent-story-generation-with-asymmetry","title":"Consistent Story Generation with Asymmetry Zigzag Sampling","date":"2025-06-11","arxiv_id":"2506.09612","repositories_listed":1,"syntology":null},{"url":"/paper/sage-exploring-the-boundaries-of-unsafe","slug":"sage-exploring-the-boundaries-of-unsafe","title":"SAGE: Exploring the Boundaries of Unsafe Concept Domain with Semantic-Augment Erasing","date":"2025-06-11","arxiv_id":"2506.09363","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-inversion-turns-clip-into-a-decoder","slug":"implicit-inversion-turns-clip-into-a-decoder","title":"Implicit Inversion turns CLIP into a Decoder","date":"2025-05-29","arxiv_id":"2505.23161","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-rag-sub-dimensional-retrieval","slug":"cross-modal-rag-sub-dimensional-retrieval","title":"Cross-modal RAG: Sub-dimensional Retrieval-Augmented Text-to-Image Generation","date":"2025-05-28","arxiv_id":"2505.21956","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-masked-autoregressive-models","slug":"hierarchical-masked-autoregressive-models","title":"Hierarchical Masked Autoregressive Models with Low-Resolution Token Pivots","date":"2025-05-26","arxiv_id":"2505.20288","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-llm-guided-semantic-correction-in","slug":"multimodal-llm-guided-semantic-correction-in","title":"Multimodal LLM-Guided Semantic Correction in Text-to-Image Diffusion","date":"2025-05-26","arxiv_id":"2505.20053","repositories_listed":1,"syntology":null},{"url":"/paper/strict-stress-test-of-rendering-images","slug":"strict-stress-test-of-rendering-images","title":"STRICT: Stress Test of Rendering Images Containing Text","date":"2025-05-25","arxiv_id":"2505.18985","repositories_listed":1,"syntology":null},{"url":"/paper/align-beyond-prompts-evaluating-world","slug":"align-beyond-prompts-evaluating-world","title":"Align Beyond Prompts: Evaluating World Knowledge Alignment in Text-to-Image Generation","date":"2025-05-24","arxiv_id":"2505.18730","repositories_listed":1,"syntology":null},{"url":"/paper/test-time-scaling-of-diffusion-models-via","slug":"test-time-scaling-of-diffusion-models-via","title":"Test-Time Scaling of Diffusion Models via Noise Trajectory Search","date":"2025-05-24","arxiv_id":"2506.03164","repositories_listed":1,"syntology":null},{"url":"/paper/co-reinforcement-learning-for-unified","slug":"co-reinforcement-learning-for-unified","title":"Co-Reinforcement Learning for Unified Multimodal Understanding and Generation","date":"2025-05-23","arxiv_id":"2505.17534","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-reinforcement-learning-for-unified#ran","syntology_url":"https://syntology.ai/paper/2505.17534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17534"}},"official":{"repos":["mm-vl/ulm-r1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/reprompt-reasoning-augmented-reprompting-for","slug":"reprompt-reasoning-augmented-reprompting-for","title":"RePrompt: Reasoning-Augmented Reprompting for Text-to-Image Generation via Reinforcement Learning","date":"2025-05-23","arxiv_id":"2505.17540","repositories_listed":1,"syntology":null},{"url":"/paper/mmada-multimodal-large-diffusion-language","slug":"mmada-multimodal-large-diffusion-language","title":"MMaDA: Multimodal Large Diffusion Language Models","date":"2025-05-21","arxiv_id":"2505.15809","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mmada-multimodal-large-diffusion-language#ran","syntology_url":"https://syntology.ai/paper/2505.15809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15809"}},"official":{"repos":["gen-verse/mmada"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-diffusion-transformers-efficiently","slug":"scaling-diffusion-transformers-efficiently","title":"Scaling Diffusion Transformers Efficiently via $μ$P","date":"2025-05-21","arxiv_id":"2505.15270","repositories_listed":1,"syntology":null},{"url":"/paper/mindomni-unleashing-reasoning-generation-in","slug":"mindomni-unleashing-reasoning-generation-in","title":"MindOmni: Unleashing Reasoning Generation in Vision Language Models with RGPO","date":"2025-05-19","arxiv_id":"2505.13031","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-sparsity-for-parameter-efficient","slug":"exploring-sparsity-for-parameter-efficient","title":"Exploring Sparsity for Parameter Efficient Fine Tuning Using Wavelets","date":"2025-05-18","arxiv_id":"2505.12532","repositories_listed":1,"syntology":null},{"url":"/paper/loft-lora-fused-training-dataset-generation","slug":"loft-lora-fused-training-dataset-generation","title":"LoFT: LoRA-fused Training Dataset Generation with Few-shot Guidance","date":"2025-05-16","arxiv_id":"2505.11703","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-deep-fusion-of-large-language","slug":"exploring-the-deep-fusion-of-large-language","title":"Exploring the Deep Fusion of Large Language Models and Diffusion Transformers for Text-to-Image Synthesis","date":"2025-05-15","arxiv_id":"2505.10046","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/exploring-the-deep-fusion-of-large-language#ran","syntology_url":"https://syntology.ai/paper/2505.10046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10046"}},"official":{"repos":["tang-bd/fuse-dit"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/hcma-hierarchical-cross-model-alignment-for","slug":"hcma-hierarchical-cross-model-alignment-for","title":"HCMA: Hierarchical Cross-model Alignment for Grounded Text-to-Image Generation","date":"2025-05-10","arxiv_id":"2505.06512","repositories_listed":1,"syntology":null},{"url":"/paper/flow-grpo-training-flow-matching-models-via","slug":"flow-grpo-training-flow-matching-models-via","title":"Flow-GRPO: Training Flow Matching Models via Online RL","date":"2025-05-08","arxiv_id":"2505.05470","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/flow-grpo-training-flow-matching-models-via#ran","syntology_url":"https://syntology.ai/paper/2505.05470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05470"}},"official":{"repos":["yifan123/flow_grpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-benchmarking-and-recommendation-of","slug":"multimodal-benchmarking-and-recommendation-of","title":"Multimodal Benchmarking and Recommendation of Text-to-Image Generation Models","date":"2025-05-06","arxiv_id":"2505.04650","repositories_listed":1,"syntology":null},{"url":"/paper/ming-lite-uni-advancements-in-unified","slug":"ming-lite-uni-advancements-in-unified","title":"Ming-Lite-Uni: Advancements in Unified Architecture for Natural Multimodal Interaction","date":"2025-05-05","arxiv_id":"2505.02471","repositories_listed":1,"syntology":null},{"url":"/paper/unified-multimodal-understanding-and","slug":"unified-multimodal-understanding-and","title":"Unified Multimodal Understanding and Generation Models: Advances, Challenges, and Opportunities","date":"2025-05-05","arxiv_id":"2505.02567","repositories_listed":1,"syntology":null},{"url":"/paper/freegraftor-training-free-cross-image-feature","slug":"freegraftor-training-free-cross-image-feature","title":"FreeGraftor: Training-Free Cross-Image Feature Grafting for Subject-Driven Text-to-Image Generation","date":"2025-04-22","arxiv_id":"2504.15958","repositories_listed":1,"syntology":null},{"url":"/paper/cross-attention-for-state-based-model-rwkv-7","slug":"cross-attention-for-state-based-model-rwkv-7","title":"Cross-attention for State-based model RWKV-7","date":"2025-04-19","arxiv_id":"2504.14260","repositories_listed":1,"syntology":null},{"url":"/paper/artistauditor-auditing-artist-style-pirate-in","slug":"artistauditor-auditing-artist-style-pirate-in","title":"ArtistAuditor: Auditing Artist Style Pirate in Text-to-Image Generation Models","date":"2025-04-17","arxiv_id":"2504.13061","repositories_listed":1,"syntology":null},{"url":"/paper/personalized-text-to-image-generation-with","slug":"personalized-text-to-image-generation-with","title":"Personalized Text-to-Image Generation with Auto-Regressive Models","date":"2025-04-17","arxiv_id":"2504.13162","repositories_listed":1,"syntology":null},{"url":"/paper/anchor-token-matching-implicit-structure","slug":"anchor-token-matching-implicit-structure","title":"Anchor Token Matching: Implicit Structure Locking for Training-free AR Image Editing","date":"2025-04-14","arxiv_id":"2504.10434","repositories_listed":1,"syntology":null},{"url":"/paper/dydit-dynamic-diffusion-transformers-for","slug":"dydit-dynamic-diffusion-transformers-for","title":"DyDiT++: Dynamic Diffusion Transformers for Efficient Visual Generation","date":"2025-04-09","arxiv_id":"2504.06803","repositories_listed":1,"syntology":null},{"url":"/paper/omnicaptioner-one-captioner-to-rule-them-all","slug":"omnicaptioner-one-captioner-to-rule-them-all","title":"OmniCaptioner: One Captioner to Rule Them All","date":"2025-04-09","arxiv_id":"2504.07089","repositories_listed":1,"syntology":null},{"url":"/paper/detection-limits-and-statistical-separability","slug":"detection-limits-and-statistical-separability","title":"Detection Limits and Statistical Separability of Tree Ring Watermarks in Rectified Flow-based Text-to-Image Generation Models","date":"2025-04-04","arxiv_id":"2504.03850","repositories_listed":1,"syntology":null},{"url":"/paper/less-to-more-generalization-unlocking-more","slug":"less-to-more-generalization-unlocking-more","title":"Less-to-More Generalization: Unlocking More Controllability by In-Context Generation","date":"2025-04-02","arxiv_id":"2504.02160","repositories_listed":1,"syntology":{"n":43,"n_ran":13,"n_constructed":0,"n_ran_checked":6,"n_instrument":7,"n_unverified":30,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 7 where Syntology's instrument failed) · 30 unverified","sample_list":"/paper/less-to-more-generalization-unlocking-more#ran","syntology_url":"https://syntology.ai/paper/2504.02160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02160"}},"official":{"repos":["bytedance/uno"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":30,"ran_from_kinds":["official"]}}},{"url":"/paper/ai2agent-an-end-to-end-framework-for","slug":"ai2agent-an-end-to-end-framework-for","title":"AI2Agent: An End-to-End Framework for Deploying AI Projects as Autonomous Agents","date":"2025-03-31","arxiv_id":"2503.23948","repositories_listed":1,"syntology":null},{"url":"/paper/lumina-image-2-0-a-unified-and-efficient","slug":"lumina-image-2-0-a-unified-and-efficient","title":"Lumina-Image 2.0: A Unified and Efficient Image Generative Framework","date":"2025-03-27","arxiv_id":"2503.21758","repositories_listed":1,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":11,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lumina-image-2-0-a-unified-and-efficient#ran","syntology_url":"https://syntology.ai/paper/2503.21758","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21758"}},"official":{"repos":["alpha-vllm/lumina-image-2.0"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/optimal-stepsize-for-diffusion-sampling","slug":"optimal-stepsize-for-diffusion-sampling","title":"Optimal Stepsize for Diffusion Sampling","date":"2025-03-27","arxiv_id":"2503.21774","repositories_listed":1,"syntology":null},{"url":"/paper/rectable-fast-modeling-tabular-data-with","slug":"rectable-fast-modeling-tabular-data-with","title":"RecTable: Fast Modeling Tabular Data with Rectified Flow","date":"2025-03-26","arxiv_id":"2503.20731","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-down-text-encoders-of-text-to-image","slug":"scaling-down-text-encoders-of-text-to-image","title":"Scaling Down Text Encoders of Text-to-Image Diffusion Models","date":"2025-03-25","arxiv_id":"2503.19897","repositories_listed":1,"syntology":null},{"url":"/paper/unseen-from-seen-rewriting-observation","slug":"unseen-from-seen-rewriting-observation","title":"Unseen from Seen: Rewriting Observation-Instruction Using Foundation Models for Augmenting Vision-Language Navigation","date":"2025-03-23","arxiv_id":"2503.18065","repositories_listed":1,"syntology":null},{"url":"/paper/verbdiff-text-only-diffusion-models-with","slug":"verbdiff-text-only-diffusion-models-with","title":"VerbDiff: Text-Only Diffusion Models with Enhanced Interaction Awareness","date":"2025-03-20","arxiv_id":"2503.16406","repositories_listed":1,"syntology":null},{"url":"/paper/reflect-dit-inference-time-scaling-for-text-1","slug":"reflect-dit-inference-time-scaling-for-text-1","title":"Reflect-DiT: Inference-Time Scaling for Text-to-Image Diffusion Transformers via In-Context Reflection","date":"2025-03-15","arxiv_id":"2503.12271","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/reflect-dit-inference-time-scaling-for-text-1#ran","syntology_url":"https://syntology.ai/paper/2503.12271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.12271"}},"official":null}},{"url":"/paper/towards-better-alignment-training-diffusion","slug":"towards-better-alignment-training-diffusion","title":"Towards Better Alignment: Training Diffusion Models with Reinforcement Learning Against Sparse Rewards","date":"2025-03-14","arxiv_id":"2503.11240","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-better-alignment-training-diffusion#ran","syntology_url":"https://syntology.ai/paper/2503.11240","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.11240"}},"official":{"repos":["hu-zijing/b2-diffurl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/got-unleashing-reasoning-capability-of","slug":"got-unleashing-reasoning-capability-of","title":"GoT: Unleashing Reasoning Capability of Multimodal Large Language Model for Visual Generation and Editing","date":"2025-03-13","arxiv_id":"2503.10639","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/got-unleashing-reasoning-capability-of#ran","syntology_url":"https://syntology.ai/paper/2503.10639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10639"}},"official":{"repos":["rongyaofang/got"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/neighboring-autoregressive-modeling-for","slug":"neighboring-autoregressive-modeling-for","title":"Neighboring Autoregressive Modeling for Efficient Visual Generation","date":"2025-03-12","arxiv_id":"2503.10696","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/neighboring-autoregressive-modeling-for#ran","syntology_url":"https://syntology.ai/paper/2503.10696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10696"}},"official":{"repos":["thisisbillhe/nar"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/aligning-text-to-image-in-diffusion-models-is","slug":"aligning-text-to-image-in-diffusion-models-is","title":"Aligning Text to Image in Diffusion Models is Easier Than You Think","date":"2025-03-11","arxiv_id":"2503.08250","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aligning-text-to-image-in-diffusion-models-is#ran","syntology_url":"https://syntology.ai/paper/2503.08250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08250"}},"official":{"repos":["softrepa/SoftREPA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lightgen-efficient-image-generation-through","slug":"lightgen-efficient-image-generation-through","title":"LightGen: Efficient Image Generation through Knowledge Distillation and Direct Preference Optimization","date":"2025-03-11","arxiv_id":"2503.08619","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":1,"n_honours":3,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 3 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lightgen-efficient-image-generation-through#ran","syntology_url":"https://syntology.ai/paper/2503.08619","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08619"}},"official":{"repos":["xianfengwu01/lightgen"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unleashing-the-potential-of-large-language-3","slug":"unleashing-the-potential-of-large-language-3","title":"Unleashing the Potential of Large Language Models for Text-to-Image Generation through Autoregressive Representation Alignment","date":"2025-03-10","arxiv_id":"2503.07334","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-alignment-and-noise-refinement","slug":"fine-grained-alignment-and-noise-refinement","title":"Fine-Grained Alignment and Noise Refinement for Compositional Text-to-Image Generation","date":"2025-03-09","arxiv_id":"2503.06506","repositories_listed":1,"syntology":null},{"url":"/paper/learning-few-step-diffusion-models-by","slug":"learning-few-step-diffusion-models-by","title":"Learning Few-Step Diffusion Models by Trajectory Distribution Matching","date":"2025-03-09","arxiv_id":"2503.06674","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-few-step-diffusion-models-by#ran","syntology_url":"https://syntology.ai/paper/2503.06674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06674"}},"official":{"repos":["Luo-Yihong/TDM"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/x2i-seamless-integration-of-multimodal","slug":"x2i-seamless-integration-of-multimodal","title":"X2I: Seamless Integration of Multimodal Understanding into Diffusion Transformer via Attention Distillation","date":"2025-03-08","arxiv_id":"2503.06134","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/x2i-seamless-integration-of-multimodal#ran","syntology_url":"https://syntology.ai/paper/2503.06134","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06134"}},"official":{"repos":["oppo-mente-lab/x2i"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/feynman-kac-correctors-in-diffusion-annealing","slug":"feynman-kac-correctors-in-diffusion-annealing","title":"Feynman-Kac Correctors in Diffusion: Annealing, Guidance, and Product of Experts","date":"2025-03-04","arxiv_id":"2503.02819","repositories_listed":1,"syntology":null},{"url":"/paper/generative-modeling-of-microweather-wind","slug":"generative-modeling-of-microweather-wind","title":"Generative Modeling of Microweather Wind Velocities for Urban Air Mobility","date":"2025-03-04","arxiv_id":"2503.02690","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-representation-alignment-for-image","slug":"multimodal-representation-alignment-for-image","title":"Multimodal Representation Alignment for Image Generation: Text-Image Interleaved Control Is Easier Than You Think","date":"2025-02-27","arxiv_id":"2502.20172","repositories_listed":1,"syntology":null},{"url":"/paper/forest-frame-of-reference-evaluation-in","slug":"forest-frame-of-reference-evaluation-in","title":"FoREST: Frame of Reference Evaluation in Spatial Reasoning Tasks","date":"2025-02-25","arxiv_id":"2502.17775","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-multimodal-models-for","slug":"multi-agent-multimodal-models-for","title":"Multi-Agent Multimodal Models for Multicultural Text to Image Generation","date":"2025-02-21","arxiv_id":"2502.15972","repositories_listed":1,"syntology":null},{"url":"/paper/chats-combining-human-aligned-optimization","slug":"chats-combining-human-aligned-optimization","title":"CHATS: Combining Human-Aligned Optimization and Test-Time Sampling for Text-to-Image Generation","date":"2025-02-18","arxiv_id":"2502.12579","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chats-combining-human-aligned-optimization#ran","syntology_url":"https://syntology.ai/paper/2502.12579","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.12579"}},"official":{"repos":["AIDC-AI/CHATS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-sample-effective-and-diverse","slug":"learning-to-sample-effective-and-diverse","title":"Learning to Sample Effective and Diverse Prompts for Text-to-Image Generation","date":"2025-02-17","arxiv_id":"2502.11477","repositories_listed":1,"syntology":null},{"url":"/paper/direct-ascent-synthesis-revealing-hidden","slug":"direct-ascent-synthesis-revealing-hidden","title":"Direct Ascent Synthesis: Revealing Hidden Generative Capabilities in Discriminative Models","date":"2025-02-11","arxiv_id":"2502.07753","repositories_listed":1,"syntology":null},{"url":"/paper/magic-1-for-1-generating-one-minute-video","slug":"magic-1-for-1-generating-one-minute-video","title":"Magic 1-For-1: Generating One Minute Video Clips within One Minute","date":"2025-02-11","arxiv_id":"2502.07701","repositories_listed":1,"syntology":null},{"url":"/paper/ruscode-russian-cultural-code-benchmark-for","slug":"ruscode-russian-cultural-code-benchmark-for","title":"RusCode: Russian Cultural Code Benchmark for Text-to-Image Generation","date":"2025-02-11","arxiv_id":"2502.07455","repositories_listed":1,"syntology":null},{"url":"/paper/sketchflex-facilitating-spatial-semantic","slug":"sketchflex-facilitating-spatial-semantic","title":"SketchFlex: Facilitating Spatial-Semantic Coherence in Text-to-Image Generation with Region-Based Sketches","date":"2025-02-11","arxiv_id":"2502.07556","repositories_listed":1,"syntology":null},{"url":"/paper/self-correcting-decoding-with-generative","slug":"self-correcting-decoding-with-generative","title":"Self-Correcting Decoding with Generative Feedback for Mitigating Hallucinations in Large Vision-Language Models","date":"2025-02-10","arxiv_id":"2502.06130","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/self-correcting-decoding-with-generative#ran","syntology_url":"https://syntology.ai/paper/2502.06130","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06130"}},"official":{"repos":["zhangce01/degf"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/beyond-fine-tuning-a-systematic-study-of","slug":"beyond-fine-tuning-a-systematic-study-of","title":"Beyond Fine-Tuning: A Systematic Study of Sampling Techniques in Personalized Image Generation","date":"2025-02-09","arxiv_id":"2502.05895","repositories_listed":1,"syntology":null},{"url":"/paper/unicms-a-unified-consistency-model-for","slug":"unicms-a-unified-consistency-model-for","title":"UniCMs: A Unified Consistency Model For Efficient Multimodal Generation and Understanding","date":"2025-02-08","arxiv_id":"2502.05415","repositories_listed":1,"syntology":null},{"url":"/paper/goku-flow-based-video-generative-foundation","slug":"goku-flow-based-video-generative-foundation","title":"Goku: Flow Based Video Generative Foundation Models","date":"2025-02-07","arxiv_id":"2502.04896","repositories_listed":1,"syntology":null},{"url":"/paper/llms-can-see-and-hear-without-any-training","slug":"llms-can-see-and-hear-without-any-training","title":"LLMs can see and hear without any training","date":"2025-01-30","arxiv_id":"2501.18096","repositories_listed":1,"syntology":null},{"url":"/paper/sana-1-5-efficient-scaling-of-training-time","slug":"sana-1-5-efficient-scaling-of-training-time","title":"SANA 1.5: Efficient Scaling of Training-Time and Inference-Time Compute in Linear Diffusion Transformer","date":"2025-01-30","arxiv_id":"2501.18427","repositories_listed":1,"syntology":null},{"url":"/paper/janus-pro-unified-multimodal-understanding","slug":"janus-pro-unified-multimodal-understanding","title":"Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling","date":"2025-01-29","arxiv_id":"2501.17811","repositories_listed":1,"syntology":null},{"url":"/paper/bringing-characters-to-new-stories-training","slug":"bringing-characters-to-new-stories-training","title":"IP-Prompter: Training-Free Theme-Specific Image Generation via Dynamic Visual Prompting","date":"2025-01-26","arxiv_id":"2501.15641","repositories_listed":1,"syntology":null},{"url":"/paper/one-prompt-one-story-free-lunch-consistent","slug":"one-prompt-one-story-free-lunch-consistent","title":"One-Prompt-One-Story: Free-Lunch Consistent Text-to-Image Generation Using a Single Prompt","date":"2025-01-23","arxiv_id":"2501.13554","repositories_listed":1,"syntology":null},{"url":"/paper/anystory-towards-unified-single-and-multiple","slug":"anystory-towards-unified-single-and-multiple","title":"AnyStory: Towards Unified Single and Multiple Subject Personalization in Text-to-Image Generation","date":"2025-01-16","arxiv_id":"2501.09503","repositories_listed":1,"syntology":null},{"url":"/paper/religious-bias-landscape-in-language-and-text","slug":"religious-bias-landscape-in-language-and-text","title":"Religious Bias Landscape in Language and Text-to-Image Models: Analysis, Detection, and Debiasing Strategies","date":"2025-01-14","arxiv_id":"2501.08441","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-text-to-image-generation-via","slug":"boosting-text-to-image-generation-via","title":"Boosting Text-To-Image Generation via Multilingual Prompting in Large Multimodal Models","date":"2025-01-13","arxiv_id":"2501.07086","repositories_listed":1,"syntology":null},{"url":"/paper/poetry-in-pixels-prompt-tuning-for-poem-image","slug":"poetry-in-pixels-prompt-tuning-for-poem-image","title":"Poetry in Pixels: Prompt Tuning for Poem Image Generation via Diffusion Models","date":"2025-01-10","arxiv_id":"2501.05839","repositories_listed":1,"syntology":null},{"url":"/paper/3dis-flux-simple-and-efficient-multi-instance","slug":"3dis-flux-simple-and-efficient-multi-instance","title":"3DIS-FLUX: simple and efficient multi-instance generation with DiT rendering","date":"2025-01-09","arxiv_id":"2501.05131","repositories_listed":1,"syntology":null},{"url":"/paper/face-makeup-multimodal-facial-prompts-for","slug":"face-makeup-multimodal-facial-prompts-for","title":"Face-MakeUp: Multimodal Facial Prompts for Text-to-Image Generation","date":"2025-01-05","arxiv_id":"2501.02523","repositories_listed":1,"syntology":null},{"url":"/paper/eligen-entity-level-controlled-image","slug":"eligen-entity-level-controlled-image","title":"EliGen: Entity-Level Controlled Image Generation with Regional Attention","date":"2025-01-02","arxiv_id":"2501.01097","repositories_listed":1,"syntology":null},{"url":"/paper/dual-diffusion-for-unified-image-generation","slug":"dual-diffusion-for-unified-image-generation","title":"Dual Diffusion for Unified Image Generation and Understanding","date":"2024-12-31","arxiv_id":"2501.00289","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dual-diffusion-for-unified-image-generation#ran","syntology_url":"https://syntology.ai/paper/2501.00289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.00289"}},"official":null}},{"url":"/paper/token-pruning-for-caching-better-9-times","slug":"token-pruning-for-caching-better-9-times","title":"Token Pruning for Caching Better: 9 Times Acceleration on Stable Diffusion for Free","date":"2024-12-31","arxiv_id":"2501.00375","repositories_listed":1,"syntology":null},{"url":"/paper/vmix-improving-text-to-image-diffusion-model","slug":"vmix-improving-text-to-image-diffusion-model","title":"VMix: Improving Text-to-Image Diffusion Model with Cross-Attention Mixing Control","date":"2024-12-30","arxiv_id":"2412.20800","repositories_listed":1,"syntology":null},{"url":"/paper/evalmuse-40k-a-reliable-and-fine-grained","slug":"evalmuse-40k-a-reliable-and-fine-grained","title":"EvalMuse-40K: A Reliable and Fine-Grained Benchmark with Comprehensive Human Annotations for Text-to-Image Generation Model Evaluation","date":"2024-12-24","arxiv_id":"2412.18150","repositories_listed":1,"syntology":null},{"url":"/paper/extract-free-dense-misalignment-from-clip","slug":"extract-free-dense-misalignment-from-clip","title":"Extract Free Dense Misalignment from CLIP","date":"2024-12-24","arxiv_id":"2412.18404","repositories_listed":1,"syntology":null},{"url":"/paper/distilled-decoding-1-one-step-sampling-of","slug":"distilled-decoding-1-one-step-sampling-of","title":"Distilled Decoding 1: One-step Sampling of Image Auto-regressive Models with Flow Matching","date":"2024-12-22","arxiv_id":"2412.17153","repositories_listed":1,"syntology":null},{"url":"/paper/autoregressive-video-generation-without","slug":"autoregressive-video-generation-without","title":"Autoregressive Video Generation without Vector Quantization","date":"2024-12-18","arxiv_id":"2412.14169","repositories_listed":1,"syntology":{"n":26,"n_ran":19,"n_constructed":18,"n_ran_checked":19,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":19,"n_pointer_only":0,"phrase":"19 ran (of which 18 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/autoregressive-video-generation-without#ran","syntology_url":"https://syntology.ai/paper/2412.14169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14169"}},"official":{"repos":["baaivision/nova"],"state":"official (archive's flag): 19 ran","n_ran":19,"n_constructed":18,"n_ran_no_instrument_failure":19,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/artaug-enhancing-text-to-image-generation","slug":"artaug-enhancing-text-to-image-generation","title":"ArtAug: Enhancing Text-to-Image Generation through Synthesis-Understanding Interaction","date":"2024-12-17","arxiv_id":"2412.12888","repositories_listed":1,"syntology":null},{"url":"/paper/fast-prompt-alignment-for-text-to-image","slug":"fast-prompt-alignment-for-text-to-image","title":"Fast Prompt Alignment for Text-to-Image Generation","date":"2024-12-11","arxiv_id":"2412.08639","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fast-prompt-alignment-for-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2412.08639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08639"}},"official":{"repos":["tiktok/fast_prompt_alignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generate-any-scene-evaluating-and-improving","slug":"generate-any-scene-evaluating-and-improving","title":"Generate Any Scene: Evaluating and Improving Text-to-Vision Generation with Scene Graph Programming","date":"2024-12-11","arxiv_id":"2412.08221","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":1,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generate-any-scene-evaluating-and-improving#ran","syntology_url":"https://syntology.ai/paper/2412.08221","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08221"}},"official":{"repos":["RAIVNLab/GenerateAnyScene"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"aa8e8309dddf1df6e11b175f28fc8f5621f8dbc808c83c8457e9f888bac9b4c0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}