{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-image-generation-1/papers/5","list_of":"/task/text-to-image-generation-1","task":"Text to Image Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":10,"rows_per_page":100,"rows":[401,500],"of":969,"counts":{"archive_papers_tagged":969,"with_a_code_link":461,"where_syntology_ran_a_sample":198,"not_listed_spam_title":0,"listed":969,"listed_where_code_ran":198,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":171,"every_run_a_failure_of_syntologys_instrument":27,"listed_with_a_run_with_no_instrument_failure":171,"listed_every_run_a_failure_of_syntologys_instrument":27,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-image-generation-1","prev":"/task/text-to-image-generation-1/papers/4","next":"/task/text-to-image-generation-1/papers/6","papers":[{"url":"/paper/multimodal-procedural-planning-via-dual-text","slug":"multimodal-procedural-planning-via-dual-text","title":"Multimodal Procedural Planning via Dual Text-Image Prompting","date":"2023-05-02","arxiv_id":"2305.01795","repositories_listed":1,"syntology":null},{"url":"/paper/pick-a-pic-an-open-dataset-of-user","slug":"pick-a-pic-an-open-dataset-of-user","title":"Pick-a-Pic: An Open Dataset of User Preferences for Text-to-Image Generation","date":"2023-05-02","arxiv_id":"2305.01569","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pick-a-pic-an-open-dataset-of-user#ran","syntology_url":"https://syntology.ai/paper/2305.01569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.01569"}},"official":{"repos":["yuvalkirstain/pickscore"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/expressive-text-to-image-generation-with-rich","slug":"expressive-text-to-image-generation-with-rich","title":"Expressive Text-to-Image Generation with Rich Text","date":"2023-04-13","arxiv_id":"2304.06720","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/expressive-text-to-image-generation-with-rich#ran","syntology_url":"https://syntology.ai/paper/2304.06720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.06720"}},"official":{"repos":["songweige/rich-text-to-image"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/controllable-textual-inversion-for","slug":"controllable-textual-inversion-for","title":"Controllable Textual Inversion for Personalized Text-to-Image Generation","date":"2023-04-11","arxiv_id":"2304.05265","repositories_listed":1,"syntology":null},{"url":"/paper/hrs-bench-holistic-reliable-and-scalable","slug":"hrs-bench-holistic-reliable-and-scalable","title":"HRS-Bench: Holistic, Reliable and Scalable Benchmark for Text-to-Image Models","date":"2023-04-11","arxiv_id":"2304.05390","repositories_listed":1,"syntology":null},{"url":"/paper/uncurated-image-text-datasets-shedding-light","slug":"uncurated-image-text-datasets-shedding-light","title":"Uncurated Image-Text Datasets: Shedding Light on Demographic Bias","date":"2023-04-06","arxiv_id":"2304.02828","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/uncurated-image-text-datasets-shedding-light#ran","syntology_url":"https://syntology.ai/paper/2304.02828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.02828"}},"official":{"repos":["noagarcia/phase"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/forget-me-not-learning-to-forget-in-text-to","slug":"forget-me-not-learning-to-forget-in-text-to","title":"Forget-Me-Not: Learning to Forget in Text-to-Image Diffusion Models","date":"2023-03-30","arxiv_id":"2303.17591","repositories_listed":1,"syntology":null},{"url":"/paper/indonesian-text-to-image-synthesis-with","slug":"indonesian-text-to-image-synthesis-with","title":"Indonesian Text-to-Image Synthesis with Sentence-BERT and FastGAN","date":"2023-03-25","arxiv_id":"2303.14517","repositories_listed":1,"syntology":null},{"url":"/paper/medical-diffusion-on-a-budget-textual","slug":"medical-diffusion-on-a-budget-textual","title":"Medical diffusion on a budget: Textual Inversion for medical image generation","date":"2023-03-23","arxiv_id":"2303.13430","repositories_listed":1,"syntology":null},{"url":"/paper/magvlt-masked-generative-vision-and-language","slug":"magvlt-masked-generative-vision-and-language","title":"MAGVLT: Masked Generative Vision-and-Language Transformer","date":"2023-03-21","arxiv_id":"2303.12208","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":5,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/magvlt-masked-generative-vision-and-language#ran","syntology_url":"https://syntology.ai/paper/2303.12208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.12208"}},"official":{"repos":["kakaobrain/magvlt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":5,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/tifa-accurate-and-interpretable-text-to-image","slug":"tifa-accurate-and-interpretable-text-to-image","title":"TIFA: Accurate and Interpretable Text-to-Image Faithfulness Evaluation with Question Answering","date":"2023-03-21","arxiv_id":"2303.11897","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tifa-accurate-and-interpretable-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2303.11897","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11897"}},"official":{"repos":["Yushi-Hu/tifa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/localizing-object-level-shape-variations-with","slug":"localizing-object-level-shape-variations-with","title":"Localizing Object-level Shape Variations with Text-to-Image Diffusion Models","date":"2023-03-20","arxiv_id":"2303.11306","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/localizing-object-level-shape-variations-with#ran","syntology_url":"https://syntology.ai/paper/2303.11306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11306"}},"official":null}},{"url":"/paper/svdiff-compact-parameter-space-for-diffusion","slug":"svdiff-compact-parameter-space-for-diffusion","title":"SVDiff: Compact Parameter Space for Diffusion Fine-Tuning","date":"2023-03-20","arxiv_id":"2303.11305","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/svdiff-compact-parameter-space-for-diffusion#ran","syntology_url":"https://syntology.ai/paper/2303.11305","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11305"}},"official":null}},{"url":"/paper/p-extended-textual-conditioning-in-text-to","slug":"p-extended-textual-conditioning-in-text-to","title":"P+: Extended Textual Conditioning in Text-to-Image Generation","date":"2023-03-16","arxiv_id":"2303.09522","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/p-extended-textual-conditioning-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2303.09522","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.09522"}},"official":null}},{"url":"/paper/a-prompt-log-analysis-of-text-to-image","slug":"a-prompt-log-analysis-of-text-to-image","title":"A Prompt Log Analysis of Text-to-Image Generation Systems","date":"2023-03-08","arxiv_id":"2303.04587","repositories_listed":1,"syntology":null},{"url":"/paper/teaching-clip-to-count-to-ten","slug":"teaching-clip-to-count-to-ten","title":"Teaching CLIP to Count to Ten","date":"2023-02-23","arxiv_id":"2302.12066","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/teaching-clip-to-count-to-ten#ran","syntology_url":"https://syntology.ai/paper/2302.12066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12066"}},"official":null}},{"url":"/paper/prompt-stealing-attacks-against-text-to-image","slug":"prompt-stealing-attacks-against-text-to-image","title":"Prompt Stealing Attacks Against Text-to-Image Generation Models","date":"2023-02-20","arxiv_id":"2302.09923","repositories_listed":1,"syntology":null},{"url":"/paper/is-this-loss-informative-faster-text-to-image-1","slug":"is-this-loss-informative-faster-text-to-image-1","title":"Is This Loss Informative? Faster Text-to-Image Customization by Tracking Objective Dynamics","date":"2023-02-09","arxiv_id":"2302.04841","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/is-this-loss-informative-faster-text-to-image-1#ran","syntology_url":"https://syntology.ai/paper/2302.04841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.04841"}},"official":{"repos":["yandex-research/dvar"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/boosting-zero-shot-classification-with","slug":"boosting-zero-shot-classification-with","title":"Diversity is Definitely Needed: Improving Model-Agnostic Zero-shot Classification via Stable Diffusion","date":"2023-02-07","arxiv_id":"2302.03298","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-zero-shot-classification-with#ran","syntology_url":"https://syntology.ai/paper/2302.03298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.03298"}},"official":{"repos":["jordan-hs/diversity_is_definitely_needed"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fair-diffusion-instructing-text-to-image","slug":"fair-diffusion-instructing-text-to-image","title":"Fair Diffusion: Instructing Text-to-Image Generation Models on Fairness","date":"2023-02-07","arxiv_id":"2302.10893","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fair-diffusion-instructing-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2302.10893","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.10893"}},"official":{"repos":["ml-research/fair-diffusion"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/eliminating-prior-bias-for-semantic-image","slug":"eliminating-prior-bias-for-semantic-image","title":"Eliminating Contextual Prior Bias for Semantic Image Editing via Dual-Cycle Diffusion","date":"2023-02-05","arxiv_id":"2302.02394","repositories_listed":1,"syntology":null},{"url":"/paper/accelerating-guided-diffusion-sampling-with","slug":"accelerating-guided-diffusion-sampling-with","title":"Accelerating Guided Diffusion Sampling with Splitting Numerical Methods","date":"2023-01-27","arxiv_id":"2301.11558","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accelerating-guided-diffusion-sampling-with#ran","syntology_url":"https://syntology.ai/paper/2301.11558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.11558"}},"official":{"repos":["swizad/split-diffusion"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/gligen-open-set-grounded-text-to-image","slug":"gligen-open-set-grounded-text-to-image","title":"GLIGEN: Open-Set Grounded Text-to-Image Generation","date":"2023-01-17","arxiv_id":"2301.07093","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/gligen-open-set-grounded-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2301.07093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.07093"}},"official":{"repos":["gligen/GLIGEN"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/riatig-reliable-and-imperceptible-adversarial","slug":"riatig-reliable-and-imperceptible-adversarial","title":"RIATIG: Reliable and Imperceptible Adversarial Text-to-Image Generation With Natural Prompts","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-spatial-relationships-in-text-to","slug":"benchmarking-spatial-relationships-in-text-to","title":"Benchmarking Spatial Relationships in Text-to-Image Generation","date":"2022-12-20","arxiv_id":"2212.10015","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-spatial-relationships-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2212.10015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.10015"}},"official":{"repos":["microsoft/VISOR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/are-multimodal-models-robust-to-image-and","slug":"are-multimodal-models-robust-to-image-and","title":"Benchmarking Robustness of Multimodal Image-Text Models under Distribution Shift","date":"2022-12-15","arxiv_id":"2212.08044","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-multimodal-models-robust-to-image-and#ran","syntology_url":"https://syntology.ai/paper/2212.08044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.08044"}},"official":null}},{"url":"/paper/smartbrush-text-and-shape-guided-object","slug":"smartbrush-text-and-shape-guided-object","title":"SmartBrush: Text and Shape Guided Object Inpainting with Diffusion Model","date":"2022-12-09","arxiv_id":"2212.05034","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/smartbrush-text-and-shape-guided-object#ran","syntology_url":"https://syntology.ai/paper/2212.05034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05034"}},"official":null}},{"url":"/paper/shifted-diffusion-for-text-to-image","slug":"shifted-diffusion-for-text-to-image","title":"Shifted Diffusion for Text-to-image Generation","date":"2022-11-24","arxiv_id":"2211.15388","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":10,"n_instrument":3,"n_unverified":4,"n_honours":2,"n_violates":1,"n_no_contract":7,"n_pointer_only":6,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/shifted-diffusion-for-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2211.15388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.15388"}},"official":{"repos":["drboog/Shifted_Diffusion"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/easily-accessible-text-to-image-generation","slug":"easily-accessible-text-to-image-generation","title":"Easily Accessible Text-to-Image Generation Amplifies Demographic Stereotypes at Large Scale","date":"2022-11-07","arxiv_id":"2211.03759","repositories_listed":1,"syntology":null},{"url":"/paper/how-well-can-text-to-image-generative-models","slug":"how-well-can-text-to-image-generative-models","title":"How well can Text-to-Image Generative Models understand Ethical Natural Language Interventions?","date":"2022-10-27","arxiv_id":"2210.15230","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/how-well-can-text-to-image-generative-models#ran","syntology_url":"https://syntology.ai/paper/2210.15230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.15230"}},"official":{"repos":["hritikbansal/entigen_emnlp"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-better-text-image-consistency-in-text","slug":"towards-better-text-image-consistency-in-text","title":"SSD: Towards Better Text-Image Consistency Metric in Text-to-Image Generation","date":"2022-10-27","arxiv_id":"2210.15235","repositories_listed":1,"syntology":null},{"url":"/paper/z-lavi-zero-shot-language-solver-fueled-by","slug":"z-lavi-zero-shot-language-solver-fueled-by","title":"Z-LaVI: Zero-Shot Language Solver Fueled by Visual Imagination","date":"2022-10-21","arxiv_id":"2210.12261","repositories_listed":1,"syntology":null},{"url":"/paper/is-synthetic-data-from-generative-models","slug":"is-synthetic-data-from-generative-models","title":"Is synthetic data from generative models ready for image recognition?","date":"2022-10-14","arxiv_id":"2210.07574","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/is-synthetic-data-from-generative-models#ran","syntology_url":"https://syntology.ai/paper/2210.07574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07574"}},"official":{"repos":["cvmi-lab/syntheticdata"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/personalizing-text-to-image-generation-via","slug":"personalizing-text-to-image-generation-via","title":"Personalizing Text-to-Image Generation via Aesthetic Gradients","date":"2022-09-25","arxiv_id":"2209.12330","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":1,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":1,"n_pointer_only":6,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 2 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/personalizing-text-to-image-generation-via#ran","syntology_url":"https://syntology.ai/paper/2209.12330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.12330"}},"official":{"repos":["vicgalle/stable-diffusion-aesthetic-gradients"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/implementing-and-experimenting-with-diffusion","slug":"implementing-and-experimenting-with-diffusion","title":"Implementing and Experimenting with Diffusion Models for Text-to-Image Generation","date":"2022-09-22","arxiv_id":"2209.10948","repositories_listed":1,"syntology":null},{"url":"/paper/t-person-gan-text-to-person-image-generation","slug":"t-person-gan-text-to-person-image-generation","title":"T-Person-GAN: Text-to-Person Image Generation with Identity-Consistency and Manifold Mix-Up","date":"2022-08-18","arxiv_id":"2208.12752","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-attention-for-vision-and-1","slug":"understanding-attention-for-vision-and-1","title":"Understanding Attention for Vision-and-Language Tasks","date":"2022-08-17","arxiv_id":"2208.08104","repositories_listed":1,"syntology":null},{"url":"/paper/txt2img-mhn-remote-sensing-image-generation","slug":"txt2img-mhn-remote-sensing-image-generation","title":"Txt2Img-MHN: Remote Sensing Image Generation from Text Using Modern Hopfield Networks","date":"2022-08-08","arxiv_id":"2208.04441","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/txt2img-mhn-remote-sensing-image-generation#ran","syntology_url":"https://syntology.ai/paper/2208.04441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.04441"}},"official":{"repos":["yonghaoxu/txt2img-mhn"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-generative-adversarial-networks-for-1","slug":"exploring-generative-adversarial-networks-for-1","title":"Exploring Generative Adversarial Networks for Text-to-Image Generation with Evolution Strategies","date":"2022-07-06","arxiv_id":"2207.02907","repositories_listed":1,"syntology":null},{"url":"/paper/prefix-language-models-are-unified-modal","slug":"prefix-language-models-are-unified-modal","title":"Write and Paint: Generative Vision-Language Models are Unified Modal Learners","date":"2022-06-15","arxiv_id":"2206.07699","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/prefix-language-models-are-unified-modal#ran","syntology_url":"https://syntology.ai/paper/2206.07699","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07699"}},"official":{"repos":["shizhediao/davinci"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mutual-information-divergence-a-unified","slug":"mutual-information-divergence-a-unified","title":"Mutual Information Divergence: A Unified Metric for Multimodal Generative Models","date":"2022-05-25","arxiv_id":"2205.13445","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mutual-information-divergence-a-unified#ran","syntology_url":"https://syntology.ai/paper/2205.13445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.13445"}},"official":{"repos":["naver-ai/mid.metric"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gr-gan-gradual-refinement-text-to-image","slug":"gr-gan-gradual-refinement-text-to-image","title":"GR-GAN: Gradual Refinement Text-to-image Generation","date":"2022-05-23","arxiv_id":"2205.11273","repositories_listed":1,"syntology":null},{"url":"/paper/conditional-vector-graphics-generation-for","slug":"conditional-vector-graphics-generation-for","title":"Conditional Vector Graphics Generation for Music Cover Images","date":"2022-05-15","arxiv_id":"2205.07301","repositories_listed":1,"syntology":null},{"url":"/paper/zero-and-r2d2-a-large-scale-chinese-cross","slug":"zero-and-r2d2-a-large-scale-chinese-cross","title":"CCMB: A Large-scale Chinese Cross-modal Benchmark","date":"2022-05-08","arxiv_id":"2205.03860","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/zero-and-r2d2-a-large-scale-chinese-cross#ran","syntology_url":"https://syntology.ai/paper/2205.03860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.03860"}},"official":{"repos":["yuxie11/R2D2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cogview2-faster-and-better-text-to-image","slug":"cogview2-faster-and-better-text-to-image","title":"CogView2: Faster and Better Text-to-Image Generation via Hierarchical Transformers","date":"2022-04-28","arxiv_id":"2204.14217","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cogview2-faster-and-better-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2204.14217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.14217"}},"official":{"repos":["thudm/cogview2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dr-gan-distribution-regularization-for-text","slug":"dr-gan-distribution-regularization-for-text","title":"DR-GAN: Distribution Regularization for Text-to-Image Generation","date":"2022-04-17","arxiv_id":"2204.07945","repositories_listed":1,"syntology":null},{"url":"/paper/make-a-scene-scene-based-text-to-image","slug":"make-a-scene-scene-based-text-to-image","title":"Make-A-Scene: Scene-Based Text-to-Image Generation with Human Priors","date":"2022-03-24","arxiv_id":"2203.13131","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/make-a-scene-scene-based-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2203.13131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13131"}},"official":null}},{"url":"/paper/gan-based-matrix-factorization-for","slug":"gan-based-matrix-factorization-for","title":"GAN-based Matrix Factorization for Recommender Systems","date":"2022-01-20","arxiv_id":"2201.08042","repositories_listed":1,"syntology":null},{"url":"/paper/fusedream-training-free-text-to-image","slug":"fusedream-training-free-text-to-image","title":"FuseDream: Training-Free Text-to-Image Generation with Improved CLIP+GAN Space Optimization","date":"2021-12-02","arxiv_id":"2112.01573","repositories_listed":1,"syntology":null},{"url":"/paper/nuwa-visual-synthesis-pre-training-for-neural","slug":"nuwa-visual-synthesis-pre-training-for-neural","title":"NÜWA: Visual Synthesis Pre-training for Neural visUal World creAtion","date":"2021-11-24","arxiv_id":"2111.12417","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nuwa-visual-synthesis-pre-training-for-neural#ran","syntology_url":"https://syntology.ai/paper/2111.12417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12417"}},"official":null}},{"url":"/paper/l-verse-bidirectional-generation-between","slug":"l-verse-bidirectional-generation-between","title":"L-Verse: Bidirectional Generation Between Image and Text","date":"2021-11-22","arxiv_id":"2111.11133","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-multimodal-transformer-for-bi","slug":"unifying-multimodal-transformer-for-bi","title":"Unifying Multimodal Transformer for Bi-directional Image and Text Generation","date":"2021-10-19","arxiv_id":"2110.09753","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unifying-multimodal-transformer-for-bi#ran","syntology_url":"https://syntology.ai/paper/2110.09753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.09753"}},"official":{"repos":["researchmm/generate-it"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/clip-forge-towards-zero-shot-text-to-shape","slug":"clip-forge-towards-zero-shot-text-to-shape","title":"CLIP-Forge: Towards Zero-Shot Text-to-Shape Generation","date":"2021-10-06","arxiv_id":"2110.02624","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-image-generation-from-bangla","slug":"fine-grained-image-generation-from-bangla","title":"Fine-Grained Image Generation from Bangla Text Description using Attentional Generative Adversarial Network","date":"2021-09-24","arxiv_id":"2109.11749","repositories_listed":1,"syntology":null},{"url":"/paper/paint4poem-a-dataset-for-artistic","slug":"paint4poem-a-dataset-for-artistic","title":"Paint4Poem: A Dataset for Artistic Visualization of Classical Chinese Poems","date":"2021-09-23","arxiv_id":"2109.11682","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-computation-of-monge-maps-with","slug":"scalable-computation-of-monge-maps-with","title":"Neural Monge Map estimation and its applications","date":"2021-06-07","arxiv_id":"2106.03812","repositories_listed":1,"syntology":null},{"url":"/paper/text-to-image-generation-with-semantic","slug":"text-to-image-generation-with-semantic","title":"Text to Image Generation with Semantic-Spatial Aware GAN","date":"2021-04-01","arxiv_id":"2104.00567","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-contrastive-learning-for-text-to","slug":"cross-modal-contrastive-learning-for-text-to","title":"Cross-Modal Contrastive Learning for Text-to-Image Generation","date":"2021-01-12","arxiv_id":"2101.04702","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cross-modal-contrastive-learning-for-text-to#ran","syntology_url":"https://syntology.ai/paper/2101.04702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.04702"}},"official":{"repos":["google-research/xmcgan_image_generation"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/text-to-image-generation-grounded-by-fine","slug":"text-to-image-generation-grounded-by-fine","title":"Text-to-Image Generation Grounded by Fine-Grained User Attention","date":"2020-11-07","arxiv_id":"2011.03775","repositories_listed":1,"syntology":null},{"url":"/paper/network-fusion-for-content-creation-with","slug":"network-fusion-for-content-creation-with","title":"Network-to-Network Translation with Conditional Invertible Neural Networks","date":"2020-05-27","arxiv_id":"2005.13580","repositories_listed":1,"syntology":null},{"url":"/paper/learn-imagine-and-create-text-to-image","slug":"learn-imagine-and-create-text-to-image","title":"Learn, Imagine and Create: Text-to-Image Generation from Prior Knowledge","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":null,"slug":"evaluating-attribute-confusion-in-fashion","title":"Evaluating Attribute Confusion in Fashion Text-to-Image Generation","date":"2025-07-09","arxiv_id":"2507.07079","repositories_listed":0,"syntology":null},{"url":null,"slug":"dc-ar-efficient-masked-autoregressive-image","title":"DC-AR: Efficient Masked Autoregressive Image Generation with Deep Compression Hybrid Tokenizer","date":"2025-07-07","arxiv_id":"2507.04947","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniglyph-unified-segmentation-conditioned","title":"UniGlyph: Unified Segmentation-Conditioned Diffusion for Precise Visual Text Synthesis","date":"2025-07-01","arxiv_id":"2507.00992","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-tree-sampling-scalable-inference","title":"Diffusion Tree Sampling: Scalable inference-time alignment of diffusion models","date":"2025-06-25","arxiv_id":"2506.20701","repositories_listed":0,"syntology":null},{"url":null,"slug":"med-art-diffusion-transformer-for-2d-medical","title":"Med-Art: Diffusion Transformer for 2D Medical Text-to-Image Generation","date":"2025-06-25","arxiv_id":"2506.20449","repositories_listed":0,"syntology":null},{"url":null,"slug":"cost-aware-routing-for-efficient-text-to","title":"Cost-Aware Routing for Efficient Text-To-Image Generation","date":"2025-06-17","arxiv_id":"2506.14753","repositories_listed":0,"syntology":null},{"url":null,"slug":"fair-generation-without-unfair-distortions","title":"Fair Generation without Unfair Distortions: Debiasing Text-to-Image Generation with Entanglement-Free Attention","date":"2025-06-16","arxiv_id":"2506.13298","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmmg-a-massive-multidisciplinary-multi-tier","title":"MMMG: A Massive, Multidisciplinary, Multi-Tier Generation Benchmark for Text-to-Image Reasoning","date":"2025-06-12","arxiv_id":"2506.10963","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-image-for-multi-label-image","title":"Text to Image for Multi-Label Image Recognition with Joint Prompt-Adapter Learning","date":"2025-06-12","arxiv_id":"2506.10575","repositories_listed":0,"syntology":null},{"url":null,"slug":"re-thinking-the-automatic-evaluation-of-image","title":"Re-Thinking the Automatic Evaluation of Image-Text Alignment in Text-to-Image Models","date":"2025-06-10","arxiv_id":"2506.08480","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-study-of-decoder-only-llms-1","title":"A Comprehensive Study of Decoder-Only LLMs for Text-to-Image Generation","date":"2025-06-09","arxiv_id":"2506.08210","repositories_listed":0,"syntology":null},{"url":null,"slug":"vivat-virtuous-improving-vae-training-through","title":"VIVAT: Virtuous Improving VAE Training through Artifact Mitigation","date":"2025-06-09","arxiv_id":"2506.07863","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-inverse-problems-with-flair","title":"Solving Inverse Problems with FLAIR","date":"2025-06-03","arxiv_id":"2506.02680","repositories_listed":0,"syntology":null},{"url":null,"slug":"pointt2i-llm-based-text-to-image-generation","title":"PointT2I: LLM-based text-to-image generation via keypoints","date":"2025-06-02","arxiv_id":"2506.01370","repositories_listed":0,"syntology":null},{"url":null,"slug":"unlocking-aha-moments-via-reinforcement","title":"Unlocking Aha Moments via Reinforcement Learning: Advancing Collaborative Visual Comprehension and Generation","date":"2025-06-02","arxiv_id":"2506.01480","repositories_listed":0,"syntology":null},{"url":null,"slug":"artiscene-language-driven-artistic-3d-scene-1","title":"ArtiScene: Language-Driven Artistic 3D Scene Generation Through Image Intermediary","date":"2025-05-31","arxiv_id":"2506.00742","repositories_listed":0,"syntology":null},{"url":null,"slug":"composeanything-composite-object-priors-for","title":"ComposeAnything: Composite Object Priors for Text-to-Image Generation","date":"2025-05-30","arxiv_id":"2505.24086","repositories_listed":0,"syntology":null},{"url":null,"slug":"r2i-bench-benchmarking-reasoning-driven-text","title":"R2I-Bench: Benchmarking Reasoning-Driven Text-to-Image Generation","date":"2025-05-29","arxiv_id":"2505.23493","repositories_listed":0,"syntology":null},{"url":null,"slug":"rhetorical-text-to-image-generation-via-two","title":"Rhetorical Text-to-Image Generation via Two-layer Diffusion Policy Optimization","date":"2025-05-28","arxiv_id":"2505.22792","repositories_listed":0,"syntology":null},{"url":null,"slug":"stylear-customizing-multimodal-autoregressive","title":"StyleAR: Customizing Multimodal Autoregressive Model for Style-Aligned Text-to-Image Generation","date":"2025-05-26","arxiv_id":"2505.19874","repositories_listed":0,"syntology":null},{"url":null,"slug":"textdiffuser-rl-efficient-and-robust-text","title":"TextDiffuser-RL: Efficient and Robust Text Layout Optimization for High-Fidelity Text-to-Image Synthesis","date":"2025-05-25","arxiv_id":"2505.19291","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-free-stylized-text-to-image","title":"Training-free Stylized Text-to-Image Generation with Fast Inference","date":"2025-05-25","arxiv_id":"2505.19063","repositories_listed":0,"syntology":null},{"url":null,"slug":"mod-adapter-tuning-free-and-versatile-multi","title":"Mod-Adapter: Tuning-Free and Versatile Multi-concept Personalization via Modulation Adapter","date":"2025-05-24","arxiv_id":"2505.18612","repositories_listed":0,"syntology":null},{"url":null,"slug":"tng-clip-training-time-negation-data","title":"TNG-CLIP:Training-Time Negation Data Generation for Negation Awareness of CLIP","date":"2025-05-24","arxiv_id":"2505.18434","repositories_listed":0,"syntology":null},{"url":null,"slug":"conditional-panoramic-image-generation-via","title":"Conditional Panoramic Image Generation via Masked Autoregressive Modeling","date":"2025-05-22","arxiv_id":"2505.16862","repositories_listed":0,"syntology":null},{"url":null,"slug":"creatively-upscaling-images-with-global","title":"Creatively Upscaling Images with Global-Regional Priors","date":"2025-05-22","arxiv_id":"2505.16976","repositories_listed":0,"syntology":null},{"url":null,"slug":"ntire-2025-challenge-on-text-to-image","title":"NTIRE 2025 challenge on Text to Image Generation Model Quality Assessment","date":"2025-05-22","arxiv_id":"2505.16314","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-rewarding-large-vision-language-models","title":"Self-Rewarding Large Vision-Language Models for Optimizing Prompts in Text-to-Image Generation","date":"2025-05-22","arxiv_id":"2505.16763","repositories_listed":0,"syntology":null},{"url":null,"slug":"harnessing-caption-detailness-for-data","title":"Harnessing Caption Detailness for Data-Efficient Text-to-Image Generation","date":"2025-05-21","arxiv_id":"2505.15172","repositories_listed":0,"syntology":null},{"url":null,"slug":"ia-t2i-internet-augmented-text-to-image","title":"IA-T2I: Internet-Augmented Text-to-Image Generation","date":"2025-05-21","arxiv_id":"2505.15779","repositories_listed":0,"syntology":null},{"url":null,"slug":"hunyuan-game-industrial-grade-intelligent","title":"Hunyuan-Game: Industrial-grade Intelligent Game Creation Model","date":"2025-05-20","arxiv_id":"2505.14135","repositories_listed":0,"syntology":null},{"url":null,"slug":"diff-mm-exploring-pre-trained-text-to-image","title":"Diff-MM: Exploring Pre-trained Text-to-Image Generation Model for Unified Multi-modal Object Tracking","date":"2025-05-19","arxiv_id":"2505.12606","repositories_listed":0,"syntology":null},{"url":null,"slug":"guiding-diffusion-with-deep-geometric-moments","title":"Guiding Diffusion with Deep Geometric Moments: Balancing Fidelity and Variation","date":"2025-05-18","arxiv_id":"2505.12486","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11178","title":"CompAlign: Improving Compositional Text-to-Image Generation with a Complex Benchmark and Fine-Grained Feedback","date":"2025-05-16","arxiv_id":"2505.11178","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11468","title":"PSDiffusion: Harmonized Multi-Layer Image Generation via Layout and Appearance Alignment","date":"2025-05-16","arxiv_id":"2505.11468","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10743","title":"IMAGE-ALCHEMY: Advancing subject fidelity in personalised text-to-image generation","date":"2025-05-15","arxiv_id":"2505.10743","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-initial-exploration-of-default-images-in","title":"An Initial Exploration of Default Images in Text-to-Image Generation","date":"2025-05-14","arxiv_id":"2505.09166","repositories_listed":0,"syntology":null},{"url":null,"slug":"don-t-forget-your-inverse-ddim-for-image","title":"Don't Forget your Inverse DDIM for Image Editing","date":"2025-05-14","arxiv_id":"2505.09571","repositories_listed":0,"syntology":null},{"url":null,"slug":"replay-based-continual-learning-with-dual","title":"Replay-Based Continual Learning with Dual-Layered Distillation and a Streamlined U-Net for Efficient Text-to-Image Generation","date":"2025-05-11","arxiv_id":"2505.06995","repositories_listed":0,"syntology":null}],"record_sha256":"951864de11bf7b2bc72dd981975245221631165faf46458e79f2c883ca945634","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}