{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-image-generation/papers/5","list_of":"/task/text-to-image-generation","task":"Text-to-Image Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":11,"rows_per_page":100,"rows":[401,500],"of":1085,"counts":{"archive_papers_tagged":1085,"with_a_code_link":546,"where_syntology_ran_a_sample":246,"not_listed_spam_title":0,"listed":1085,"listed_where_code_ran":246,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":215,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":215,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-image-generation","prev":"/task/text-to-image-generation/papers/4","next":"/task/text-to-image-generation/papers/6","papers":[{"url":"/paper/conditional-diffusion-distillation","slug":"conditional-diffusion-distillation","title":"CoDi: Conditional Diffusion Distillation for Higher-Fidelity and Faster Image Generation","date":"2023-10-02","arxiv_id":"2310.01407","repositories_listed":1,"syntology":null},{"url":"/paper/instructcv-instruction-tuned-text-to-image","slug":"instructcv-instruction-tuned-text-to-image","title":"InstructCV: Instruction-Tuned Text-to-Image Diffusion Models as Vision Generalists","date":"2023-09-30","arxiv_id":"2310.00390","repositories_listed":1,"syntology":{"n":12,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":12,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/instructcv-instruction-tuned-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2310.00390","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00390"}},"official":{"repos":["AlaaLab/InstructCV"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/ccedit-creative-and-controllable-video","slug":"ccedit-creative-and-controllable-video","title":"CCEdit: Creative and Controllable Video Editing via Diffusion Models","date":"2023-09-28","arxiv_id":"2309.16496","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-foundation-models-from-specialists","slug":"multimodal-foundation-models-from-specialists","title":"Multimodal Foundation Models: From Specialists to General-Purpose Assistants","date":"2023-09-18","arxiv_id":"2309.10020","repositories_listed":1,"syntology":null},{"url":"/paper/progressive-text-to-image-diffusion-with-soft","slug":"progressive-text-to-image-diffusion-with-soft","title":"Progressive Text-to-Image Diffusion with Soft Latent Direction","date":"2023-09-18","arxiv_id":"2309.09466","repositories_listed":1,"syntology":null},{"url":"/paper/viewpoint-textual-inversion-unleashing-novel","slug":"viewpoint-textual-inversion-unleashing-novel","title":"Viewpoint Textual Inversion: Discovering Scene Representations and 3D View Control in 2D Diffusion Models","date":"2023-09-14","arxiv_id":"2309.07986","repositories_listed":1,"syntology":null},{"url":"/paper/language-models-as-black-box-optimizers-for","slug":"language-models-as-black-box-optimizers-for","title":"Language Models as Black-Box Optimizers for Vision-Language Models","date":"2023-09-12","arxiv_id":"2309.05950","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/language-models-as-black-box-optimizers-for#ran","syntology_url":"https://syntology.ai/paper/2309.05950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05950"}},"official":{"repos":["shihongl1998/llm-as-a-blackbox-optimizer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/prompting4debugging-red-teaming-text-to-image","slug":"prompting4debugging-red-teaming-text-to-image","title":"Prompting4Debugging: Red-Teaming Text-to-Image Diffusion Models by Finding Problematic Prompts","date":"2023-09-12","arxiv_id":"2309.06135","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prompting4debugging-red-teaming-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2309.06135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06135"}},"official":{"repos":["joycenerd/p4d"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/iti-gen-inclusive-text-to-image-generation","slug":"iti-gen-inclusive-text-to-image-generation","title":"ITI-GEN: Inclusive Text-to-Image Generation","date":"2023-09-11","arxiv_id":"2309.05569","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/iti-gen-inclusive-text-to-image-generation#ran","syntology_url":"https://syntology.ai/paper/2309.05569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05569"}},"official":{"repos":["humansensinglab/ITI-GEN"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/photoverse-tuning-free-image-customization","slug":"photoverse-tuning-free-image-customization","title":"PhotoVerse: Tuning-Free Image Customization with Text-to-Image Diffusion Models","date":"2023-09-11","arxiv_id":"2309.05793","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/photoverse-tuning-free-image-customization#ran","syntology_url":"https://syntology.ai/paper/2309.05793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05793"}},"official":null}},{"url":"/paper/from-text-to-mask-localizing-entities-using","slug":"from-text-to-mask-localizing-entities-using","title":"From Text to Mask: Localizing Entities Using the Attention of Text-to-Image Diffusion Models","date":"2023-09-08","arxiv_id":"2309.04109","repositories_listed":1,"syntology":null},{"url":"/paper/diffusion-model-is-secretly-a-training-free","slug":"diffusion-model-is-secretly-a-training-free","title":"Diffusion Model is Secretly a Training-free Open Vocabulary Semantic Segmenter","date":"2023-09-06","arxiv_id":"2309.02773","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffusion-model-is-secretly-a-training-free#ran","syntology_url":"https://syntology.ai/paper/2309.02773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.02773"}},"official":{"repos":["VCG-team/DiffSegmenter"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exchanging-based-multimodal-fusion-with","slug":"exchanging-based-multimodal-fusion-with","title":"Exchanging-based Multimodal Fusion with Transformer","date":"2023-09-05","arxiv_id":"2309.02190","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exchanging-based-multimodal-fusion-with#ran","syntology_url":"https://syntology.ai/paper/2309.02190","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.02190"}},"official":{"repos":["recklessronan/muse"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-autoregressive-multi-modal-models","slug":"scaling-autoregressive-multi-modal-models","title":"Scaling Autoregressive Multi-Modal Models: Pretraining and Instruction Tuning","date":"2023-09-05","arxiv_id":"2309.02591","repositories_listed":1,"syntology":null},{"url":"/paper/pathldm-text-conditioned-latent-diffusion","slug":"pathldm-text-conditioned-latent-diffusion","title":"PathLDM: Text conditioned Latent Diffusion Model for Histopathology","date":"2023-09-01","arxiv_id":"2309.00748","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":5,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":3,"n_no_contract":2,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 3 violated, 2 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pathldm-text-conditioned-latent-diffusion#ran","syntology_url":"https://syntology.ai/paper/2309.00748","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.00748"}},"official":{"repos":["cvlab-stonybrook/pathldm"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dense-text-to-image-generation-with-attention","slug":"dense-text-to-image-generation-with-attention","title":"Dense Text-to-Image Generation with Attention Modulation","date":"2023-08-24","arxiv_id":"2308.12964","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dense-text-to-image-generation-with-attention#ran","syntology_url":"https://syntology.ai/paper/2308.12964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12964"}},"official":{"repos":["naver-ai/densediffusion"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/aspire-language-guided-augmentation-for","slug":"aspire-language-guided-augmentation-for","title":"ASPIRE: Language-Guided Data Augmentation for Improving Robustness Against Spurious Correlations","date":"2023-08-19","arxiv_id":"2308.10103","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-generate-semantic-layouts-for","slug":"learning-to-generate-semantic-layouts-for","title":"Learning to Generate Semantic Layouts for Higher Text-Image Correspondence in Text-to-Image Synthesis","date":"2023-08-16","arxiv_id":"2308.08157","repositories_listed":1,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":2,"n_honours":4,"n_violates":3,"n_no_contract":7,"n_pointer_only":17,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 4 honoured, 3 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-generate-semantic-layouts-for#ran","syntology_url":"https://syntology.ai/paper/2308.08157","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08157"}},"official":{"repos":["pmh9960/GCDP"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/likelihood-based-text-to-image-evaluation","slug":"likelihood-based-text-to-image-evaluation","title":"Likelihood-Based Text-to-Image Evaluation with Patch-Level Perceptual and Semantic Credit Assignment","date":"2023-08-16","arxiv_id":"2308.08525","repositories_listed":1,"syntology":null},{"url":"/paper/story-visualization-by-online-text","slug":"story-visualization-by-online-text","title":"Story Visualization by Online Text Augmentation with Context Memory","date":"2023-08-15","arxiv_id":"2308.07575","repositories_listed":1,"syntology":null},{"url":"/paper/masked-attention-diffusion-guidance-for","slug":"masked-attention-diffusion-guidance-for","title":"Masked-Attention Diffusion Guidance for Spatially Controlling Text-to-Image Generation","date":"2023-08-11","arxiv_id":"2308.06027","repositories_listed":1,"syntology":null},{"url":"/paper/layoutllm-t2i-eliciting-layout-guidance-from","slug":"layoutllm-t2i-eliciting-layout-guidance-from","title":"LayoutLLM-T2I: Eliciting Layout Guidance from LLM for Text-to-Image Generation","date":"2023-08-09","arxiv_id":"2308.05095","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":4,"n_ran_checked":9,"n_instrument":2,"n_unverified":6,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":17,"phrase":"11 ran (of which 4 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/layoutllm-t2i-eliciting-layout-guidance-from#ran","syntology_url":"https://syntology.ai/paper/2308.05095","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.05095"}},"official":null}},{"url":"/paper/promptpaint-steering-text-to-image-generation","slug":"promptpaint-steering-text-to-image-generation","title":"PromptPaint: Steering Text-to-Image Generation Through Paint Medium-like Interactions","date":"2023-08-09","arxiv_id":"2308.05184","repositories_listed":1,"syntology":null},{"url":"/paper/conceptlab-creative-generation-using","slug":"conceptlab-creative-generation-using","title":"ConceptLab: Creative Concept Generation using VLM-Guided Diffusion Prior Constraints","date":"2023-08-03","arxiv_id":"2308.02669","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conceptlab-creative-generation-using#ran","syntology_url":"https://syntology.ai/paper/2308.02669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.02669"}},"official":{"repos":["kfirgoldberg/ConceptLab"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reverse-stable-diffusion-what-prompt-was-used","slug":"reverse-stable-diffusion-what-prompt-was-used","title":"Reverse Stable Diffusion: What prompt was used to generate this image?","date":"2023-08-02","arxiv_id":"2308.01472","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reverse-stable-diffusion-what-prompt-was-used#ran","syntology_url":"https://syntology.ai/paper/2308.01472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.01472"}},"official":{"repos":["croitorualin/reverse-stable-diffusion"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-bias-amplification-paradox-in-text-to","slug":"the-bias-amplification-paradox-in-text-to","title":"The Bias Amplification Paradox in Text-to-Image Generation","date":"2023-08-01","arxiv_id":"2308.00755","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-bias-amplification-paradox-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2308.00755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.00755"}},"official":{"repos":["preethiseshadri518/bias-amplification-paradox"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bagm-a-backdoor-attack-for-manipulating-text","slug":"bagm-a-backdoor-attack-for-manipulating-text","title":"BAGM: A Backdoor Attack for Manipulating Text-to-Image Generative Models","date":"2023-07-31","arxiv_id":"2307.16489","repositories_listed":1,"syntology":null},{"url":"/paper/learning-disentangled-discrete","slug":"learning-disentangled-discrete","title":"Learning Disentangled Discrete Representations","date":"2023-07-26","arxiv_id":"2307.14151","repositories_listed":1,"syntology":null},{"url":"/paper/subject-diffusion-open-domain-personalized","slug":"subject-diffusion-open-domain-personalized","title":"Subject-Diffusion:Open Domain Personalized Text-to-Image Generation without Test-time Fine-tuning","date":"2023-07-21","arxiv_id":"2307.11410","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/subject-diffusion-open-domain-personalized#ran","syntology_url":"https://syntology.ai/paper/2307.11410","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11410"}},"official":{"repos":["OPPO-Mente-Lab/Subject-Diffusion"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-sampling-with-momentum-for","slug":"diffusion-sampling-with-momentum-for","title":"Diffusion Sampling with Momentum for Mitigating Divergence Artifacts","date":"2023-07-20","arxiv_id":"2307.11118","repositories_listed":1,"syntology":null},{"url":"/paper/divide-bind-your-attention-for-improved","slug":"divide-bind-your-attention-for-improved","title":"Divide & Bind Your Attention for Improved Generative Semantic Nursing","date":"2023-07-20","arxiv_id":"2307.10864","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/divide-bind-your-attention-for-improved#ran","syntology_url":"https://syntology.ai/paper/2307.10864","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.10864"}},"official":{"repos":["boschresearch/Divide-and-Bind"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/planting-a-seed-of-vision-in-large-language","slug":"planting-a-seed-of-vision-in-large-language","title":"Planting a SEED of Vision in Large Language Model","date":"2023-07-16","arxiv_id":"2307.08041","repositories_listed":1,"syntology":null},{"url":"/paper/t2i-compbench-a-comprehensive-benchmark-for-1","slug":"t2i-compbench-a-comprehensive-benchmark-for-1","title":"T2I-CompBench: A Comprehensive Benchmark for Open-world Compositional Text-to-image Generation","date":"2023-07-12","arxiv_id":"2307.06350","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/t2i-compbench-a-comprehensive-benchmark-for-1#ran","syntology_url":"https://syntology.ai/paper/2307.06350","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.06350"}},"official":null}},{"url":"/paper/tiam-a-metric-for-evaluating-alignment-in","slug":"tiam-a-metric-for-evaluating-alignment-in","title":"TIAM -- A Metric for Evaluating Alignment in Text-to-Image Generation","date":"2023-07-11","arxiv_id":"2307.05134","repositories_listed":1,"syntology":null},{"url":"/paper/exact-diffusion-inversion-via-bi-directional","slug":"exact-diffusion-inversion-via-bi-directional","title":"Exact Diffusion Inversion via Bi-directional Integration Approximation","date":"2023-07-10","arxiv_id":"2307.10829","repositories_listed":1,"syntology":{"n":16,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":16,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/exact-diffusion-inversion-via-bi-directional#ran","syntology_url":"https://syntology.ai/paper/2307.10829","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.10829"}},"official":{"repos":["guoqiang-zhang-x/BDIA"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/aigciqa2023-a-large-scale-image-quality","slug":"aigciqa2023-a-large-scale-image-quality","title":"AIGCIQA2023: A Large-scale Image Quality Assessment Database for AI Generated Images: from the Perspectives of Quality, Authenticity and Correspondence","date":"2023-07-01","arxiv_id":"2307.00211","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-multimodal-representation","slug":"semi-supervised-multimodal-representation","title":"Semi-supervised Multimodal Representation Learning through a Global Workspace","date":"2023-06-27","arxiv_id":"2306.15711","repositories_listed":1,"syntology":null},{"url":"/paper/norm-guided-latent-space-exploration-for-text-1","slug":"norm-guided-latent-space-exploration-for-text-1","title":"Norm-guided latent space exploration for text-to-image generation","date":"2023-06-14","arxiv_id":"2306.08687","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/norm-guided-latent-space-exploration-for-text-1#ran","syntology_url":"https://syntology.ai/paper/2306.08687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08687"}},"official":{"repos":["dvirsamuel/SeedSelect"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ai-generated-image-detection-using-a-cross","slug":"ai-generated-image-detection-using-a-cross","title":"AI-Generated Image Detection using a Cross-Attention Enhanced Dual-Stream Network","date":"2023-06-12","arxiv_id":"2306.07005","repositories_listed":1,"syntology":null},{"url":"/paper/rewarded-soups-towards-pareto-optimal-1","slug":"rewarded-soups-towards-pareto-optimal-1","title":"Rewarded soups: towards Pareto-optimal alignment by interpolating weights fine-tuned on diverse rewards","date":"2023-06-07","arxiv_id":"2306.04488","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-difference-of-bert-style-and-clip","slug":"on-the-difference-of-bert-style-and-clip","title":"On the Difference of BERT-style and CLIP-style Text Encoders","date":"2023-06-06","arxiv_id":"2306.03678","repositories_listed":1,"syntology":null},{"url":"/paper/composition-and-deformance-measuring","slug":"composition-and-deformance-measuring","title":"Composition and Deformance: Measuring Imageability with a Text-to-Image Model","date":"2023-06-05","arxiv_id":"2306.03168","repositories_listed":1,"syntology":null},{"url":"/paper/detector-guidance-for-multi-object-text-to","slug":"detector-guidance-for-multi-object-text-to","title":"Detector Guidance for Multi-Object Text-to-Image Generation","date":"2023-06-04","arxiv_id":"2306.02236","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-grimm-open-ended-visual","slug":"intelligent-grimm-open-ended-visual","title":"Intelligent Grimm -- Open-ended Visual Storytelling via Latent Diffusion Models","date":"2023-06-01","arxiv_id":"2306.00973","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/intelligent-grimm-open-ended-visual#ran","syntology_url":"https://syntology.ai/paper/2306.00973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00973"}},"official":{"repos":["haoningwu3639/StoryGen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/vico-detail-preserving-visual-condition-for","slug":"vico-detail-preserving-visual-condition-for","title":"ViCo: Plug-and-play Visual Condition for Personalized Text-to-image Generation","date":"2023-06-01","arxiv_id":"2306.00971","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":9,"n_instrument":6,"n_unverified":0,"n_honours":2,"n_violates":3,"n_no_contract":4,"n_pointer_only":3,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 3 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vico-detail-preserving-visual-condition-for#ran","syntology_url":"https://syntology.ai/paper/2306.00971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00971"}},"official":{"repos":["haoosz/vico"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/nested-diffusion-processes-for-anytime-image","slug":"nested-diffusion-processes-for-anytime-image","title":"Nested Diffusion Processes for Anytime Image Generation","date":"2023-05-30","arxiv_id":"2305.19066","repositories_listed":1,"syntology":null},{"url":"/paper/raphael-text-to-image-generation-via-large","slug":"raphael-text-to-image-generation-via-large","title":"RAPHAEL: Text-to-Image Generation via Large Mixture of Diffusion Paths","date":"2023-05-29","arxiv_id":"2305.18295","repositories_listed":1,"syntology":null},{"url":"/paper/talecrafter-interactive-story-visualization","slug":"talecrafter-interactive-story-visualization","title":"TaleCrafter: Interactive Story Visualization with Multiple Characters","date":"2023-05-29","arxiv_id":"2305.18247","repositories_listed":1,"syntology":null},{"url":"/paper/generating-images-with-multimodal-language","slug":"generating-images-with-multimodal-language","title":"Generating Images with Multimodal Language Models","date":"2023-05-26","arxiv_id":"2305.17216","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/generating-images-with-multimodal-language#ran","syntology_url":"https://syntology.ai/paper/2305.17216","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17216"}},"official":{"repos":["kohjingyu/gill"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/blip-diffusion-pre-trained-subject-1","slug":"blip-diffusion-pre-trained-subject-1","title":"BLIP-Diffusion: Pre-trained Subject Representation for Controllable Text-to-Image Generation and Editing","date":"2023-05-24","arxiv_id":"2305.14720","repositories_listed":1,"syntology":null},{"url":"/paper/diffblender-scalable-and-composable","slug":"diffblender-scalable-and-composable","title":"DiffBlender: Scalable and Composable Multimodal Text-to-Image Diffusion Models","date":"2023-05-24","arxiv_id":"2305.15194","repositories_listed":1,"syntology":null},{"url":"/paper/layoutgpt-compositional-visual-planning-and","slug":"layoutgpt-compositional-visual-planning-and","title":"LayoutGPT: Compositional Visual Planning and Generation with Large Language Models","date":"2023-05-24","arxiv_id":"2305.15393","repositories_listed":1,"syntology":{"n":19,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/layoutgpt-compositional-visual-planning-and#ran","syntology_url":"https://syntology.ai/paper/2305.15393","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15393"}},"official":{"repos":["weixi-feng/layoutgpt"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-applications-a-survey","slug":"vision-language-applications-a-survey","title":"Vision + Language Applications: A Survey","date":"2023-05-24","arxiv_id":"2305.14598","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-detail-preservation-for-customized","slug":"enhancing-detail-preservation-for-customized","title":"Enhancing Detail Preservation for Customized Text-to-Image Generation: A Regularization-Free Approach","date":"2023-05-23","arxiv_id":"2305.13579","repositories_listed":1,"syntology":null},{"url":"/paper/if-at-first-you-don-t-succeed-try-try-again","slug":"if-at-first-you-don-t-succeed-try-try-again","title":"If at First You Don't Succeed, Try, Try Again: Faithful Diffusion-based Text-to-Image Generation by Selection","date":"2023-05-22","arxiv_id":"2305.13308","repositories_listed":1,"syntology":null},{"url":"/paper/interactive-data-synthesis-for-systematic","slug":"interactive-data-synthesis-for-systematic","title":"Interactive Data Synthesis for Systematic Vision Adaptation via LLMs-AIGCs Collaboration","date":"2023-05-22","arxiv_id":"2305.12799","repositories_listed":1,"syntology":null},{"url":"/paper/late-constraint-diffusion-guidance-for","slug":"late-constraint-diffusion-guidance-for","title":"LaCon: Late-Constraint Diffusion for Steerable Guided Image Synthesis","date":"2023-05-19","arxiv_id":"2305.11520","repositories_listed":1,"syntology":null},{"url":"/paper/discriminative-diffusion-models-as-few-shot","slug":"discriminative-diffusion-models-as-few-shot","title":"Discffusion: Discriminative Diffusion Models as Few-shot Vision and Language Learners","date":"2023-05-18","arxiv_id":"2305.10722","repositories_listed":1,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":1,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/discriminative-diffusion-models-as-few-shot#ran","syntology_url":"https://syntology.ai/paper/2305.10722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10722"}},"official":{"repos":["eric-ai-lab/dsd"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/instruct2act-mapping-multi-modality","slug":"instruct2act-mapping-multi-modality","title":"Instruct2Act: Mapping Multi-modality Instructions to Robotic Actions with Large Language Model","date":"2023-05-18","arxiv_id":"2305.11176","repositories_listed":1,"syntology":null},{"url":"/paper/videofactory-swap-attention-in-spatiotemporal","slug":"videofactory-swap-attention-in-spatiotemporal","title":"Swap Attention in Spatiotemporal Diffusions for Text-to-Video Generation","date":"2023-05-18","arxiv_id":"2305.10874","repositories_listed":1,"syntology":null},{"url":"/paper/x-iqe-explainable-image-quality-evaluation","slug":"x-iqe-explainable-image-quality-evaluation","title":"X-IQE: eXplainable Image Quality Evaluation for Text-to-Image Generation with Visual Large Language Models","date":"2023-05-18","arxiv_id":"2305.10843","repositories_listed":1,"syntology":null},{"url":"/paper/fastcomposer-tuning-free-multi-subject-image","slug":"fastcomposer-tuning-free-multi-subject-image","title":"FastComposer: Tuning-Free Multi-Subject Image Generation with Localized Attention","date":"2023-05-17","arxiv_id":"2305.10431","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":7,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/fastcomposer-tuning-free-multi-subject-image#ran","syntology_url":"https://syntology.ai/paper/2305.10431","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10431"}},"official":{"repos":["mit-han-lab/fastcomposer"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/what-you-see-is-what-you-read-improving-text-1","slug":"what-you-see-is-what-you-read-improving-text-1","title":"What You See is What You Read? Improving Text-Image Alignment Evaluation","date":"2023-05-17","arxiv_id":"2305.10400","repositories_listed":1,"syntology":null},{"url":"/paper/sur-adapter-enhancing-text-to-image-pre","slug":"sur-adapter-enhancing-text-to-image-pre","title":"SUR-adapter: Enhancing Text-to-Image Pre-trained Diffusion Models with Large Language Models","date":"2023-05-09","arxiv_id":"2305.05189","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sur-adapter-enhancing-text-to-image-pre#ran","syntology_url":"https://syntology.ai/paper/2305.05189","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.05189"}},"official":{"repos":["Qrange-group/SUR-adapter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/data-curation-for-image-captioning-with-text","slug":"data-curation-for-image-captioning-with-text","title":"The Role of Data Curation in Image Captioning","date":"2023-05-05","arxiv_id":"2305.03610","repositories_listed":1,"syntology":null},{"url":"/paper/disenbooth-disentangled-parameter-efficient","slug":"disenbooth-disentangled-parameter-efficient","title":"DisenBooth: Identity-Preserving Disentangled Tuning for Subject-Driven Text-to-Image Generation","date":"2023-05-05","arxiv_id":"2305.03374","repositories_listed":1,"syntology":null},{"url":"/paper/personalize-segment-anything-model-with-one","slug":"personalize-segment-anything-model-with-one","title":"Personalize Segment Anything Model with One Shot","date":"2023-05-04","arxiv_id":"2305.03048","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-procedural-planning-via-dual-text","slug":"multimodal-procedural-planning-via-dual-text","title":"Multimodal Procedural Planning via Dual Text-Image Prompting","date":"2023-05-02","arxiv_id":"2305.01795","repositories_listed":1,"syntology":null},{"url":"/paper/pick-a-pic-an-open-dataset-of-user","slug":"pick-a-pic-an-open-dataset-of-user","title":"Pick-a-Pic: An Open Dataset of User Preferences for Text-to-Image Generation","date":"2023-05-02","arxiv_id":"2305.01569","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pick-a-pic-an-open-dataset-of-user#ran","syntology_url":"https://syntology.ai/paper/2305.01569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.01569"}},"official":{"repos":["yuvalkirstain/pickscore"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/it-is-all-about-where-you-start-text-to-image","slug":"it-is-all-about-where-you-start-text-to-image","title":"Generating images of rare concepts using pre-trained diffusion models","date":"2023-04-27","arxiv_id":"2304.14530","repositories_listed":1,"syntology":null},{"url":"/paper/upgpt-universal-diffusion-model-for-person","slug":"upgpt-universal-diffusion-model-for-person","title":"UPGPT: Universal Diffusion Model for Person Image Generation, Editing and Pose Transfer","date":"2023-04-18","arxiv_id":"2304.08870","repositories_listed":1,"syntology":null},{"url":"/paper/expressive-text-to-image-generation-with-rich","slug":"expressive-text-to-image-generation-with-rich","title":"Expressive Text-to-Image Generation with Rich Text","date":"2023-04-13","arxiv_id":"2304.06720","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/expressive-text-to-image-generation-with-rich#ran","syntology_url":"https://syntology.ai/paper/2304.06720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.06720"}},"official":{"repos":["songweige/rich-text-to-image"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/controllable-textual-inversion-for","slug":"controllable-textual-inversion-for","title":"Controllable Textual Inversion for Personalized Text-to-Image Generation","date":"2023-04-11","arxiv_id":"2304.05265","repositories_listed":1,"syntology":null},{"url":"/paper/hrs-bench-holistic-reliable-and-scalable","slug":"hrs-bench-holistic-reliable-and-scalable","title":"HRS-Bench: Holistic, Reliable and Scalable Benchmark for Text-to-Image Models","date":"2023-04-11","arxiv_id":"2304.05390","repositories_listed":1,"syntology":null},{"url":"/paper/uncurated-image-text-datasets-shedding-light","slug":"uncurated-image-text-datasets-shedding-light","title":"Uncurated Image-Text Datasets: Shedding Light on Demographic Bias","date":"2023-04-06","arxiv_id":"2304.02828","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/uncurated-image-text-datasets-shedding-light#ran","syntology_url":"https://syntology.ai/paper/2304.02828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.02828"}},"official":{"repos":["noagarcia/phase"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/forget-me-not-learning-to-forget-in-text-to","slug":"forget-me-not-learning-to-forget-in-text-to","title":"Forget-Me-Not: Learning to Forget in Text-to-Image Diffusion Models","date":"2023-03-30","arxiv_id":"2303.17591","repositories_listed":1,"syntology":null},{"url":"/paper/indonesian-text-to-image-synthesis-with","slug":"indonesian-text-to-image-synthesis-with","title":"Indonesian Text-to-Image Synthesis with Sentence-BERT and FastGAN","date":"2023-03-25","arxiv_id":"2303.14517","repositories_listed":1,"syntology":null},{"url":"/paper/medical-diffusion-on-a-budget-textual","slug":"medical-diffusion-on-a-budget-textual","title":"Medical diffusion on a budget: Textual Inversion for medical image generation","date":"2023-03-23","arxiv_id":"2303.13430","repositories_listed":1,"syntology":null},{"url":"/paper/magvlt-masked-generative-vision-and-language","slug":"magvlt-masked-generative-vision-and-language","title":"MAGVLT: Masked Generative Vision-and-Language Transformer","date":"2023-03-21","arxiv_id":"2303.12208","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":5,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/magvlt-masked-generative-vision-and-language#ran","syntology_url":"https://syntology.ai/paper/2303.12208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.12208"}},"official":{"repos":["kakaobrain/magvlt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":5,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/tifa-accurate-and-interpretable-text-to-image","slug":"tifa-accurate-and-interpretable-text-to-image","title":"TIFA: Accurate and Interpretable Text-to-Image Faithfulness Evaluation with Question Answering","date":"2023-03-21","arxiv_id":"2303.11897","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tifa-accurate-and-interpretable-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2303.11897","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11897"}},"official":{"repos":["Yushi-Hu/tifa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/localizing-object-level-shape-variations-with","slug":"localizing-object-level-shape-variations-with","title":"Localizing Object-level Shape Variations with Text-to-Image Diffusion Models","date":"2023-03-20","arxiv_id":"2303.11306","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/localizing-object-level-shape-variations-with#ran","syntology_url":"https://syntology.ai/paper/2303.11306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11306"}},"official":null}},{"url":"/paper/svdiff-compact-parameter-space-for-diffusion","slug":"svdiff-compact-parameter-space-for-diffusion","title":"SVDiff: Compact Parameter Space for Diffusion Fine-Tuning","date":"2023-03-20","arxiv_id":"2303.11305","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/svdiff-compact-parameter-space-for-diffusion#ran","syntology_url":"https://syntology.ai/paper/2303.11305","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11305"}},"official":null}},{"url":"/paper/p-extended-textual-conditioning-in-text-to","slug":"p-extended-textual-conditioning-in-text-to","title":"P+: Extended Textual Conditioning in Text-to-Image Generation","date":"2023-03-16","arxiv_id":"2303.09522","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/p-extended-textual-conditioning-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2303.09522","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.09522"}},"official":null}},{"url":"/paper/scaling-up-gans-for-text-to-image-synthesis","slug":"scaling-up-gans-for-text-to-image-synthesis","title":"Scaling up GANs for Text-to-Image Synthesis","date":"2023-03-09","arxiv_id":"2303.05511","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":2,"n_honours":1,"n_violates":3,"n_no_contract":5,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 3 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/scaling-up-gans-for-text-to-image-synthesis#ran","syntology_url":"https://syntology.ai/paper/2303.05511","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.05511"}},"official":null}},{"url":"/paper/a-prompt-log-analysis-of-text-to-image","slug":"a-prompt-log-analysis-of-text-to-image","title":"A Prompt Log Analysis of Text-to-Image Generation Systems","date":"2023-03-08","arxiv_id":"2303.04587","repositories_listed":1,"syntology":null},{"url":"/paper/teaching-clip-to-count-to-ten","slug":"teaching-clip-to-count-to-ten","title":"Teaching CLIP to Count to Ten","date":"2023-02-23","arxiv_id":"2302.12066","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/teaching-clip-to-count-to-ten#ran","syntology_url":"https://syntology.ai/paper/2302.12066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12066"}},"official":null}},{"url":"/paper/prompt-stealing-attacks-against-text-to-image","slug":"prompt-stealing-attacks-against-text-to-image","title":"Prompt Stealing Attacks Against Text-to-Image Generation Models","date":"2023-02-20","arxiv_id":"2302.09923","repositories_listed":1,"syntology":null},{"url":"/paper/is-this-loss-informative-faster-text-to-image-1","slug":"is-this-loss-informative-faster-text-to-image-1","title":"Is This Loss Informative? Faster Text-to-Image Customization by Tracking Objective Dynamics","date":"2023-02-09","arxiv_id":"2302.04841","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/is-this-loss-informative-faster-text-to-image-1#ran","syntology_url":"https://syntology.ai/paper/2302.04841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.04841"}},"official":{"repos":["yandex-research/dvar"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/boosting-zero-shot-classification-with","slug":"boosting-zero-shot-classification-with","title":"Diversity is Definitely Needed: Improving Model-Agnostic Zero-shot Classification via Stable Diffusion","date":"2023-02-07","arxiv_id":"2302.03298","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-zero-shot-classification-with#ran","syntology_url":"https://syntology.ai/paper/2302.03298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.03298"}},"official":{"repos":["jordan-hs/diversity_is_definitely_needed"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fair-diffusion-instructing-text-to-image","slug":"fair-diffusion-instructing-text-to-image","title":"Fair Diffusion: Instructing Text-to-Image Generation Models on Fairness","date":"2023-02-07","arxiv_id":"2302.10893","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fair-diffusion-instructing-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2302.10893","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.10893"}},"official":{"repos":["ml-research/fair-diffusion"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/eliminating-prior-bias-for-semantic-image","slug":"eliminating-prior-bias-for-semantic-image","title":"Eliminating Contextual Prior Bias for Semantic Image Editing via Dual-Cycle Diffusion","date":"2023-02-05","arxiv_id":"2302.02394","repositories_listed":1,"syntology":null},{"url":"/paper/accelerating-guided-diffusion-sampling-with","slug":"accelerating-guided-diffusion-sampling-with","title":"Accelerating Guided Diffusion Sampling with Splitting Numerical Methods","date":"2023-01-27","arxiv_id":"2301.11558","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accelerating-guided-diffusion-sampling-with#ran","syntology_url":"https://syntology.ai/paper/2301.11558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.11558"}},"official":{"repos":["swizad/split-diffusion"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/simple-diffusion-end-to-end-diffusion-for","slug":"simple-diffusion-end-to-end-diffusion-for","title":"Simple diffusion: End-to-end diffusion for high resolution images","date":"2023-01-26","arxiv_id":"2301.11093","repositories_listed":1,"syntology":null},{"url":"/paper/stylegan-t-unlocking-the-power-of-gans-for","slug":"stylegan-t-unlocking-the-power-of-gans-for","title":"StyleGAN-T: Unlocking the Power of GANs for Fast Large-Scale Text-to-Image Synthesis","date":"2023-01-23","arxiv_id":"2301.09515","repositories_listed":1,"syntology":null},{"url":"/paper/gligen-open-set-grounded-text-to-image","slug":"gligen-open-set-grounded-text-to-image","title":"GLIGEN: Open-Set Grounded Text-to-Image Generation","date":"2023-01-17","arxiv_id":"2301.07093","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/gligen-open-set-grounded-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2301.07093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.07093"}},"official":{"repos":["gligen/GLIGEN"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/riatig-reliable-and-imperceptible-adversarial","slug":"riatig-reliable-and-imperceptible-adversarial","title":"RIATIG: Reliable and Imperceptible Adversarial Text-to-Image Generation With Natural Prompts","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-spatial-relationships-in-text-to","slug":"benchmarking-spatial-relationships-in-text-to","title":"Benchmarking Spatial Relationships in Text-to-Image Generation","date":"2022-12-20","arxiv_id":"2212.10015","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-spatial-relationships-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2212.10015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.10015"}},"official":{"repos":["microsoft/VISOR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/are-multimodal-models-robust-to-image-and","slug":"are-multimodal-models-robust-to-image-and","title":"Benchmarking Robustness of Multimodal Image-Text Models under Distribution Shift","date":"2022-12-15","arxiv_id":"2212.08044","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-multimodal-models-robust-to-image-and#ran","syntology_url":"https://syntology.ai/paper/2212.08044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.08044"}},"official":null}},{"url":"/paper/smartbrush-text-and-shape-guided-object","slug":"smartbrush-text-and-shape-guided-object","title":"SmartBrush: Text and Shape Guided Object Inpainting with Diffusion Model","date":"2022-12-09","arxiv_id":"2212.05034","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/smartbrush-text-and-shape-guided-object#ran","syntology_url":"https://syntology.ai/paper/2212.05034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05034"}},"official":null}},{"url":"/paper/unite-and-conquer-cross-dataset-multimodal","slug":"unite-and-conquer-cross-dataset-multimodal","title":"Unite and Conquer: Plug & Play Multi-Modal Synthesis using Diffusion Models","date":"2022-12-01","arxiv_id":"2212.00793","repositories_listed":1,"syntology":null}],"record_sha256":"dc4ed0722a55b469cfbeb0cea02183987c2cdeb91c2990187e168a6fc17dd04f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}