{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-image-generation-1/papers/4","list_of":"/task/text-to-image-generation-1","task":"Text to Image Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":10,"rows_per_page":100,"rows":[301,400],"of":969,"counts":{"archive_papers_tagged":969,"with_a_code_link":461,"where_syntology_ran_a_sample":198,"not_listed_spam_title":0,"listed":969,"listed_where_code_ran":198,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":171,"every_run_a_failure_of_syntologys_instrument":27,"listed_with_a_run_with_no_instrument_failure":171,"listed_every_run_a_failure_of_syntologys_instrument":27,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-image-generation-1","prev":"/task/text-to-image-generation-1/papers/3","next":"/task/text-to-image-generation-1/papers/5","papers":[{"url":"/paper/mastering-text-to-image-diffusion","slug":"mastering-text-to-image-diffusion","title":"Mastering Text-to-Image Diffusion: Recaptioning, Planning, and Generating with Multimodal LLMs","date":"2024-01-22","arxiv_id":"2401.11708","repositories_listed":1,"syntology":{"n":24,"n_ran":21,"n_constructed":0,"n_ran_checked":12,"n_instrument":9,"n_unverified":3,"n_honours":2,"n_violates":3,"n_no_contract":7,"n_pointer_only":16,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 2 honoured, 3 violated, 7 with no contract checked; 9 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mastering-text-to-image-diffusion#ran","syntology_url":"https://syntology.ai/paper/2401.11708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.11708"}},"official":{"repos":["yangling0818/rpg-diffusionmaster"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/connect-collapse-corrupt-learning-cross-modal","slug":"connect-collapse-corrupt-learning-cross-modal","title":"Connect, Collapse, Corrupt: Learning Cross-Modal Tasks with Uni-Modal Data","date":"2024-01-16","arxiv_id":"2401.08567","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/connect-collapse-corrupt-learning-cross-modal#ran","syntology_url":"https://syntology.ai/paper/2401.08567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08567"}},"official":{"repos":["yuhui-zh15/c3"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-the-surface-a-global-scale-analysis-of","slug":"beyond-the-surface-a-global-scale-analysis-of","title":"ViSAGe: A Global-Scale Analysis of Visual Stereotypes in Text-to-Image Generation","date":"2024-01-12","arxiv_id":"2401.06310","repositories_listed":1,"syntology":null},{"url":"/paper/a-dataset-and-benchmark-for-copyright","slug":"a-dataset-and-benchmark-for-copyright","title":"A Dataset and Benchmark for Copyright Infringement Unlearning from Text-to-Image Diffusion Models","date":"2024-01-04","arxiv_id":"2403.12052","repositories_listed":1,"syntology":null},{"url":"/paper/amused-an-open-muse-reproduction","slug":"amused-an-open-muse-reproduction","title":"aMUSEd: An Open MUSE Reproduction","date":"2024-01-03","arxiv_id":"2401.01808","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-grimm-open-ended-visual-1","slug":"intelligent-grimm-open-ended-visual-1","title":"Intelligent Grimm - Open-ended Visual Storytelling via Latent Diffusion Models","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/cross-initialization-for-personalized-text-to","slug":"cross-initialization-for-personalized-text-to","title":"Cross Initialization for Personalized Text-to-Image Generation","date":"2023-12-26","arxiv_id":"2312.15905","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cross-initialization-for-personalized-text-to#ran","syntology_url":"https://syntology.ai/paper/2312.15905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15905"}},"official":{"repos":["lyupang/crossinitialization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/one-dimensional-adapter-to-rule-them-all","slug":"one-dimensional-adapter-to-rule-them-all","title":"One-Dimensional Adapter to Rule Them All: Concepts, Diffusion Models and Erasing Applications","date":"2023-12-26","arxiv_id":"2312.16145","repositories_listed":1,"syntology":null},{"url":"/paper/a-recipe-for-scaling-up-text-to-video","slug":"a-recipe-for-scaling-up-text-to-video","title":"A Recipe for Scaling up Text-to-Video Generation with Text-free Videos","date":"2023-12-25","arxiv_id":"2312.15770","repositories_listed":1,"syntology":null},{"url":"/paper/asymmetric-bias-in-text-to-image-generation","slug":"asymmetric-bias-in-text-to-image-generation","title":"Asymmetric Bias in Text-to-Image Generation with Adversarial Attacks","date":"2023-12-22","arxiv_id":"2312.14440","repositories_listed":1,"syntology":null},{"url":"/paper/brush-your-text-synthesize-any-scene-text-on","slug":"brush-your-text-synthesize-any-scene-text-on","title":"Brush Your Text: Synthesize Any Scene Text on Images via Diffusion Model","date":"2023-12-19","arxiv_id":"2312.12232","repositories_listed":1,"syntology":{"n":19,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":19,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/brush-your-text-synthesize-any-scene-text-on#ran","syntology_url":"https://syntology.ai/paper/2312.12232","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.12232"}},"official":{"repos":["ecnuljzhang/brush-your-text"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/decoupled-textual-embeddings-for-customized","slug":"decoupled-textual-embeddings-for-customized","title":"Decoupled Textual Embeddings for Customized Image Generation","date":"2023-12-19","arxiv_id":"2312.11826","repositories_listed":1,"syntology":null},{"url":"/paper/vl-gpt-a-generative-pre-trained-transformer","slug":"vl-gpt-a-generative-pre-trained-transformer","title":"VL-GPT: A Generative Pre-trained Transformer for Vision and Language Understanding and Generation","date":"2023-12-14","arxiv_id":"2312.09251","repositories_listed":1,"syntology":null},{"url":"/paper/adapedit-spatio-temporal-guided-adaptive","slug":"adapedit-spatio-temporal-guided-adaptive","title":"AdapEdit: Spatio-Temporal Guided Adaptive Editing Algorithm for Text-Based Continuity-Sensitive Image Editing","date":"2023-12-13","arxiv_id":"2312.08019","repositories_listed":1,"syntology":null},{"url":"/paper/clockwork-diffusion-efficient-generation-with","slug":"clockwork-diffusion-efficient-generation-with","title":"Clockwork Diffusion: Efficient Generation With Model-Step Distillation","date":"2023-12-13","arxiv_id":"2312.08128","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-driven-initial-image-construction","slug":"semantic-driven-initial-image-construction","title":"The Lottery Ticket Hypothesis in Denoising: Towards Semantic-Driven Initialization","date":"2023-12-13","arxiv_id":"2312.08872","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":7,"n_instrument":6,"n_unverified":1,"n_honours":2,"n_violates":3,"n_no_contract":2,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 3 violated, 2 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/semantic-driven-initial-image-construction#ran","syntology_url":"https://syntology.ai/paper/2312.08872","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08872"}},"official":{"repos":["UT-Mao/Initial-Noise-Construction"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/correcting-diffusion-generation-through","slug":"correcting-diffusion-generation-through","title":"Correcting Diffusion Generation through Resampling","date":"2023-12-10","arxiv_id":"2312.06038","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/correcting-diffusion-generation-through#ran","syntology_url":"https://syntology.ai/paper/2312.06038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06038"}},"official":{"repos":["ucsb-nlp-chang/diffusion_resampling"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generating-illustrated-instructions","slug":"generating-illustrated-instructions","title":"Generating Illustrated Instructions","date":"2023-12-07","arxiv_id":"2312.04552","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/generating-illustrated-instructions#ran","syntology_url":"https://syntology.ai/paper/2312.04552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.04552"}},"official":{"repos":["sachit-menon/generating-illustrated-instructions-reproduction"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/photomaker-customizing-realistic-human-photos","slug":"photomaker-customizing-realistic-human-photos","title":"PhotoMaker: Customizing Realistic Human Photos via Stacked ID Embedding","date":"2023-12-07","arxiv_id":"2312.04461","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/photomaker-customizing-realistic-human-photos#ran","syntology_url":"https://syntology.ai/paper/2312.04461","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.04461"}},"official":{"repos":["TencentARC/PhotoMaker"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kandinsky-3-0-technical-report","slug":"kandinsky-3-0-technical-report","title":"Kandinsky 3.0 Technical Report","date":"2023-12-06","arxiv_id":"2312.03511","repositories_listed":1,"syntology":{"n":14,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/kandinsky-3-0-technical-report#ran","syntology_url":"https://syntology.ai/paper/2312.03511","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03511"}},"official":{"repos":["ai-forever/kandinsky-3"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/tokencompose-grounding-diffusion-with-token","slug":"tokencompose-grounding-diffusion-with-token","title":"TokenCompose: Text-to-Image Diffusion with Token-level Supervision","date":"2023-12-06","arxiv_id":"2312.03626","repositories_listed":1,"syntology":null},{"url":"/paper/customization-assistant-for-text-to-image","slug":"customization-assistant-for-text-to-image","title":"Customization Assistant for Text-to-image Generation","date":"2023-12-05","arxiv_id":"2312.03045","repositories_listed":1,"syntology":null},{"url":"/paper/diversified-in-domain-synthesis-with","slug":"diversified-in-domain-synthesis-with","title":"Diversified in-domain synthesis with efficient fine-tuning for few-shot classification","date":"2023-12-05","arxiv_id":"2312.03046","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diversified-in-domain-synthesis-with#ran","syntology_url":"https://syntology.ai/paper/2312.03046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03046"}},"official":{"repos":["vturrisi/disef"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fergi-automatic-annotation-of-user","slug":"fergi-automatic-annotation-of-user","title":"FERGI: Automatic Scoring of User Preferences for Text-to-Image Generation from Spontaneous Facial Expression Reaction","date":"2023-12-05","arxiv_id":"2312.03187","repositories_listed":1,"syntology":null},{"url":"/paper/pea-diffusion-parameter-efficient-adapter","slug":"pea-diffusion-parameter-efficient-adapter","title":"PEA-Diffusion: Parameter-Efficient Adapter with Knowledge Distillation in non-English Text-to-Image Generation","date":"2023-11-28","arxiv_id":"2311.17086","repositories_listed":1,"syntology":null},{"url":"/paper/self-discovering-interpretable-diffusion","slug":"self-discovering-interpretable-diffusion","title":"Self-Discovering Interpretable Diffusion Latent Directions for Responsible Text-to-Image Generation","date":"2023-11-28","arxiv_id":"2311.17216","repositories_listed":1,"syntology":null},{"url":"/paper/self-correcting-llm-controlled-diffusion","slug":"self-correcting-llm-controlled-diffusion","title":"Self-correcting LLM-controlled Diffusion Models","date":"2023-11-27","arxiv_id":"2311.16090","repositories_listed":1,"syntology":null},{"url":"/paper/instastyle-inversion-noise-of-a-stylized","slug":"instastyle-inversion-noise-of-a-stylized","title":"InstaStyle: Inversion Noise of a Stylized Image is Secretly a Style Adviser","date":"2023-11-25","arxiv_id":"2311.15040","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/instastyle-inversion-noise-of-a-stylized#ran","syntology_url":"https://syntology.ai/paper/2311.15040","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.15040"}},"official":{"repos":["cuixing100876/instastyle"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neuroprompts-an-adaptive-framework-to","slug":"neuroprompts-an-adaptive-framework-to","title":"NeuroPrompts: An Adaptive Framework to Optimize Prompts for Text-to-Image Generation","date":"2023-11-20","arxiv_id":"2311.12229","repositories_listed":1,"syntology":null},{"url":"/paper/the-chosen-one-consistent-characters-in-text","slug":"the-chosen-one-consistent-characters-in-text","title":"The Chosen One: Consistent Characters in Text-to-Image Diffusion Models","date":"2023-11-16","arxiv_id":"2311.10093","repositories_listed":1,"syntology":null},{"url":"/paper/ufogen-you-forward-once-large-scale-text-to","slug":"ufogen-you-forward-once-large-scale-text-to","title":"UFOGen: You Forward Once Large Scale Text-to-Image Generation via Diffusion GANs","date":"2023-11-14","arxiv_id":"2311.09257","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ufogen-you-forward-once-large-scale-text-to#ran","syntology_url":"https://syntology.ai/paper/2311.09257","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09257"}},"official":{"repos":["xuyanwu/SIDDMs-UFOGen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/antifakeprompt-prompt-tuned-vision-language","slug":"antifakeprompt-prompt-tuned-vision-language","title":"AntifakePrompt: Prompt-Tuned Vision-Language Models are Fake Image Detectors","date":"2023-10-26","arxiv_id":"2310.17419","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/antifakeprompt-prompt-tuned-vision-language#ran","syntology_url":"https://syntology.ai/paper/2310.17419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17419"}},"official":{"repos":["nctu-eva-lab/antifakeprompt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-the-gap-between-synthetic-and","slug":"bridging-the-gap-between-synthetic-and","title":"Bridging the Gap between Synthetic and Authentic Images for Multimodal Machine Translation","date":"2023-10-20","arxiv_id":"2310.13361","repositories_listed":1,"syntology":null},{"url":"/paper/quality-diversity-through-human-feedback","slug":"quality-diversity-through-human-feedback","title":"Quality Diversity through Human Feedback: Towards Open-Ended Diversity-Driven Optimization","date":"2023-10-18","arxiv_id":"2310.12103","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quality-diversity-through-human-feedback#ran","syntology_url":"https://syntology.ai/paper/2310.12103","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12103"}},"official":{"repos":["ld-ing/qdhf"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/elucidating-the-design-space-of-classifier","slug":"elucidating-the-design-space-of-classifier","title":"Elucidating The Design Space of Classifier-Guided Diffusion Generation","date":"2023-10-17","arxiv_id":"2310.11311","repositories_listed":1,"syntology":null},{"url":"/paper/lamp-learn-a-motion-pattern-for-few-shot","slug":"lamp-learn-a-motion-pattern-for-few-shot","title":"LAMP: Learn A Motion Pattern for Few-Shot-Based Video Generation","date":"2023-10-16","arxiv_id":"2310.10769","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lamp-learn-a-motion-pattern-for-few-shot#ran","syntology_url":"https://syntology.ai/paper/2310.10769","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10769"}},"official":{"repos":["RQ-Wu/LAMP"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-blueprint-enabling-text-to-image","slug":"llm-blueprint-enabling-text-to-image","title":"LLM Blueprint: Enabling Text-to-Image Generation with Complex and Detailed Prompts","date":"2023-10-16","arxiv_id":"2310.10640","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llm-blueprint-enabling-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2310.10640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10640"}},"official":{"repos":["hananshafi/llmblueprint"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/tailored-visions-enhancing-text-to-image","slug":"tailored-visions-enhancing-text-to-image","title":"Tailored Visions: Enhancing Text-to-Image Generation with Personalized Prompt Rewriting","date":"2023-10-12","arxiv_id":"2310.08129","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":15,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tailored-visions-enhancing-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2310.08129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08129"}},"official":{"repos":["zzjchen/tailored-visions"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/conditionvideo-training-free-condition-guided","slug":"conditionvideo-training-free-condition-guided","title":"ConditionVideo: Training-Free Condition-Guided Text-to-Video Generation","date":"2023-10-11","arxiv_id":"2310.07697","repositories_listed":1,"syntology":null},{"url":"/paper/kandinsky-an-improved-text-to-image-synthesis","slug":"kandinsky-an-improved-text-to-image-synthesis","title":"Kandinsky: an Improved Text-to-Image Synthesis with Image Prior and Latent Diffusion","date":"2023-10-05","arxiv_id":"2310.03502","repositories_listed":1,"syntology":{"n":18,"n_ran":12,"n_constructed":1,"n_ran_checked":7,"n_instrument":5,"n_unverified":6,"n_honours":3,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"12 ran (of which 1 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/kandinsky-an-improved-text-to-image-synthesis#ran","syntology_url":"https://syntology.ai/paper/2310.03502","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03502"}},"official":{"repos":["ai-forever/Kandinsky-2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/conditional-diffusion-distillation","slug":"conditional-diffusion-distillation","title":"CoDi: Conditional Diffusion Distillation for Higher-Fidelity and Faster Image Generation","date":"2023-10-02","arxiv_id":"2310.01407","repositories_listed":1,"syntology":null},{"url":"/paper/instructcv-instruction-tuned-text-to-image","slug":"instructcv-instruction-tuned-text-to-image","title":"InstructCV: Instruction-Tuned Text-to-Image Diffusion Models as Vision Generalists","date":"2023-09-30","arxiv_id":"2310.00390","repositories_listed":1,"syntology":{"n":12,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":12,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/instructcv-instruction-tuned-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2310.00390","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00390"}},"official":{"repos":["AlaaLab/InstructCV"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-foundation-models-from-specialists","slug":"multimodal-foundation-models-from-specialists","title":"Multimodal Foundation Models: From Specialists to General-Purpose Assistants","date":"2023-09-18","arxiv_id":"2309.10020","repositories_listed":1,"syntology":null},{"url":"/paper/progressive-text-to-image-diffusion-with-soft","slug":"progressive-text-to-image-diffusion-with-soft","title":"Progressive Text-to-Image Diffusion with Soft Latent Direction","date":"2023-09-18","arxiv_id":"2309.09466","repositories_listed":1,"syntology":null},{"url":"/paper/viewpoint-textual-inversion-unleashing-novel","slug":"viewpoint-textual-inversion-unleashing-novel","title":"Viewpoint Textual Inversion: Discovering Scene Representations and 3D View Control in 2D Diffusion Models","date":"2023-09-14","arxiv_id":"2309.07986","repositories_listed":1,"syntology":null},{"url":"/paper/language-models-as-black-box-optimizers-for","slug":"language-models-as-black-box-optimizers-for","title":"Language Models as Black-Box Optimizers for Vision-Language Models","date":"2023-09-12","arxiv_id":"2309.05950","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/language-models-as-black-box-optimizers-for#ran","syntology_url":"https://syntology.ai/paper/2309.05950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05950"}},"official":{"repos":["shihongl1998/llm-as-a-blackbox-optimizer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/iti-gen-inclusive-text-to-image-generation","slug":"iti-gen-inclusive-text-to-image-generation","title":"ITI-GEN: Inclusive Text-to-Image Generation","date":"2023-09-11","arxiv_id":"2309.05569","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/iti-gen-inclusive-text-to-image-generation#ran","syntology_url":"https://syntology.ai/paper/2309.05569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05569"}},"official":{"repos":["humansensinglab/ITI-GEN"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/photoverse-tuning-free-image-customization","slug":"photoverse-tuning-free-image-customization","title":"PhotoVerse: Tuning-Free Image Customization with Text-to-Image Diffusion Models","date":"2023-09-11","arxiv_id":"2309.05793","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/photoverse-tuning-free-image-customization#ran","syntology_url":"https://syntology.ai/paper/2309.05793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05793"}},"official":null}},{"url":"/paper/from-text-to-mask-localizing-entities-using","slug":"from-text-to-mask-localizing-entities-using","title":"From Text to Mask: Localizing Entities Using the Attention of Text-to-Image Diffusion Models","date":"2023-09-08","arxiv_id":"2309.04109","repositories_listed":1,"syntology":null},{"url":"/paper/exchanging-based-multimodal-fusion-with","slug":"exchanging-based-multimodal-fusion-with","title":"Exchanging-based Multimodal Fusion with Transformer","date":"2023-09-05","arxiv_id":"2309.02190","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exchanging-based-multimodal-fusion-with#ran","syntology_url":"https://syntology.ai/paper/2309.02190","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.02190"}},"official":{"repos":["recklessronan/muse"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-autoregressive-multi-modal-models","slug":"scaling-autoregressive-multi-modal-models","title":"Scaling Autoregressive Multi-Modal Models: Pretraining and Instruction Tuning","date":"2023-09-05","arxiv_id":"2309.02591","repositories_listed":1,"syntology":null},{"url":"/paper/pathldm-text-conditioned-latent-diffusion","slug":"pathldm-text-conditioned-latent-diffusion","title":"PathLDM: Text conditioned Latent Diffusion Model for Histopathology","date":"2023-09-01","arxiv_id":"2309.00748","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":5,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":3,"n_no_contract":2,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 3 violated, 2 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pathldm-text-conditioned-latent-diffusion#ran","syntology_url":"https://syntology.ai/paper/2309.00748","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.00748"}},"official":{"repos":["cvlab-stonybrook/pathldm"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dense-text-to-image-generation-with-attention","slug":"dense-text-to-image-generation-with-attention","title":"Dense Text-to-Image Generation with Attention Modulation","date":"2023-08-24","arxiv_id":"2308.12964","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dense-text-to-image-generation-with-attention#ran","syntology_url":"https://syntology.ai/paper/2308.12964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12964"}},"official":{"repos":["naver-ai/densediffusion"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/aspire-language-guided-augmentation-for","slug":"aspire-language-guided-augmentation-for","title":"ASPIRE: Language-Guided Data Augmentation for Improving Robustness Against Spurious Correlations","date":"2023-08-19","arxiv_id":"2308.10103","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-generate-semantic-layouts-for","slug":"learning-to-generate-semantic-layouts-for","title":"Learning to Generate Semantic Layouts for Higher Text-Image Correspondence in Text-to-Image Synthesis","date":"2023-08-16","arxiv_id":"2308.08157","repositories_listed":1,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":2,"n_honours":4,"n_violates":3,"n_no_contract":7,"n_pointer_only":17,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 4 honoured, 3 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-generate-semantic-layouts-for#ran","syntology_url":"https://syntology.ai/paper/2308.08157","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08157"}},"official":{"repos":["pmh9960/GCDP"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/likelihood-based-text-to-image-evaluation","slug":"likelihood-based-text-to-image-evaluation","title":"Likelihood-Based Text-to-Image Evaluation with Patch-Level Perceptual and Semantic Credit Assignment","date":"2023-08-16","arxiv_id":"2308.08525","repositories_listed":1,"syntology":null},{"url":"/paper/story-visualization-by-online-text","slug":"story-visualization-by-online-text","title":"Story Visualization by Online Text Augmentation with Context Memory","date":"2023-08-15","arxiv_id":"2308.07575","repositories_listed":1,"syntology":null},{"url":"/paper/masked-attention-diffusion-guidance-for","slug":"masked-attention-diffusion-guidance-for","title":"Masked-Attention Diffusion Guidance for Spatially Controlling Text-to-Image Generation","date":"2023-08-11","arxiv_id":"2308.06027","repositories_listed":1,"syntology":null},{"url":"/paper/layoutllm-t2i-eliciting-layout-guidance-from","slug":"layoutllm-t2i-eliciting-layout-guidance-from","title":"LayoutLLM-T2I: Eliciting Layout Guidance from LLM for Text-to-Image Generation","date":"2023-08-09","arxiv_id":"2308.05095","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":4,"n_ran_checked":9,"n_instrument":2,"n_unverified":6,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":17,"phrase":"11 ran (of which 4 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/layoutllm-t2i-eliciting-layout-guidance-from#ran","syntology_url":"https://syntology.ai/paper/2308.05095","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.05095"}},"official":null}},{"url":"/paper/promptpaint-steering-text-to-image-generation","slug":"promptpaint-steering-text-to-image-generation","title":"PromptPaint: Steering Text-to-Image Generation Through Paint Medium-like Interactions","date":"2023-08-09","arxiv_id":"2308.05184","repositories_listed":1,"syntology":null},{"url":"/paper/conceptlab-creative-generation-using","slug":"conceptlab-creative-generation-using","title":"ConceptLab: Creative Concept Generation using VLM-Guided Diffusion Prior Constraints","date":"2023-08-03","arxiv_id":"2308.02669","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conceptlab-creative-generation-using#ran","syntology_url":"https://syntology.ai/paper/2308.02669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.02669"}},"official":{"repos":["kfirgoldberg/ConceptLab"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reverse-stable-diffusion-what-prompt-was-used","slug":"reverse-stable-diffusion-what-prompt-was-used","title":"Reverse Stable Diffusion: What prompt was used to generate this image?","date":"2023-08-02","arxiv_id":"2308.01472","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reverse-stable-diffusion-what-prompt-was-used#ran","syntology_url":"https://syntology.ai/paper/2308.01472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.01472"}},"official":{"repos":["croitorualin/reverse-stable-diffusion"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-bias-amplification-paradox-in-text-to","slug":"the-bias-amplification-paradox-in-text-to","title":"The Bias Amplification Paradox in Text-to-Image Generation","date":"2023-08-01","arxiv_id":"2308.00755","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-bias-amplification-paradox-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2308.00755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.00755"}},"official":{"repos":["preethiseshadri518/bias-amplification-paradox"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bagm-a-backdoor-attack-for-manipulating-text","slug":"bagm-a-backdoor-attack-for-manipulating-text","title":"BAGM: A Backdoor Attack for Manipulating Text-to-Image Generative Models","date":"2023-07-31","arxiv_id":"2307.16489","repositories_listed":1,"syntology":null},{"url":"/paper/learning-disentangled-discrete","slug":"learning-disentangled-discrete","title":"Learning Disentangled Discrete Representations","date":"2023-07-26","arxiv_id":"2307.14151","repositories_listed":1,"syntology":null},{"url":"/paper/subject-diffusion-open-domain-personalized","slug":"subject-diffusion-open-domain-personalized","title":"Subject-Diffusion:Open Domain Personalized Text-to-Image Generation without Test-time Fine-tuning","date":"2023-07-21","arxiv_id":"2307.11410","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/subject-diffusion-open-domain-personalized#ran","syntology_url":"https://syntology.ai/paper/2307.11410","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11410"}},"official":{"repos":["OPPO-Mente-Lab/Subject-Diffusion"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/planting-a-seed-of-vision-in-large-language","slug":"planting-a-seed-of-vision-in-large-language","title":"Planting a SEED of Vision in Large Language Model","date":"2023-07-16","arxiv_id":"2307.08041","repositories_listed":1,"syntology":null},{"url":"/paper/t2i-compbench-a-comprehensive-benchmark-for-1","slug":"t2i-compbench-a-comprehensive-benchmark-for-1","title":"T2I-CompBench: A Comprehensive Benchmark for Open-world Compositional Text-to-image Generation","date":"2023-07-12","arxiv_id":"2307.06350","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/t2i-compbench-a-comprehensive-benchmark-for-1#ran","syntology_url":"https://syntology.ai/paper/2307.06350","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.06350"}},"official":null}},{"url":"/paper/tiam-a-metric-for-evaluating-alignment-in","slug":"tiam-a-metric-for-evaluating-alignment-in","title":"TIAM -- A Metric for Evaluating Alignment in Text-to-Image Generation","date":"2023-07-11","arxiv_id":"2307.05134","repositories_listed":1,"syntology":null},{"url":"/paper/exact-diffusion-inversion-via-bi-directional","slug":"exact-diffusion-inversion-via-bi-directional","title":"Exact Diffusion Inversion via Bi-directional Integration Approximation","date":"2023-07-10","arxiv_id":"2307.10829","repositories_listed":1,"syntology":{"n":16,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":16,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/exact-diffusion-inversion-via-bi-directional#ran","syntology_url":"https://syntology.ai/paper/2307.10829","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.10829"}},"official":{"repos":["guoqiang-zhang-x/BDIA"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/aigciqa2023-a-large-scale-image-quality","slug":"aigciqa2023-a-large-scale-image-quality","title":"AIGCIQA2023: A Large-scale Image Quality Assessment Database for AI Generated Images: from the Perspectives of Quality, Authenticity and Correspondence","date":"2023-07-01","arxiv_id":"2307.00211","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-multimodal-representation","slug":"semi-supervised-multimodal-representation","title":"Semi-supervised Multimodal Representation Learning through a Global Workspace","date":"2023-06-27","arxiv_id":"2306.15711","repositories_listed":1,"syntology":null},{"url":"/paper/norm-guided-latent-space-exploration-for-text-1","slug":"norm-guided-latent-space-exploration-for-text-1","title":"Norm-guided latent space exploration for text-to-image generation","date":"2023-06-14","arxiv_id":"2306.08687","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/norm-guided-latent-space-exploration-for-text-1#ran","syntology_url":"https://syntology.ai/paper/2306.08687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08687"}},"official":{"repos":["dvirsamuel/SeedSelect"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ai-generated-image-detection-using-a-cross","slug":"ai-generated-image-detection-using-a-cross","title":"AI-Generated Image Detection using a Cross-Attention Enhanced Dual-Stream Network","date":"2023-06-12","arxiv_id":"2306.07005","repositories_listed":1,"syntology":null},{"url":"/paper/rewarded-soups-towards-pareto-optimal-1","slug":"rewarded-soups-towards-pareto-optimal-1","title":"Rewarded soups: towards Pareto-optimal alignment by interpolating weights fine-tuned on diverse rewards","date":"2023-06-07","arxiv_id":"2306.04488","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-difference-of-bert-style-and-clip","slug":"on-the-difference-of-bert-style-and-clip","title":"On the Difference of BERT-style and CLIP-style Text Encoders","date":"2023-06-06","arxiv_id":"2306.03678","repositories_listed":1,"syntology":null},{"url":"/paper/composition-and-deformance-measuring","slug":"composition-and-deformance-measuring","title":"Composition and Deformance: Measuring Imageability with a Text-to-Image Model","date":"2023-06-05","arxiv_id":"2306.03168","repositories_listed":1,"syntology":null},{"url":"/paper/detector-guidance-for-multi-object-text-to","slug":"detector-guidance-for-multi-object-text-to","title":"Detector Guidance for Multi-Object Text-to-Image Generation","date":"2023-06-04","arxiv_id":"2306.02236","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-grimm-open-ended-visual","slug":"intelligent-grimm-open-ended-visual","title":"Intelligent Grimm -- Open-ended Visual Storytelling via Latent Diffusion Models","date":"2023-06-01","arxiv_id":"2306.00973","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/intelligent-grimm-open-ended-visual#ran","syntology_url":"https://syntology.ai/paper/2306.00973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00973"}},"official":{"repos":["haoningwu3639/StoryGen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/vico-detail-preserving-visual-condition-for","slug":"vico-detail-preserving-visual-condition-for","title":"ViCo: Plug-and-play Visual Condition for Personalized Text-to-image Generation","date":"2023-06-01","arxiv_id":"2306.00971","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":9,"n_instrument":6,"n_unverified":0,"n_honours":2,"n_violates":3,"n_no_contract":4,"n_pointer_only":3,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 3 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vico-detail-preserving-visual-condition-for#ran","syntology_url":"https://syntology.ai/paper/2306.00971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00971"}},"official":{"repos":["haoosz/vico"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/nested-diffusion-processes-for-anytime-image","slug":"nested-diffusion-processes-for-anytime-image","title":"Nested Diffusion Processes for Anytime Image Generation","date":"2023-05-30","arxiv_id":"2305.19066","repositories_listed":1,"syntology":null},{"url":"/paper/raphael-text-to-image-generation-via-large","slug":"raphael-text-to-image-generation-via-large","title":"RAPHAEL: Text-to-Image Generation via Large Mixture of Diffusion Paths","date":"2023-05-29","arxiv_id":"2305.18295","repositories_listed":1,"syntology":null},{"url":"/paper/talecrafter-interactive-story-visualization","slug":"talecrafter-interactive-story-visualization","title":"TaleCrafter: Interactive Story Visualization with Multiple Characters","date":"2023-05-29","arxiv_id":"2305.18247","repositories_listed":1,"syntology":null},{"url":"/paper/generating-images-with-multimodal-language","slug":"generating-images-with-multimodal-language","title":"Generating Images with Multimodal Language Models","date":"2023-05-26","arxiv_id":"2305.17216","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/generating-images-with-multimodal-language#ran","syntology_url":"https://syntology.ai/paper/2305.17216","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17216"}},"official":{"repos":["kohjingyu/gill"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/blip-diffusion-pre-trained-subject-1","slug":"blip-diffusion-pre-trained-subject-1","title":"BLIP-Diffusion: Pre-trained Subject Representation for Controllable Text-to-Image Generation and Editing","date":"2023-05-24","arxiv_id":"2305.14720","repositories_listed":1,"syntology":null},{"url":"/paper/layoutgpt-compositional-visual-planning-and","slug":"layoutgpt-compositional-visual-planning-and","title":"LayoutGPT: Compositional Visual Planning and Generation with Large Language Models","date":"2023-05-24","arxiv_id":"2305.15393","repositories_listed":1,"syntology":{"n":19,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/layoutgpt-compositional-visual-planning-and#ran","syntology_url":"https://syntology.ai/paper/2305.15393","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15393"}},"official":{"repos":["weixi-feng/layoutgpt"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-applications-a-survey","slug":"vision-language-applications-a-survey","title":"Vision + Language Applications: A Survey","date":"2023-05-24","arxiv_id":"2305.14598","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-detail-preservation-for-customized","slug":"enhancing-detail-preservation-for-customized","title":"Enhancing Detail Preservation for Customized Text-to-Image Generation: A Regularization-Free Approach","date":"2023-05-23","arxiv_id":"2305.13579","repositories_listed":1,"syntology":null},{"url":"/paper/if-at-first-you-don-t-succeed-try-try-again","slug":"if-at-first-you-don-t-succeed-try-try-again","title":"If at First You Don't Succeed, Try, Try Again: Faithful Diffusion-based Text-to-Image Generation by Selection","date":"2023-05-22","arxiv_id":"2305.13308","repositories_listed":1,"syntology":null},{"url":"/paper/interactive-data-synthesis-for-systematic","slug":"interactive-data-synthesis-for-systematic","title":"Interactive Data Synthesis for Systematic Vision Adaptation via LLMs-AIGCs Collaboration","date":"2023-05-22","arxiv_id":"2305.12799","repositories_listed":1,"syntology":null},{"url":"/paper/discriminative-diffusion-models-as-few-shot","slug":"discriminative-diffusion-models-as-few-shot","title":"Discffusion: Discriminative Diffusion Models as Few-shot Vision and Language Learners","date":"2023-05-18","arxiv_id":"2305.10722","repositories_listed":1,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":1,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/discriminative-diffusion-models-as-few-shot#ran","syntology_url":"https://syntology.ai/paper/2305.10722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10722"}},"official":{"repos":["eric-ai-lab/dsd"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/instruct2act-mapping-multi-modality","slug":"instruct2act-mapping-multi-modality","title":"Instruct2Act: Mapping Multi-modality Instructions to Robotic Actions with Large Language Model","date":"2023-05-18","arxiv_id":"2305.11176","repositories_listed":1,"syntology":null},{"url":"/paper/videofactory-swap-attention-in-spatiotemporal","slug":"videofactory-swap-attention-in-spatiotemporal","title":"Swap Attention in Spatiotemporal Diffusions for Text-to-Video Generation","date":"2023-05-18","arxiv_id":"2305.10874","repositories_listed":1,"syntology":null},{"url":"/paper/x-iqe-explainable-image-quality-evaluation","slug":"x-iqe-explainable-image-quality-evaluation","title":"X-IQE: eXplainable Image Quality Evaluation for Text-to-Image Generation with Visual Large Language Models","date":"2023-05-18","arxiv_id":"2305.10843","repositories_listed":1,"syntology":null},{"url":"/paper/fastcomposer-tuning-free-multi-subject-image","slug":"fastcomposer-tuning-free-multi-subject-image","title":"FastComposer: Tuning-Free Multi-Subject Image Generation with Localized Attention","date":"2023-05-17","arxiv_id":"2305.10431","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":7,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/fastcomposer-tuning-free-multi-subject-image#ran","syntology_url":"https://syntology.ai/paper/2305.10431","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10431"}},"official":{"repos":["mit-han-lab/fastcomposer"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/what-you-see-is-what-you-read-improving-text-1","slug":"what-you-see-is-what-you-read-improving-text-1","title":"What You See is What You Read? Improving Text-Image Alignment Evaluation","date":"2023-05-17","arxiv_id":"2305.10400","repositories_listed":1,"syntology":null},{"url":"/paper/sur-adapter-enhancing-text-to-image-pre","slug":"sur-adapter-enhancing-text-to-image-pre","title":"SUR-adapter: Enhancing Text-to-Image Pre-trained Diffusion Models with Large Language Models","date":"2023-05-09","arxiv_id":"2305.05189","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sur-adapter-enhancing-text-to-image-pre#ran","syntology_url":"https://syntology.ai/paper/2305.05189","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.05189"}},"official":{"repos":["Qrange-group/SUR-adapter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/data-curation-for-image-captioning-with-text","slug":"data-curation-for-image-captioning-with-text","title":"The Role of Data Curation in Image Captioning","date":"2023-05-05","arxiv_id":"2305.03610","repositories_listed":1,"syntology":null},{"url":"/paper/disenbooth-disentangled-parameter-efficient","slug":"disenbooth-disentangled-parameter-efficient","title":"DisenBooth: Identity-Preserving Disentangled Tuning for Subject-Driven Text-to-Image Generation","date":"2023-05-05","arxiv_id":"2305.03374","repositories_listed":1,"syntology":null},{"url":"/paper/personalize-segment-anything-model-with-one","slug":"personalize-segment-anything-model-with-one","title":"Personalize Segment Anything Model with One Shot","date":"2023-05-04","arxiv_id":"2305.03048","repositories_listed":1,"syntology":null}],"record_sha256":"2bda3e4b084c1e91a2496596fba7bfcf84ef152e82a9a209c2f13a253af5b198","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}