{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-image-generation/papers/3","list_of":"/task/text-to-image-generation","task":"Text-to-Image Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":11,"rows_per_page":100,"rows":[201,300],"of":1085,"counts":{"archive_papers_tagged":1085,"with_a_code_link":546,"where_syntology_ran_a_sample":246,"not_listed_spam_title":0,"listed":1085,"listed_where_code_ran":246,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":215,"every_run_a_failure_of_syntologys_instrument":31,"listed_with_a_run_with_no_instrument_failure":215,"listed_every_run_a_failure_of_syntologys_instrument":31,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-image-generation","prev":"/task/text-to-image-generation/papers/2","next":"/task/text-to-image-generation/papers/4","papers":[{"url":"/paper/precise-fast-and-low-cost-concept-erasure-in","slug":"precise-fast-and-low-cost-concept-erasure-in","title":"Precise, Fast, and Low-cost Concept Erasure in Value Space: Orthogonal Complement Matters","date":"2024-12-09","arxiv_id":"2412.06143","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/precise-fast-and-low-cost-concept-erasure-in#ran","syntology_url":"https://syntology.ai/paper/2412.06143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06143"}},"official":{"repos":["wyuan1001/adavd"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/proactive-agents-for-multi-turn-text-to-image","slug":"proactive-agents-for-multi-turn-text-to-image","title":"Proactive Agents for Multi-Turn Text-to-Image Generation Under Uncertainty","date":"2024-12-09","arxiv_id":"2412.06771","repositories_listed":1,"syntology":null},{"url":"/paper/flexdit-dynamic-token-density-control-for","slug":"flexdit-dynamic-token-density-control-for","title":"FlexDiT: Dynamic Token Density Control for Diffusion Transformer","date":"2024-12-08","arxiv_id":"2412.06028","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/flexdit-dynamic-token-density-control-for#ran","syntology_url":"https://syntology.ai/paper/2412.06028","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06028"}},"official":{"repos":["changsn/FlexDiT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/safeguarding-text-to-image-generation-via","slug":"safeguarding-text-to-image-generation-via","title":"Safeguarding Text-to-Image Generation via Inference-Time Prompt-Noise Optimization","date":"2024-12-05","arxiv_id":"2412.03876","repositories_listed":1,"syntology":null},{"url":"/paper/scimage-how-good-are-multimodal-large","slug":"scimage-how-good-are-multimodal-large","title":"ScImage: How Good Are Multimodal Large Language Models at Scientific Text-to-Image Generation?","date":"2024-12-03","arxiv_id":"2412.02368","repositories_listed":1,"syntology":null},{"url":"/paper/mftf-mask-free-training-free-object-level","slug":"mftf-mask-free-training-free-object-level","title":"MFTF: Mask-free Training-free Object Level Layout Control Diffusion Model","date":"2024-12-02","arxiv_id":"2412.01284","repositories_listed":1,"syntology":null},{"url":"/paper/x-prompt-towards-universal-in-context-image","slug":"x-prompt-towards-universal-in-context-image","title":"X-Prompt: Towards Universal In-Context Image Generation in Auto-Regressive Vision Language Foundation Models","date":"2024-12-02","arxiv_id":"2412.01824","repositories_listed":1,"syntology":null},{"url":"/paper/playable-game-generation","slug":"playable-game-generation","title":"Playable Game Generation","date":"2024-12-01","arxiv_id":"2412.00887","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/playable-game-generation#ran","syntology_url":"https://syntology.ai/paper/2412.00887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.00887"}},"official":{"repos":["greatx3/playable-game-generation"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/jetformer-an-autoregressive-generative-model","slug":"jetformer-an-autoregressive-generative-model","title":"JetFormer: An Autoregressive Generative Model of Raw Images and Text","date":"2024-11-29","arxiv_id":"2411.19722","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/jetformer-an-autoregressive-generative-model#ran","syntology_url":"https://syntology.ai/paper/2411.19722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19722"}},"official":{"repos":["google-research/big_vision"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/amo-sampler-enhancing-text-rendering-with","slug":"amo-sampler-enhancing-text-rendering-with","title":"AMO Sampler: Enhancing Text Rendering with Overshooting","date":"2024-11-28","arxiv_id":"2411.19415","repositories_listed":1,"syntology":null},{"url":"/paper/relations-negations-and-numbers-looking-for","slug":"relations-negations-and-numbers-looking-for","title":"Relations, Negations, and Numbers: Looking for Logic in Generative Text-to-Image Models","date":"2024-11-26","arxiv_id":"2411.17066","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-evaluation-for-text-to-image","slug":"automatic-evaluation-for-text-to-image","title":"Automatic Evaluation for Text-to-image Generation: Task-decomposed Framework, Distilled Training, and Meta-evaluation Benchmark","date":"2024-11-23","arxiv_id":"2411.15488","repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-text-to-image-model-with","slug":"large-scale-text-to-image-model-with","title":"Large-Scale Text-to-Image Model with Inpainting is a Zero-Shot Subject-Driven Image Generator","date":"2024-11-23","arxiv_id":"2411.15466","repositories_listed":1,"syntology":null},{"url":"/paper/what-makes-a-scene-scene-graph-based","slug":"what-makes-a-scene-scene-graph-based","title":"What Makes a Scene ? Scene Graph-based Evaluation and Feedback for Controllable Generation","date":"2024-11-23","arxiv_id":"2411.15435","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-human-artifacts-from-text-to-image","slug":"detecting-human-artifacts-from-text-to-image","title":"Detecting Human Artifacts from Text-to-Image Models","date":"2024-11-21","arxiv_id":"2411.13842","repositories_listed":1,"syntology":null},{"url":"/paper/mmgenbench-evaluating-the-limits-of-lmms-from","slug":"mmgenbench-evaluating-the-limits-of-lmms-from","title":"MMGenBench: Evaluating the Limits of LMMs from the Text-to-Image Generation Perspective","date":"2024-11-21","arxiv_id":"2411.14062","repositories_listed":1,"syntology":null},{"url":"/paper/janusflow-harmonizing-autoregression-and","slug":"janusflow-harmonizing-autoregression-and","title":"JanusFlow: Harmonizing Autoregression and Rectified Flow for Unified Multimodal Understanding and Generation","date":"2024-11-12","arxiv_id":"2411.07975","repositories_listed":1,"syntology":null},{"url":"/paper/region-aware-text-to-image-generation-via","slug":"region-aware-text-to-image-generation-via","title":"Region-Aware Text-to-Image Generation via Hard Binding and Soft Refinement","date":"2024-11-10","arxiv_id":"2411.06558","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/region-aware-text-to-image-generation-via#ran","syntology_url":"https://syntology.ai/paper/2411.06558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.06558"}},"official":{"repos":["nju-pcalab/rag-diffusion"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/precision-or-recall-an-analysis-of-image","slug":"precision-or-recall-an-analysis-of-image","title":"Precision or Recall? An Analysis of Image Captions for Training Text-to-Image Generation Model","date":"2024-11-07","arxiv_id":"2411.05079","repositories_listed":1,"syntology":null},{"url":"/paper/taming-rectified-flow-for-inversion-and","slug":"taming-rectified-flow-for-inversion-and","title":"Taming Rectified Flow for Inversion and Editing","date":"2024-11-07","arxiv_id":"2411.04746","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":10,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/taming-rectified-flow-for-inversion-and#ran","syntology_url":"https://syntology.ai/paper/2411.04746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04746"}},"official":{"repos":["wangjiangshan0725/rf-solver-edit"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/training-free-regional-prompting-for","slug":"training-free-regional-prompting-for","title":"Training-free Regional Prompting for Diffusion Transformers","date":"2024-11-04","arxiv_id":"2411.02395","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-free-regional-prompting-for#ran","syntology_url":"https://syntology.ai/paper/2411.02395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02395"}},"official":{"repos":["instantX-research/Regional-Prompting-FLUX"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/groundit-grounding-diffusion-transformers-via","slug":"groundit-grounding-diffusion-transformers-via","title":"GrounDiT: Grounding Diffusion Transformers via Noisy Patch Transplantation","date":"2024-10-27","arxiv_id":"2410.20474","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/groundit-grounding-diffusion-transformers-via#ran","syntology_url":"https://syntology.ai/paper/2410.20474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20474"}},"official":{"repos":["KAIST-Visual-AI-Group/GrounDiT"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/mmm-rs-a-multi-modal-multi-gsd-multi-scene","slug":"mmm-rs-a-multi-modal-multi-gsd-multi-scene","title":"MMM-RS: A Multi-modal, Multi-GSD, Multi-scene Remote Sensing Dataset and Benchmark for Text-to-Image Generation","date":"2024-10-26","arxiv_id":"2410.22362","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mmm-rs-a-multi-modal-multi-gsd-multi-scene#ran","syntology_url":"https://syntology.ai/paper/2410.22362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22362"}},"official":{"repos":["ljl5261/mmm-rs"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/altogether-image-captioning-via-re-aligning","slug":"altogether-image-captioning-via-re-aligning","title":"Altogether: Image Captioning via Re-aligning Alt-text","date":"2024-10-22","arxiv_id":"2410.17251","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/altogether-image-captioning-via-re-aligning#ran","syntology_url":"https://syntology.ai/paper/2410.17251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17251"}},"official":{"repos":["facebookresearch/metaclip"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/offline-evaluation-of-set-based-text-to-image","slug":"offline-evaluation-of-set-based-text-to-image","title":"Offline Evaluation of Set-Based Text-to-Image Generation","date":"2024-10-22","arxiv_id":"2410.17331","repositories_listed":1,"syntology":null},{"url":"/paper/bigr-harnessing-binary-latent-codes-for-image","slug":"bigr-harnessing-binary-latent-codes-for-image","title":"BiGR: Harnessing Binary Latent Codes for Image Generation and Improved Visual Representation Capabilities","date":"2024-10-18","arxiv_id":"2410.14672","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bigr-harnessing-binary-latent-codes-for-image#ran","syntology_url":"https://syntology.ai/paper/2410.14672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14672"}},"official":{"repos":["haoosz/BiGR"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/boosting-imperceptibility-of-stable-diffusion","slug":"boosting-imperceptibility-of-stable-diffusion","title":"Boosting Imperceptibility of Stable Diffusion-based Adversarial Examples Generation with Momentum","date":"2024-10-17","arxiv_id":"2410.13122","repositories_listed":1,"syntology":null},{"url":"/paper/fluid-scaling-autoregressive-text-to-image","slug":"fluid-scaling-autoregressive-text-to-image","title":"Fluid: Scaling Autoregressive Text-to-image Generative Models with Continuous Tokens","date":"2024-10-17","arxiv_id":"2410.13863","repositories_listed":1,"syntology":null},{"url":"/paper/puma-empowering-unified-mllm-with-multi","slug":"puma-empowering-unified-mllm-with-multi","title":"PUMA: Empowering Unified MLLM with Multi-granular Visual Generation","date":"2024-10-17","arxiv_id":"2410.13861","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/puma-empowering-unified-mllm-with-multi#ran","syntology_url":"https://syntology.ai/paper/2410.13861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13861"}},"official":{"repos":["rongyaofang/puma"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/3dis-depth-driven-decoupled-instance","slug":"3dis-depth-driven-decoupled-instance","title":"3DIS: Depth-Driven Decoupled Instance Synthesis for Text-to-Image Generation","date":"2024-10-16","arxiv_id":"2410.12669","repositories_listed":1,"syntology":null},{"url":"/paper/facechain-fact-face-adapter-with-decoupled","slug":"facechain-fact-face-adapter-with-decoupled","title":"FaceChain-FACT: Face Adapter with Decoupled Training for Identity-preserved Personalization","date":"2024-10-16","arxiv_id":"2410.12312","repositories_listed":1,"syntology":null},{"url":"/paper/intermediate-representations-for-enhanced","slug":"intermediate-representations-for-enhanced","title":"Generating Intermediate Representations for Compositional Text-To-Image Generation","date":"2024-10-13","arxiv_id":"2410.09792","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intermediate-representations-for-enhanced#ran","syntology_url":"https://syntology.ai/paper/2410.09792","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09792"}},"official":{"repos":["rang1991/public-intermediate-semantics-for-generation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tulip-token-length-upgraded-clip","slug":"tulip-token-length-upgraded-clip","title":"TULIP: Token-length Upgraded CLIP","date":"2024-10-13","arxiv_id":"2410.10034","repositories_listed":1,"syntology":{"n":20,"n_ran":13,"n_constructed":0,"n_ran_checked":9,"n_instrument":4,"n_unverified":7,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":7,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/tulip-token-length-upgraded-clip#ran","syntology_url":"https://syntology.ai/paper/2410.10034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10034"}},"official":{"repos":["ivonajdenkoska/tulip"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/a-unified-debiasing-approach-for-vision","slug":"a-unified-debiasing-approach-for-vision","title":"A Unified Debiasing Approach for Vision-Language Models across Modalities and Tasks","date":"2024-10-10","arxiv_id":"2410.07593","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/a-unified-debiasing-approach-for-vision#ran","syntology_url":"https://syntology.ai/paper/2410.07593","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07593"}},"official":{"repos":["HoinJung/Unified-Debiaisng-VLM-SFID"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/minorityprompt-text-to-minority-image","slug":"minorityprompt-text-to-minority-image","title":"Minority-Focused Text-to-Image Generation via Prompt Optimization","date":"2024-10-10","arxiv_id":"2410.07838","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/minorityprompt-text-to-minority-image#ran","syntology_url":"https://syntology.ai/paper/2410.07838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07838"}},"official":{"repos":["anonymous5293/minorityprompt"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/evolvedirector-approaching-advanced-text-to","slug":"evolvedirector-approaching-advanced-text-to","title":"EvolveDirector: Approaching Advanced Text-to-Image Generation with Large Vision-Language Models","date":"2024-10-09","arxiv_id":"2410.07133","repositories_listed":1,"syntology":null},{"url":"/paper/training-free-diffusion-model-alignment-with","slug":"training-free-diffusion-model-alignment-with","title":"Training-free Diffusion Model Alignment with Sampling Demons","date":"2024-10-08","arxiv_id":"2410.05760","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/training-free-diffusion-model-alignment-with#ran","syntology_url":"https://syntology.ai/paper/2410.05760","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05760"}},"official":{"repos":["aiiu-lab/DemonSampling"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/images-speak-volumes-user-centric-assessment","slug":"images-speak-volumes-user-centric-assessment","title":"Images Speak Volumes: User-Centric Assessment of Image Generation for Accessible Communication","date":"2024-10-04","arxiv_id":"2410.03430","repositories_listed":1,"syntology":null},{"url":"/paper/data-extrapolation-for-text-to-image","slug":"data-extrapolation-for-text-to-image","title":"Data Extrapolation for Text-to-image Generation on Small Datasets","date":"2024-10-02","arxiv_id":"2410.01638","repositories_listed":1,"syntology":null},{"url":"/paper/cusconcept-customized-visual-concept","slug":"cusconcept-customized-visual-concept","title":"CusConcept: Customized Visual Concept Decomposition with Diffusion Models","date":"2024-10-01","arxiv_id":"2410.00398","repositories_listed":1,"syntology":null},{"url":"/paper/flowturbo-towards-real-time-flow-based-image","slug":"flowturbo-towards-real-time-flow-based-image","title":"FlowTurbo: Towards Real-time Flow-Based Image Generation with Velocity Refiner","date":"2024-09-26","arxiv_id":"2409.18128","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/flowturbo-towards-real-time-flow-based-image#ran","syntology_url":"https://syntology.ai/paper/2409.18128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.18128"}},"official":{"repos":["shiml20/flowturbo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/resolving-multi-condition-confusion-for","slug":"resolving-multi-condition-confusion-for","title":"Resolving Multi-Condition Confusion for Finetuning-Free Personalized Image Generation","date":"2024-09-26","arxiv_id":"2409.17920","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/resolving-multi-condition-confusion-for#ran","syntology_url":"https://syntology.ai/paper/2409.17920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.17920"}},"official":{"repos":["hqhqaq/mip-adapter"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/pixwizard-versatile-image-to-image-visual","slug":"pixwizard-versatile-image-to-image-visual","title":"PixWizard: Versatile Image-to-Image Visual Assistant with Open-Language Instructions","date":"2024-09-23","arxiv_id":"2409.15278","repositories_listed":1,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":11,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":2,"n_no_contract":9,"n_pointer_only":1,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 2 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pixwizard-versatile-image-to-image-visual#ran","syntology_url":"https://syntology.ai/paper/2409.15278","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.15278"}},"official":{"repos":["afeng-x/pixwizard"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarial-attacks-on-parts-of-speech-an","slug":"adversarial-attacks-on-parts-of-speech-an","title":"Adversarial Attacks on Parts of Speech: An Empirical Study in Text-to-Image Generation","date":"2024-09-21","arxiv_id":"2409.15381","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-image-hallucination-in-text-to","slug":"evaluating-image-hallucination-in-text-to","title":"Evaluating Image Hallucination in Text-to-Image Generation with Question-Answering","date":"2024-09-19","arxiv_id":"2409.12784","repositories_listed":1,"syntology":null},{"url":"/paper/storymaker-towards-holistic-consistent","slug":"storymaker-towards-holistic-consistent","title":"StoryMaker: Towards Holistic Consistent Characters in Text-to-image Generation","date":"2024-09-19","arxiv_id":"2409.12576","repositories_listed":1,"syntology":null},{"url":"/paper/omnigen-unified-image-generation","slug":"omnigen-unified-image-generation","title":"OmniGen: Unified Image Generation","date":"2024-09-17","arxiv_id":"2409.11340","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/omnigen-unified-image-generation#ran","syntology_url":"https://syntology.ai/paper/2409.11340","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.11340"}},"official":{"repos":["vectorspacelab/omnigen"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/finetuning-clip-to-reason-about-pairwise","slug":"finetuning-clip-to-reason-about-pairwise","title":"Finetuning CLIP to Reason about Pairwise Differences","date":"2024-09-15","arxiv_id":"2409.09721","repositories_listed":1,"syntology":null},{"url":"/paper/scribble-guided-diffusion-for-training-free","slug":"scribble-guided-diffusion-for-training-free","title":"Scribble-Guided Diffusion for Training-free Text-to-Image Generation","date":"2024-09-12","arxiv_id":"2409.08026","repositories_listed":1,"syntology":null},{"url":"/paper/textboost-towards-one-shot-personalization-of","slug":"textboost-towards-one-shot-personalization-of","title":"TextBoost: Towards One-Shot Personalization of Text-to-Image Models via Fine-tuning Text Encoder","date":"2024-09-12","arxiv_id":"2409.08248","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/textboost-towards-one-shot-personalization-of#ran","syntology_url":"https://syntology.ai/paper/2409.08248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.08248"}},"official":{"repos":["nahyeonkaty/textboost"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/styletokenizer-defining-image-style-by-a","slug":"styletokenizer-defining-image-style-by-a","title":"StyleTokenizer: Defining Image Style by a Single Instance for Controlling Diffusion Models","date":"2024-09-04","arxiv_id":"2409.02543","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/styletokenizer-defining-image-style-by-a#ran","syntology_url":"https://syntology.ai/paper/2409.02543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02543"}},"official":{"repos":["alipay/style-tokenizer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/resvg-enhancing-relation-and-semantic","slug":"resvg-enhancing-relation-and-semantic","title":"ResVG: Enhancing Relation and Semantic Understanding in Multiple Instances for Visual Grounding","date":"2024-08-29","arxiv_id":"2408.16314","repositories_listed":1,"syntology":null},{"url":"/paper/stereo-towards-adversarially-robust-concept","slug":"stereo-towards-adversarially-robust-concept","title":"STEREO: Towards Adversarially Robust Concept Erasing from Text-to-Image Generation Models","date":"2024-08-29","arxiv_id":"2408.16807","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/stereo-towards-adversarially-robust-concept#ran","syntology_url":"https://syntology.ai/paper/2408.16807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.16807"}},"official":{"repos":["koushiksrivats/robust-concept-erasing"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/merging-and-splitting-diffusion-paths-for","slug":"merging-and-splitting-diffusion-paths-for","title":"Merging and Splitting Diffusion Paths for Semantically Coherent Panoramas","date":"2024-08-28","arxiv_id":"2408.15660","repositories_listed":1,"syntology":null},{"url":"/paper/show-o-one-single-transformer-to-unify","slug":"show-o-one-single-transformer-to-unify","title":"Show-o: One Single Transformer to Unify Multimodal Understanding and Generation","date":"2024-08-22","arxiv_id":"2408.12528","repositories_listed":1,"syntology":null},{"url":"/paper/megafusion-extend-diffusion-models-towards","slug":"megafusion-extend-diffusion-models-towards","title":"MegaFusion: Extend Diffusion Models towards Higher-resolution Image Generation without Further Tuning","date":"2024-08-20","arxiv_id":"2408.11001","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/megafusion-extend-diffusion-models-towards#ran","syntology_url":"https://syntology.ai/paper/2408.11001","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11001"}},"official":{"repos":["haoningwu3639/MegaFusion"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/muses-3d-controllable-image-generation-via","slug":"muses-3d-controllable-image-generation-via","title":"MUSES: 3D-Controllable Image Generation via Multi-Modal Agent Collaboration","date":"2024-08-20","arxiv_id":"2408.10605","repositories_listed":1,"syntology":null},{"url":"/paper/zepo-zero-shot-portrait-stylization-with","slug":"zepo-zero-shot-portrait-stylization-with","title":"ZePo: Zero-Shot Portrait Stylization with Faster Sampling","date":"2024-08-10","arxiv_id":"2408.05492","repositories_listed":1,"syntology":null},{"url":"/paper/2408-02814","slug":"2408-02814","title":"Pre-trained Encoder Inference: Revealing Upstream Encoders In Downstream Machine Learning Services","date":"2024-08-05","arxiv_id":"2408.02814","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00523","slug":"2408-00523","title":"Fuzz-Testing Meets LLM-Based Agents: An Automated and Efficient Framework for Jailbreaking Text-To-Image Generation Models","date":"2024-08-01","arxiv_id":"2408.00523","repositories_listed":1,"syntology":null},{"url":"/paper/reproducibility-study-of-iti-gen-inclusive","slug":"reproducibility-study-of-iti-gen-inclusive","title":"Reproducibility Study of \"ITI-GEN: Inclusive Text-to-Image Generation\"","date":"2024-07-29","arxiv_id":"2407.19996","repositories_listed":1,"syntology":null},{"url":"/paper/attentionhand-text-driven-controllable-hand","slug":"attentionhand-text-driven-controllable-hand","title":"AttentionHand: Text-driven Controllable Hand Image Generation for 3D Hand Reconstruction in the Wild","date":"2024-07-25","arxiv_id":"2407.18034","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/attentionhand-text-driven-controllable-hand#ran","syntology_url":"https://syntology.ai/paper/2407.18034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18034"}},"official":{"repos":["redorangeyellowy/AttentionHand"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/record-reasoning-and-correcting-diffusion-for","slug":"record-reasoning-and-correcting-diffusion-for","title":"ReCorD: Reasoning and Correcting Diffusion for HOI Generation","date":"2024-07-25","arxiv_id":"2407.17911","repositories_listed":1,"syntology":null},{"url":"/paper/membench-memorized-image-trigger-prompt","slug":"membench-memorized-image-trigger-prompt","title":"MemBench: Memorized Image Trigger Prompt Dataset for Diffusion Models","date":"2024-07-24","arxiv_id":"2407.17095","repositories_listed":1,"syntology":null},{"url":"/paper/artist-aesthetically-controllable-text-driven","slug":"artist-aesthetically-controllable-text-driven","title":"DiffArtist: Towards Structure and Appearance Controllable Image Stylization","date":"2024-07-22","arxiv_id":"2407.15842","repositories_listed":1,"syntology":null},{"url":"/paper/greenstableyolo-optimizing-inference-time-and","slug":"greenstableyolo-optimizing-inference-time-and","title":"GreenStableYolo: Optimizing Inference Time and Image Quality of Text-to-Image Generation","date":"2024-07-20","arxiv_id":"2407.14982","repositories_listed":1,"syntology":null},{"url":"/paper/subject-driven-text-to-image-generation-via-1","slug":"subject-driven-text-to-image-generation-via-1","title":"Subject-driven Text-to-Image Generation via Preference-based Reinforcement Learning","date":"2024-07-16","arxiv_id":"2407.12164","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/subject-driven-text-to-image-generation-via-1#ran","syntology_url":"https://syntology.ai/paper/2407.12164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12164"}},"official":{"repos":["andrew-miao/RPO"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/temporalstory-enhancing-consistency-in-story","slug":"temporalstory-enhancing-consistency-in-story","title":"ContextualStory: Consistent Visual Storytelling with Spatially-Enhanced and Storyline Context","date":"2024-07-13","arxiv_id":"2407.09774","repositories_listed":1,"syntology":null},{"url":"/paper/conceptexpress-harnessing-diffusion-models","slug":"conceptexpress-harnessing-diffusion-models","title":"ConceptExpress: Harnessing Diffusion Models for Single-image Unsupervised Concept Extraction","date":"2024-07-09","arxiv_id":"2407.07077","repositories_listed":1,"syntology":null},{"url":"/paper/humanrefiner-benchmarking-abnormal-human","slug":"humanrefiner-benchmarking-abnormal-human","title":"HumanRefiner: Benchmarking Abnormal Human Generation and Refining with Coarse-to-fine Pose-Reversible Guidance","date":"2024-07-09","arxiv_id":"2407.06937","repositories_listed":1,"syntology":null},{"url":"/paper/powerful-and-flexible-personalized-text-to","slug":"powerful-and-flexible-personalized-text-to","title":"Powerful and Flexible: Personalized Text-to-Image Generation via Reinforcement Learning","date":"2024-07-09","arxiv_id":"2407.06642","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/powerful-and-flexible-personalized-text-to#ran","syntology_url":"https://syntology.ai/paper/2407.06642","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06642"}},"official":{"repos":["wfanyue/dpg-t2i-personalization"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mj-bench-is-your-multimodal-reward-model","slug":"mj-bench-is-your-multimodal-reward-model","title":"MJ-Bench: Is Your Multimodal Reward Model Really a Good Judge for Text-to-Image Generation?","date":"2024-07-05","arxiv_id":"2407.04842","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mj-bench-is-your-multimodal-reward-model#ran","syntology_url":"https://syntology.ai/paper/2407.04842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04842"}},"official":{"repos":["MJ-Bench/MJ-Bench"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-personalized-text-to-image","slug":"efficient-personalized-text-to-image","title":"Efficient Personalized Text-to-image Generation by Leveraging Textual Subspace","date":"2024-06-30","arxiv_id":"2407.00608","repositories_listed":1,"syntology":null},{"url":"/paper/instantstyle-plus-style-transfer-with-content","slug":"instantstyle-plus-style-transfer-with-content","title":"InstantStyle-Plus: Style Transfer with Content-Preserving in Text-to-Image Generation","date":"2024-06-30","arxiv_id":"2407.00788","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/instantstyle-plus-style-transfer-with-content#ran","syntology_url":"https://syntology.ai/paper/2407.00788","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00788"}},"official":{"repos":["instantx-research/instantstyle-plus"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/llm4gen-leveraging-semantic-representation-of","slug":"llm4gen-leveraging-semantic-representation-of","title":"LLM4GEN: Leveraging Semantic Representation of LLMs for Text-to-Image Generation","date":"2024-06-30","arxiv_id":"2407.00737","repositories_listed":1,"syntology":null},{"url":"/paper/the-factuality-tax-of-diversity-intervened","slug":"the-factuality-tax-of-diversity-intervened","title":"The Factuality Tax of Diversity-Intervened Text-to-Image Generation: Benchmark and Fact-Augmented Intervention","date":"2024-06-29","arxiv_id":"2407.00377","repositories_listed":1,"syntology":null},{"url":"/paper/popalign-population-level-alignment-for-fair","slug":"popalign-population-level-alignment-for-fair","title":"PopAlign: Population-Level Alignment for Fair Text-to-Image Generation","date":"2024-06-28","arxiv_id":"2406.19668","repositories_listed":1,"syntology":null},{"url":"/paper/prompt-refinement-with-image-pivot-for-text","slug":"prompt-refinement-with-image-pivot-for-text","title":"Prompt Refinement with Image Pivot for Text-to-Image Generation","date":"2024-06-28","arxiv_id":"2407.00247","repositories_listed":1,"syntology":null},{"url":"/paper/anycontrol-create-your-artwork-with-versatile","slug":"anycontrol-create-your-artwork-with-versatile","title":"AnyControl: Create Your Artwork with Versatile Control on Text-to-Image Generation","date":"2024-06-27","arxiv_id":"2406.18958","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":3,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/anycontrol-create-your-artwork-with-versatile#ran","syntology_url":"https://syntology.ai/paper/2406.18958","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18958"}},"official":{"repos":["open-mmlab/anycontrol"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evalalign-evaluating-text-to-image-models","slug":"evalalign-evaluating-text-to-image-models","title":"EVALALIGN: Supervised Fine-Tuning Multimodal LLMs with Human-Aligned Data for Evaluating Text-to-Image Models","date":"2024-06-24","arxiv_id":"2406.16562","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evalalign-evaluating-text-to-image-models#ran","syntology_url":"https://syntology.ai/paper/2406.16562","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16562"}},"official":{"repos":["sais-fuxi/evalalign"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/fine-tuning-diffusion-models-for-enhancing","slug":"fine-tuning-diffusion-models-for-enhancing","title":"FaceScore: Benchmarking and Enhancing Face Quality in Human Generation","date":"2024-06-24","arxiv_id":"2406.17100","repositories_listed":1,"syntology":null},{"url":"/paper/repulsive-score-distillation-for-diverse","slug":"repulsive-score-distillation-for-diverse","title":"Repulsive Latent Score Distillation for Solving Inverse Problems","date":"2024-06-24","arxiv_id":"2406.16683","repositories_listed":1,"syntology":null},{"url":"/paper/injecting-bias-in-text-to-image-models-via","slug":"injecting-bias-in-text-to-image-models-via","title":"Backdooring Bias into Text-to-Image Models","date":"2024-06-21","arxiv_id":"2406.15213","repositories_listed":1,"syntology":null},{"url":"/paper/aitti-learning-adaptive-inclusive-token-for","slug":"aitti-learning-adaptive-inclusive-token-for","title":"AITTI: Learning Adaptive Inclusive Token for Text-to-Image Generation","date":"2024-06-18","arxiv_id":"2406.12805","repositories_listed":1,"syntology":null},{"url":"/paper/generative-visual-instruction-tuning","slug":"generative-visual-instruction-tuning","title":"Generative Visual Instruction Tuning","date":"2024-06-17","arxiv_id":"2406.11262","repositories_listed":1,"syntology":null},{"url":"/paper/mixture-of-subspaces-in-low-rank-adaptation","slug":"mixture-of-subspaces-in-low-rank-adaptation","title":"Mixture-of-Subspaces in Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.11909","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mixture-of-subspaces-in-low-rank-adaptation#ran","syntology_url":"https://syntology.ai/paper/2406.11909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11909"}},"official":{"repos":["wutaiqiang/moslora"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/star-scale-wise-text-to-image-generation-via","slug":"star-scale-wise-text-to-image-generation-via","title":"STAR: Scale-wise Text-to-image generation via Auto-Regressive representations","date":"2024-06-16","arxiv_id":"2406.10797","repositories_listed":1,"syntology":null},{"url":"/paper/make-it-count-text-to-image-generation-with","slug":"make-it-count-text-to-image-generation-with","title":"Make It Count: Text-to-Image Generation with an Accurate Number of Objects","date":"2024-06-14","arxiv_id":"2406.10210","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/make-it-count-text-to-image-generation-with#ran","syntology_url":"https://syntology.ai/paper/2406.10210","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10210"}},"official":null}},{"url":"/paper/batch-instructed-gradient-for-prompt","slug":"batch-instructed-gradient-for-prompt","title":"Batch-Instructed Gradient for Prompt Evolution:Systematic Prompt Optimization for Enhanced Text-to-Image Synthesis","date":"2024-06-13","arxiv_id":"2406.08713","repositories_listed":1,"syntology":null},{"url":"/paper/cfg-manifold-constrained-classifier-free","slug":"cfg-manifold-constrained-classifier-free","title":"CFG++: Manifold-constrained Classifier Free Guidance for Diffusion Models","date":"2024-06-12","arxiv_id":"2406.08070","repositories_listed":1,"syntology":null},{"url":"/paper/image-textualization-an-automatic-framework","slug":"image-textualization-an-automatic-framework","title":"Image Textualization: An Automatic Framework for Creating Accurate and Detailed Image Descriptions","date":"2024-06-11","arxiv_id":"2406.07502","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/image-textualization-an-automatic-framework#ran","syntology_url":"https://syntology.ai/paper/2406.07502","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07502"}},"official":{"repos":["sterzhang/image-textualization"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ms-diffusion-multi-subject-zero-shot-image","slug":"ms-diffusion-multi-subject-zero-shot-image","title":"MS-Diffusion: Multi-subject Zero-shot Image Personalization with Layout Guidance","date":"2024-06-11","arxiv_id":"2406.07209","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ms-diffusion-multi-subject-zero-shot-image#ran","syntology_url":"https://syntology.ai/paper/2406.07209","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07209"}},"official":{"repos":["MS-Diffusion/MS-Diffusion"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-visual-concepts-across-models","slug":"understanding-visual-concepts-across-models","title":"Understanding Visual Concepts Across Models","date":"2024-06-11","arxiv_id":"2406.07506","repositories_listed":1,"syntology":null},{"url":"/paper/regularized-training-with-generated-datasets","slug":"regularized-training-with-generated-datasets","title":"Regularized Training with Generated Datasets for Name-Only Transfer of Vision-Language Models","date":"2024-06-08","arxiv_id":"2406.05432","repositories_listed":1,"syntology":null},{"url":"/paper/pqpp-a-joint-benchmark-for-text-to-image","slug":"pqpp-a-joint-benchmark-for-text-to-image","title":"PQPP: A Joint Benchmark for Text-to-Image Prompt and Query Performance Prediction","date":"2024-06-07","arxiv_id":"2406.04746","repositories_listed":1,"syntology":null},{"url":"/paper/genai-arena-an-open-evaluation-platform-for","slug":"genai-arena-an-open-evaluation-platform-for","title":"GenAI Arena: An Open Evaluation Platform for Generative Models","date":"2024-06-06","arxiv_id":"2406.04485","repositories_listed":1,"syntology":null},{"url":"/paper/step-aware-preference-optimization-aligning","slug":"step-aware-preference-optimization-aligning","title":"Aesthetic Post-Training Diffusion Models from Generic Preferences with Step-by-step Preference Optimization","date":"2024-06-06","arxiv_id":"2406.04314","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/step-aware-preference-optimization-aligning#ran","syntology_url":"https://syntology.ai/paper/2406.04314","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04314"}},"official":{"repos":["rockeycoss/spo"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/lumina-next-making-lumina-t2x-stronger-and","slug":"lumina-next-making-lumina-t2x-stronger-and","title":"Lumina-Next: Making Lumina-T2X Stronger and Faster with Next-DiT","date":"2024-06-05","arxiv_id":"2406.18583","repositories_listed":1,"syntology":null},{"url":"/paper/stable-pose-leveraging-transformers-for-pose","slug":"stable-pose-leveraging-transformers-for-pose","title":"Stable-Pose: Leveraging Transformers for Pose-Guided Text-to-Image Generation","date":"2024-06-04","arxiv_id":"2406.02485","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":4,"n_no_contract":6,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 4 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stable-pose-leveraging-transformers-for-pose#ran","syntology_url":"https://syntology.ai/paper/2406.02485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.02485"}},"official":{"repos":["ai-med/stablepose"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reflection-reinforced-self-training-for","slug":"reflection-reinforced-self-training-for","title":"Re-ReST: Reflection-Reinforced Self-Training for Language Agents","date":"2024-06-03","arxiv_id":"2406.01495","repositories_listed":1,"syntology":null}],"record_sha256":"cc65abc888f30a172c3c5239de74e58f3ce6936ae4773f3b5c5bcd6b7ebbce68","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}