{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-image-generation-1/papers/2","list_of":"/task/text-to-image-generation-1","task":"Text to Image Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":10,"rows_per_page":100,"rows":[101,200],"of":969,"counts":{"archive_papers_tagged":969,"with_a_code_link":461,"where_syntology_ran_a_sample":198,"not_listed_spam_title":0,"listed":969,"listed_where_code_ran":198,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":171,"every_run_a_failure_of_syntologys_instrument":27,"listed_with_a_run_with_no_instrument_failure":171,"listed_every_run_a_failure_of_syntologys_instrument":27,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-image-generation-1","prev":"/task/text-to-image-generation-1","next":"/task/text-to-image-generation-1/papers/3","papers":[{"url":"/paper/freegraftor-training-free-cross-image-feature","slug":"freegraftor-training-free-cross-image-feature","title":"FreeGraftor: Training-Free Cross-Image Feature Grafting for Subject-Driven Text-to-Image Generation","date":"2025-04-22","arxiv_id":"2504.15958","repositories_listed":1,"syntology":null},{"url":"/paper/cross-attention-for-state-based-model-rwkv-7","slug":"cross-attention-for-state-based-model-rwkv-7","title":"Cross-attention for State-based model RWKV-7","date":"2025-04-19","arxiv_id":"2504.14260","repositories_listed":1,"syntology":null},{"url":"/paper/artistauditor-auditing-artist-style-pirate-in","slug":"artistauditor-auditing-artist-style-pirate-in","title":"ArtistAuditor: Auditing Artist Style Pirate in Text-to-Image Generation Models","date":"2025-04-17","arxiv_id":"2504.13061","repositories_listed":1,"syntology":null},{"url":"/paper/personalized-text-to-image-generation-with","slug":"personalized-text-to-image-generation-with","title":"Personalized Text-to-Image Generation with Auto-Regressive Models","date":"2025-04-17","arxiv_id":"2504.13162","repositories_listed":1,"syntology":null},{"url":"/paper/anchor-token-matching-implicit-structure","slug":"anchor-token-matching-implicit-structure","title":"Anchor Token Matching: Implicit Structure Locking for Training-free AR Image Editing","date":"2025-04-14","arxiv_id":"2504.10434","repositories_listed":1,"syntology":null},{"url":"/paper/dydit-dynamic-diffusion-transformers-for","slug":"dydit-dynamic-diffusion-transformers-for","title":"DyDiT++: Dynamic Diffusion Transformers for Efficient Visual Generation","date":"2025-04-09","arxiv_id":"2504.06803","repositories_listed":1,"syntology":null},{"url":"/paper/omnicaptioner-one-captioner-to-rule-them-all","slug":"omnicaptioner-one-captioner-to-rule-them-all","title":"OmniCaptioner: One Captioner to Rule Them All","date":"2025-04-09","arxiv_id":"2504.07089","repositories_listed":1,"syntology":null},{"url":"/paper/detection-limits-and-statistical-separability","slug":"detection-limits-and-statistical-separability","title":"Detection Limits and Statistical Separability of Tree Ring Watermarks in Rectified Flow-based Text-to-Image Generation Models","date":"2025-04-04","arxiv_id":"2504.03850","repositories_listed":1,"syntology":null},{"url":"/paper/ai2agent-an-end-to-end-framework-for","slug":"ai2agent-an-end-to-end-framework-for","title":"AI2Agent: An End-to-End Framework for Deploying AI Projects as Autonomous Agents","date":"2025-03-31","arxiv_id":"2503.23948","repositories_listed":1,"syntology":null},{"url":"/paper/lumina-image-2-0-a-unified-and-efficient","slug":"lumina-image-2-0-a-unified-and-efficient","title":"Lumina-Image 2.0: A Unified and Efficient Image Generative Framework","date":"2025-03-27","arxiv_id":"2503.21758","repositories_listed":1,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":11,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lumina-image-2-0-a-unified-and-efficient#ran","syntology_url":"https://syntology.ai/paper/2503.21758","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21758"}},"official":{"repos":["alpha-vllm/lumina-image-2.0"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/optimal-stepsize-for-diffusion-sampling","slug":"optimal-stepsize-for-diffusion-sampling","title":"Optimal Stepsize for Diffusion Sampling","date":"2025-03-27","arxiv_id":"2503.21774","repositories_listed":1,"syntology":null},{"url":"/paper/rectable-fast-modeling-tabular-data-with","slug":"rectable-fast-modeling-tabular-data-with","title":"RecTable: Fast Modeling Tabular Data with Rectified Flow","date":"2025-03-26","arxiv_id":"2503.20731","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-down-text-encoders-of-text-to-image","slug":"scaling-down-text-encoders-of-text-to-image","title":"Scaling Down Text Encoders of Text-to-Image Diffusion Models","date":"2025-03-25","arxiv_id":"2503.19897","repositories_listed":1,"syntology":null},{"url":"/paper/unseen-from-seen-rewriting-observation","slug":"unseen-from-seen-rewriting-observation","title":"Unseen from Seen: Rewriting Observation-Instruction Using Foundation Models for Augmenting Vision-Language Navigation","date":"2025-03-23","arxiv_id":"2503.18065","repositories_listed":1,"syntology":null},{"url":"/paper/verbdiff-text-only-diffusion-models-with","slug":"verbdiff-text-only-diffusion-models-with","title":"VerbDiff: Text-Only Diffusion Models with Enhanced Interaction Awareness","date":"2025-03-20","arxiv_id":"2503.16406","repositories_listed":1,"syntology":null},{"url":"/paper/reflect-dit-inference-time-scaling-for-text-1","slug":"reflect-dit-inference-time-scaling-for-text-1","title":"Reflect-DiT: Inference-Time Scaling for Text-to-Image Diffusion Transformers via In-Context Reflection","date":"2025-03-15","arxiv_id":"2503.12271","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/reflect-dit-inference-time-scaling-for-text-1#ran","syntology_url":"https://syntology.ai/paper/2503.12271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.12271"}},"official":null}},{"url":"/paper/towards-better-alignment-training-diffusion","slug":"towards-better-alignment-training-diffusion","title":"Towards Better Alignment: Training Diffusion Models with Reinforcement Learning Against Sparse Rewards","date":"2025-03-14","arxiv_id":"2503.11240","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-better-alignment-training-diffusion#ran","syntology_url":"https://syntology.ai/paper/2503.11240","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.11240"}},"official":{"repos":["hu-zijing/b2-diffurl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/got-unleashing-reasoning-capability-of","slug":"got-unleashing-reasoning-capability-of","title":"GoT: Unleashing Reasoning Capability of Multimodal Large Language Model for Visual Generation and Editing","date":"2025-03-13","arxiv_id":"2503.10639","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/got-unleashing-reasoning-capability-of#ran","syntology_url":"https://syntology.ai/paper/2503.10639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10639"}},"official":{"repos":["rongyaofang/got"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/neighboring-autoregressive-modeling-for","slug":"neighboring-autoregressive-modeling-for","title":"Neighboring Autoregressive Modeling for Efficient Visual Generation","date":"2025-03-12","arxiv_id":"2503.10696","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/neighboring-autoregressive-modeling-for#ran","syntology_url":"https://syntology.ai/paper/2503.10696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10696"}},"official":{"repos":["thisisbillhe/nar"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/aligning-text-to-image-in-diffusion-models-is","slug":"aligning-text-to-image-in-diffusion-models-is","title":"Aligning Text to Image in Diffusion Models is Easier Than You Think","date":"2025-03-11","arxiv_id":"2503.08250","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aligning-text-to-image-in-diffusion-models-is#ran","syntology_url":"https://syntology.ai/paper/2503.08250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08250"}},"official":{"repos":["softrepa/SoftREPA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lightgen-efficient-image-generation-through","slug":"lightgen-efficient-image-generation-through","title":"LightGen: Efficient Image Generation through Knowledge Distillation and Direct Preference Optimization","date":"2025-03-11","arxiv_id":"2503.08619","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":1,"n_honours":3,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 3 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lightgen-efficient-image-generation-through#ran","syntology_url":"https://syntology.ai/paper/2503.08619","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08619"}},"official":{"repos":["xianfengwu01/lightgen"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unleashing-the-potential-of-large-language-3","slug":"unleashing-the-potential-of-large-language-3","title":"Unleashing the Potential of Large Language Models for Text-to-Image Generation through Autoregressive Representation Alignment","date":"2025-03-10","arxiv_id":"2503.07334","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-alignment-and-noise-refinement","slug":"fine-grained-alignment-and-noise-refinement","title":"Fine-Grained Alignment and Noise Refinement for Compositional Text-to-Image Generation","date":"2025-03-09","arxiv_id":"2503.06506","repositories_listed":1,"syntology":null},{"url":"/paper/learning-few-step-diffusion-models-by","slug":"learning-few-step-diffusion-models-by","title":"Learning Few-Step Diffusion Models by Trajectory Distribution Matching","date":"2025-03-09","arxiv_id":"2503.06674","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-few-step-diffusion-models-by#ran","syntology_url":"https://syntology.ai/paper/2503.06674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06674"}},"official":{"repos":["Luo-Yihong/TDM"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/x2i-seamless-integration-of-multimodal","slug":"x2i-seamless-integration-of-multimodal","title":"X2I: Seamless Integration of Multimodal Understanding into Diffusion Transformer via Attention Distillation","date":"2025-03-08","arxiv_id":"2503.06134","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/x2i-seamless-integration-of-multimodal#ran","syntology_url":"https://syntology.ai/paper/2503.06134","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06134"}},"official":{"repos":["oppo-mente-lab/x2i"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/feynman-kac-correctors-in-diffusion-annealing","slug":"feynman-kac-correctors-in-diffusion-annealing","title":"Feynman-Kac Correctors in Diffusion: Annealing, Guidance, and Product of Experts","date":"2025-03-04","arxiv_id":"2503.02819","repositories_listed":1,"syntology":null},{"url":"/paper/generative-modeling-of-microweather-wind","slug":"generative-modeling-of-microweather-wind","title":"Generative Modeling of Microweather Wind Velocities for Urban Air Mobility","date":"2025-03-04","arxiv_id":"2503.02690","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-representation-alignment-for-image","slug":"multimodal-representation-alignment-for-image","title":"Multimodal Representation Alignment for Image Generation: Text-Image Interleaved Control Is Easier Than You Think","date":"2025-02-27","arxiv_id":"2502.20172","repositories_listed":1,"syntology":null},{"url":"/paper/forest-frame-of-reference-evaluation-in","slug":"forest-frame-of-reference-evaluation-in","title":"FoREST: Frame of Reference Evaluation in Spatial Reasoning Tasks","date":"2025-02-25","arxiv_id":"2502.17775","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-multimodal-models-for","slug":"multi-agent-multimodal-models-for","title":"Multi-Agent Multimodal Models for Multicultural Text to Image Generation","date":"2025-02-21","arxiv_id":"2502.15972","repositories_listed":1,"syntology":null},{"url":"/paper/chats-combining-human-aligned-optimization","slug":"chats-combining-human-aligned-optimization","title":"CHATS: Combining Human-Aligned Optimization and Test-Time Sampling for Text-to-Image Generation","date":"2025-02-18","arxiv_id":"2502.12579","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chats-combining-human-aligned-optimization#ran","syntology_url":"https://syntology.ai/paper/2502.12579","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.12579"}},"official":{"repos":["AIDC-AI/CHATS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-sample-effective-and-diverse","slug":"learning-to-sample-effective-and-diverse","title":"Learning to Sample Effective and Diverse Prompts for Text-to-Image Generation","date":"2025-02-17","arxiv_id":"2502.11477","repositories_listed":1,"syntology":null},{"url":"/paper/direct-ascent-synthesis-revealing-hidden","slug":"direct-ascent-synthesis-revealing-hidden","title":"Direct Ascent Synthesis: Revealing Hidden Generative Capabilities in Discriminative Models","date":"2025-02-11","arxiv_id":"2502.07753","repositories_listed":1,"syntology":null},{"url":"/paper/magic-1-for-1-generating-one-minute-video","slug":"magic-1-for-1-generating-one-minute-video","title":"Magic 1-For-1: Generating One Minute Video Clips within One Minute","date":"2025-02-11","arxiv_id":"2502.07701","repositories_listed":1,"syntology":null},{"url":"/paper/ruscode-russian-cultural-code-benchmark-for","slug":"ruscode-russian-cultural-code-benchmark-for","title":"RusCode: Russian Cultural Code Benchmark for Text-to-Image Generation","date":"2025-02-11","arxiv_id":"2502.07455","repositories_listed":1,"syntology":null},{"url":"/paper/sketchflex-facilitating-spatial-semantic","slug":"sketchflex-facilitating-spatial-semantic","title":"SketchFlex: Facilitating Spatial-Semantic Coherence in Text-to-Image Generation with Region-Based Sketches","date":"2025-02-11","arxiv_id":"2502.07556","repositories_listed":1,"syntology":null},{"url":"/paper/self-correcting-decoding-with-generative","slug":"self-correcting-decoding-with-generative","title":"Self-Correcting Decoding with Generative Feedback for Mitigating Hallucinations in Large Vision-Language Models","date":"2025-02-10","arxiv_id":"2502.06130","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/self-correcting-decoding-with-generative#ran","syntology_url":"https://syntology.ai/paper/2502.06130","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06130"}},"official":{"repos":["zhangce01/degf"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/beyond-fine-tuning-a-systematic-study-of","slug":"beyond-fine-tuning-a-systematic-study-of","title":"Beyond Fine-Tuning: A Systematic Study of Sampling Techniques in Personalized Image Generation","date":"2025-02-09","arxiv_id":"2502.05895","repositories_listed":1,"syntology":null},{"url":"/paper/unicms-a-unified-consistency-model-for","slug":"unicms-a-unified-consistency-model-for","title":"UniCMs: A Unified Consistency Model For Efficient Multimodal Generation and Understanding","date":"2025-02-08","arxiv_id":"2502.05415","repositories_listed":1,"syntology":null},{"url":"/paper/goku-flow-based-video-generative-foundation","slug":"goku-flow-based-video-generative-foundation","title":"Goku: Flow Based Video Generative Foundation Models","date":"2025-02-07","arxiv_id":"2502.04896","repositories_listed":1,"syntology":null},{"url":"/paper/llms-can-see-and-hear-without-any-training","slug":"llms-can-see-and-hear-without-any-training","title":"LLMs can see and hear without any training","date":"2025-01-30","arxiv_id":"2501.18096","repositories_listed":1,"syntology":null},{"url":"/paper/sana-1-5-efficient-scaling-of-training-time","slug":"sana-1-5-efficient-scaling-of-training-time","title":"SANA 1.5: Efficient Scaling of Training-Time and Inference-Time Compute in Linear Diffusion Transformer","date":"2025-01-30","arxiv_id":"2501.18427","repositories_listed":1,"syntology":null},{"url":"/paper/janus-pro-unified-multimodal-understanding","slug":"janus-pro-unified-multimodal-understanding","title":"Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling","date":"2025-01-29","arxiv_id":"2501.17811","repositories_listed":1,"syntology":null},{"url":"/paper/bringing-characters-to-new-stories-training","slug":"bringing-characters-to-new-stories-training","title":"IP-Prompter: Training-Free Theme-Specific Image Generation via Dynamic Visual Prompting","date":"2025-01-26","arxiv_id":"2501.15641","repositories_listed":1,"syntology":null},{"url":"/paper/one-prompt-one-story-free-lunch-consistent","slug":"one-prompt-one-story-free-lunch-consistent","title":"One-Prompt-One-Story: Free-Lunch Consistent Text-to-Image Generation Using a Single Prompt","date":"2025-01-23","arxiv_id":"2501.13554","repositories_listed":1,"syntology":null},{"url":"/paper/anystory-towards-unified-single-and-multiple","slug":"anystory-towards-unified-single-and-multiple","title":"AnyStory: Towards Unified Single and Multiple Subject Personalization in Text-to-Image Generation","date":"2025-01-16","arxiv_id":"2501.09503","repositories_listed":1,"syntology":null},{"url":"/paper/religious-bias-landscape-in-language-and-text","slug":"religious-bias-landscape-in-language-and-text","title":"Religious Bias Landscape in Language and Text-to-Image Models: Analysis, Detection, and Debiasing Strategies","date":"2025-01-14","arxiv_id":"2501.08441","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-text-to-image-generation-via","slug":"boosting-text-to-image-generation-via","title":"Boosting Text-To-Image Generation via Multilingual Prompting in Large Multimodal Models","date":"2025-01-13","arxiv_id":"2501.07086","repositories_listed":1,"syntology":null},{"url":"/paper/poetry-in-pixels-prompt-tuning-for-poem-image","slug":"poetry-in-pixels-prompt-tuning-for-poem-image","title":"Poetry in Pixels: Prompt Tuning for Poem Image Generation via Diffusion Models","date":"2025-01-10","arxiv_id":"2501.05839","repositories_listed":1,"syntology":null},{"url":"/paper/3dis-flux-simple-and-efficient-multi-instance","slug":"3dis-flux-simple-and-efficient-multi-instance","title":"3DIS-FLUX: simple and efficient multi-instance generation with DiT rendering","date":"2025-01-09","arxiv_id":"2501.05131","repositories_listed":1,"syntology":null},{"url":"/paper/face-makeup-multimodal-facial-prompts-for","slug":"face-makeup-multimodal-facial-prompts-for","title":"Face-MakeUp: Multimodal Facial Prompts for Text-to-Image Generation","date":"2025-01-05","arxiv_id":"2501.02523","repositories_listed":1,"syntology":null},{"url":"/paper/eligen-entity-level-controlled-image","slug":"eligen-entity-level-controlled-image","title":"EliGen: Entity-Level Controlled Image Generation with Regional Attention","date":"2025-01-02","arxiv_id":"2501.01097","repositories_listed":1,"syntology":null},{"url":"/paper/dual-diffusion-for-unified-image-generation","slug":"dual-diffusion-for-unified-image-generation","title":"Dual Diffusion for Unified Image Generation and Understanding","date":"2024-12-31","arxiv_id":"2501.00289","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dual-diffusion-for-unified-image-generation#ran","syntology_url":"https://syntology.ai/paper/2501.00289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.00289"}},"official":null}},{"url":"/paper/token-pruning-for-caching-better-9-times","slug":"token-pruning-for-caching-better-9-times","title":"Token Pruning for Caching Better: 9 Times Acceleration on Stable Diffusion for Free","date":"2024-12-31","arxiv_id":"2501.00375","repositories_listed":1,"syntology":null},{"url":"/paper/vmix-improving-text-to-image-diffusion-model","slug":"vmix-improving-text-to-image-diffusion-model","title":"VMix: Improving Text-to-Image Diffusion Model with Cross-Attention Mixing Control","date":"2024-12-30","arxiv_id":"2412.20800","repositories_listed":1,"syntology":null},{"url":"/paper/evalmuse-40k-a-reliable-and-fine-grained","slug":"evalmuse-40k-a-reliable-and-fine-grained","title":"EvalMuse-40K: A Reliable and Fine-Grained Benchmark with Comprehensive Human Annotations for Text-to-Image Generation Model Evaluation","date":"2024-12-24","arxiv_id":"2412.18150","repositories_listed":1,"syntology":null},{"url":"/paper/extract-free-dense-misalignment-from-clip","slug":"extract-free-dense-misalignment-from-clip","title":"Extract Free Dense Misalignment from CLIP","date":"2024-12-24","arxiv_id":"2412.18404","repositories_listed":1,"syntology":null},{"url":"/paper/distilled-decoding-1-one-step-sampling-of","slug":"distilled-decoding-1-one-step-sampling-of","title":"Distilled Decoding 1: One-step Sampling of Image Auto-regressive Models with Flow Matching","date":"2024-12-22","arxiv_id":"2412.17153","repositories_listed":1,"syntology":null},{"url":"/paper/autoregressive-video-generation-without","slug":"autoregressive-video-generation-without","title":"Autoregressive Video Generation without Vector Quantization","date":"2024-12-18","arxiv_id":"2412.14169","repositories_listed":1,"syntology":{"n":26,"n_ran":19,"n_constructed":18,"n_ran_checked":19,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":19,"n_pointer_only":0,"phrase":"19 ran (of which 18 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/autoregressive-video-generation-without#ran","syntology_url":"https://syntology.ai/paper/2412.14169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14169"}},"official":{"repos":["baaivision/nova"],"state":"official (archive's flag): 19 ran","n_ran":19,"n_constructed":18,"n_ran_no_instrument_failure":19,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/artaug-enhancing-text-to-image-generation","slug":"artaug-enhancing-text-to-image-generation","title":"ArtAug: Enhancing Text-to-Image Generation through Synthesis-Understanding Interaction","date":"2024-12-17","arxiv_id":"2412.12888","repositories_listed":1,"syntology":null},{"url":"/paper/fast-prompt-alignment-for-text-to-image","slug":"fast-prompt-alignment-for-text-to-image","title":"Fast Prompt Alignment for Text-to-Image Generation","date":"2024-12-11","arxiv_id":"2412.08639","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fast-prompt-alignment-for-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2412.08639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08639"}},"official":{"repos":["tiktok/fast_prompt_alignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/proactive-agents-for-multi-turn-text-to-image","slug":"proactive-agents-for-multi-turn-text-to-image","title":"Proactive Agents for Multi-Turn Text-to-Image Generation Under Uncertainty","date":"2024-12-09","arxiv_id":"2412.06771","repositories_listed":1,"syntology":null},{"url":"/paper/flexdit-dynamic-token-density-control-for","slug":"flexdit-dynamic-token-density-control-for","title":"FlexDiT: Dynamic Token Density Control for Diffusion Transformer","date":"2024-12-08","arxiv_id":"2412.06028","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/flexdit-dynamic-token-density-control-for#ran","syntology_url":"https://syntology.ai/paper/2412.06028","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06028"}},"official":{"repos":["changsn/FlexDiT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/safeguarding-text-to-image-generation-via","slug":"safeguarding-text-to-image-generation-via","title":"Safeguarding Text-to-Image Generation via Inference-Time Prompt-Noise Optimization","date":"2024-12-05","arxiv_id":"2412.03876","repositories_listed":1,"syntology":null},{"url":"/paper/scimage-how-good-are-multimodal-large","slug":"scimage-how-good-are-multimodal-large","title":"ScImage: How Good Are Multimodal Large Language Models at Scientific Text-to-Image Generation?","date":"2024-12-03","arxiv_id":"2412.02368","repositories_listed":1,"syntology":null},{"url":"/paper/mftf-mask-free-training-free-object-level","slug":"mftf-mask-free-training-free-object-level","title":"MFTF: Mask-free Training-free Object Level Layout Control Diffusion Model","date":"2024-12-02","arxiv_id":"2412.01284","repositories_listed":1,"syntology":null},{"url":"/paper/x-prompt-towards-universal-in-context-image","slug":"x-prompt-towards-universal-in-context-image","title":"X-Prompt: Towards Universal In-Context Image Generation in Auto-Regressive Vision Language Foundation Models","date":"2024-12-02","arxiv_id":"2412.01824","repositories_listed":1,"syntology":null},{"url":"/paper/playable-game-generation","slug":"playable-game-generation","title":"Playable Game Generation","date":"2024-12-01","arxiv_id":"2412.00887","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/playable-game-generation#ran","syntology_url":"https://syntology.ai/paper/2412.00887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.00887"}},"official":{"repos":["greatx3/playable-game-generation"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/jetformer-an-autoregressive-generative-model","slug":"jetformer-an-autoregressive-generative-model","title":"JetFormer: An Autoregressive Generative Model of Raw Images and Text","date":"2024-11-29","arxiv_id":"2411.19722","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/jetformer-an-autoregressive-generative-model#ran","syntology_url":"https://syntology.ai/paper/2411.19722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19722"}},"official":{"repos":["google-research/big_vision"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/amo-sampler-enhancing-text-rendering-with","slug":"amo-sampler-enhancing-text-rendering-with","title":"AMO Sampler: Enhancing Text Rendering with Overshooting","date":"2024-11-28","arxiv_id":"2411.19415","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-evaluation-for-text-to-image","slug":"automatic-evaluation-for-text-to-image","title":"Automatic Evaluation for Text-to-image Generation: Task-decomposed Framework, Distilled Training, and Meta-evaluation Benchmark","date":"2024-11-23","arxiv_id":"2411.15488","repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-text-to-image-model-with","slug":"large-scale-text-to-image-model-with","title":"Large-Scale Text-to-Image Model with Inpainting is a Zero-Shot Subject-Driven Image Generator","date":"2024-11-23","arxiv_id":"2411.15466","repositories_listed":1,"syntology":null},{"url":"/paper/what-makes-a-scene-scene-graph-based","slug":"what-makes-a-scene-scene-graph-based","title":"What Makes a Scene ? Scene Graph-based Evaluation and Feedback for Controllable Generation","date":"2024-11-23","arxiv_id":"2411.15435","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-human-artifacts-from-text-to-image","slug":"detecting-human-artifacts-from-text-to-image","title":"Detecting Human Artifacts from Text-to-Image Models","date":"2024-11-21","arxiv_id":"2411.13842","repositories_listed":1,"syntology":null},{"url":"/paper/mmgenbench-evaluating-the-limits-of-lmms-from","slug":"mmgenbench-evaluating-the-limits-of-lmms-from","title":"MMGenBench: Evaluating the Limits of LMMs from the Text-to-Image Generation Perspective","date":"2024-11-21","arxiv_id":"2411.14062","repositories_listed":1,"syntology":null},{"url":"/paper/region-aware-text-to-image-generation-via","slug":"region-aware-text-to-image-generation-via","title":"Region-Aware Text-to-Image Generation via Hard Binding and Soft Refinement","date":"2024-11-10","arxiv_id":"2411.06558","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/region-aware-text-to-image-generation-via#ran","syntology_url":"https://syntology.ai/paper/2411.06558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.06558"}},"official":{"repos":["nju-pcalab/rag-diffusion"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/precision-or-recall-an-analysis-of-image","slug":"precision-or-recall-an-analysis-of-image","title":"Precision or Recall? An Analysis of Image Captions for Training Text-to-Image Generation Model","date":"2024-11-07","arxiv_id":"2411.05079","repositories_listed":1,"syntology":null},{"url":"/paper/training-free-regional-prompting-for","slug":"training-free-regional-prompting-for","title":"Training-free Regional Prompting for Diffusion Transformers","date":"2024-11-04","arxiv_id":"2411.02395","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-free-regional-prompting-for#ran","syntology_url":"https://syntology.ai/paper/2411.02395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02395"}},"official":{"repos":["instantX-research/Regional-Prompting-FLUX"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/groundit-grounding-diffusion-transformers-via","slug":"groundit-grounding-diffusion-transformers-via","title":"GrounDiT: Grounding Diffusion Transformers via Noisy Patch Transplantation","date":"2024-10-27","arxiv_id":"2410.20474","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/groundit-grounding-diffusion-transformers-via#ran","syntology_url":"https://syntology.ai/paper/2410.20474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20474"}},"official":{"repos":["KAIST-Visual-AI-Group/GrounDiT"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/mmm-rs-a-multi-modal-multi-gsd-multi-scene","slug":"mmm-rs-a-multi-modal-multi-gsd-multi-scene","title":"MMM-RS: A Multi-modal, Multi-GSD, Multi-scene Remote Sensing Dataset and Benchmark for Text-to-Image Generation","date":"2024-10-26","arxiv_id":"2410.22362","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mmm-rs-a-multi-modal-multi-gsd-multi-scene#ran","syntology_url":"https://syntology.ai/paper/2410.22362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22362"}},"official":{"repos":["ljl5261/mmm-rs"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/altogether-image-captioning-via-re-aligning","slug":"altogether-image-captioning-via-re-aligning","title":"Altogether: Image Captioning via Re-aligning Alt-text","date":"2024-10-22","arxiv_id":"2410.17251","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/altogether-image-captioning-via-re-aligning#ran","syntology_url":"https://syntology.ai/paper/2410.17251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17251"}},"official":{"repos":["facebookresearch/metaclip"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/offline-evaluation-of-set-based-text-to-image","slug":"offline-evaluation-of-set-based-text-to-image","title":"Offline Evaluation of Set-Based Text-to-Image Generation","date":"2024-10-22","arxiv_id":"2410.17331","repositories_listed":1,"syntology":null},{"url":"/paper/bigr-harnessing-binary-latent-codes-for-image","slug":"bigr-harnessing-binary-latent-codes-for-image","title":"BiGR: Harnessing Binary Latent Codes for Image Generation and Improved Visual Representation Capabilities","date":"2024-10-18","arxiv_id":"2410.14672","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bigr-harnessing-binary-latent-codes-for-image#ran","syntology_url":"https://syntology.ai/paper/2410.14672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14672"}},"official":{"repos":["haoosz/BiGR"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/boosting-imperceptibility-of-stable-diffusion","slug":"boosting-imperceptibility-of-stable-diffusion","title":"Boosting Imperceptibility of Stable Diffusion-based Adversarial Examples Generation with Momentum","date":"2024-10-17","arxiv_id":"2410.13122","repositories_listed":1,"syntology":null},{"url":"/paper/fluid-scaling-autoregressive-text-to-image","slug":"fluid-scaling-autoregressive-text-to-image","title":"Fluid: Scaling Autoregressive Text-to-image Generative Models with Continuous Tokens","date":"2024-10-17","arxiv_id":"2410.13863","repositories_listed":1,"syntology":null},{"url":"/paper/puma-empowering-unified-mllm-with-multi","slug":"puma-empowering-unified-mllm-with-multi","title":"PUMA: Empowering Unified MLLM with Multi-granular Visual Generation","date":"2024-10-17","arxiv_id":"2410.13861","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/puma-empowering-unified-mllm-with-multi#ran","syntology_url":"https://syntology.ai/paper/2410.13861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13861"}},"official":{"repos":["rongyaofang/puma"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/3dis-depth-driven-decoupled-instance","slug":"3dis-depth-driven-decoupled-instance","title":"3DIS: Depth-Driven Decoupled Instance Synthesis for Text-to-Image Generation","date":"2024-10-16","arxiv_id":"2410.12669","repositories_listed":1,"syntology":null},{"url":"/paper/facechain-fact-face-adapter-with-decoupled","slug":"facechain-fact-face-adapter-with-decoupled","title":"FaceChain-FACT: Face Adapter with Decoupled Training for Identity-preserved Personalization","date":"2024-10-16","arxiv_id":"2410.12312","repositories_listed":1,"syntology":null},{"url":"/paper/intermediate-representations-for-enhanced","slug":"intermediate-representations-for-enhanced","title":"Generating Intermediate Representations for Compositional Text-To-Image Generation","date":"2024-10-13","arxiv_id":"2410.09792","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intermediate-representations-for-enhanced#ran","syntology_url":"https://syntology.ai/paper/2410.09792","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09792"}},"official":{"repos":["rang1991/public-intermediate-semantics-for-generation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tulip-token-length-upgraded-clip","slug":"tulip-token-length-upgraded-clip","title":"TULIP: Token-length Upgraded CLIP","date":"2024-10-13","arxiv_id":"2410.10034","repositories_listed":1,"syntology":{"n":20,"n_ran":13,"n_constructed":0,"n_ran_checked":9,"n_instrument":4,"n_unverified":7,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":7,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/tulip-token-length-upgraded-clip#ran","syntology_url":"https://syntology.ai/paper/2410.10034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10034"}},"official":{"repos":["ivonajdenkoska/tulip"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/a-unified-debiasing-approach-for-vision","slug":"a-unified-debiasing-approach-for-vision","title":"A Unified Debiasing Approach for Vision-Language Models across Modalities and Tasks","date":"2024-10-10","arxiv_id":"2410.07593","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/a-unified-debiasing-approach-for-vision#ran","syntology_url":"https://syntology.ai/paper/2410.07593","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07593"}},"official":{"repos":["HoinJung/Unified-Debiaisng-VLM-SFID"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/minorityprompt-text-to-minority-image","slug":"minorityprompt-text-to-minority-image","title":"Minority-Focused Text-to-Image Generation via Prompt Optimization","date":"2024-10-10","arxiv_id":"2410.07838","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/minorityprompt-text-to-minority-image#ran","syntology_url":"https://syntology.ai/paper/2410.07838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07838"}},"official":{"repos":["anonymous5293/minorityprompt"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/evolvedirector-approaching-advanced-text-to","slug":"evolvedirector-approaching-advanced-text-to","title":"EvolveDirector: Approaching Advanced Text-to-Image Generation with Large Vision-Language Models","date":"2024-10-09","arxiv_id":"2410.07133","repositories_listed":1,"syntology":null},{"url":"/paper/training-free-diffusion-model-alignment-with","slug":"training-free-diffusion-model-alignment-with","title":"Training-free Diffusion Model Alignment with Sampling Demons","date":"2024-10-08","arxiv_id":"2410.05760","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/training-free-diffusion-model-alignment-with#ran","syntology_url":"https://syntology.ai/paper/2410.05760","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05760"}},"official":{"repos":["aiiu-lab/DemonSampling"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/images-speak-volumes-user-centric-assessment","slug":"images-speak-volumes-user-centric-assessment","title":"Images Speak Volumes: User-Centric Assessment of Image Generation for Accessible Communication","date":"2024-10-04","arxiv_id":"2410.03430","repositories_listed":1,"syntology":null},{"url":"/paper/data-extrapolation-for-text-to-image","slug":"data-extrapolation-for-text-to-image","title":"Data Extrapolation for Text-to-image Generation on Small Datasets","date":"2024-10-02","arxiv_id":"2410.01638","repositories_listed":1,"syntology":null},{"url":"/paper/cusconcept-customized-visual-concept","slug":"cusconcept-customized-visual-concept","title":"CusConcept: Customized Visual Concept Decomposition with Diffusion Models","date":"2024-10-01","arxiv_id":"2410.00398","repositories_listed":1,"syntology":null},{"url":"/paper/flowturbo-towards-real-time-flow-based-image","slug":"flowturbo-towards-real-time-flow-based-image","title":"FlowTurbo: Towards Real-time Flow-Based Image Generation with Velocity Refiner","date":"2024-09-26","arxiv_id":"2409.18128","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/flowturbo-towards-real-time-flow-based-image#ran","syntology_url":"https://syntology.ai/paper/2409.18128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.18128"}},"official":{"repos":["shiml20/flowturbo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/resolving-multi-condition-confusion-for","slug":"resolving-multi-condition-confusion-for","title":"Resolving Multi-Condition Confusion for Finetuning-Free Personalized Image Generation","date":"2024-09-26","arxiv_id":"2409.17920","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/resolving-multi-condition-confusion-for#ran","syntology_url":"https://syntology.ai/paper/2409.17920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.17920"}},"official":{"repos":["hqhqaq/mip-adapter"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/pixwizard-versatile-image-to-image-visual","slug":"pixwizard-versatile-image-to-image-visual","title":"PixWizard: Versatile Image-to-Image Visual Assistant with Open-Language Instructions","date":"2024-09-23","arxiv_id":"2409.15278","repositories_listed":1,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":11,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":2,"n_no_contract":9,"n_pointer_only":1,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 2 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pixwizard-versatile-image-to-image-visual#ran","syntology_url":"https://syntology.ai/paper/2409.15278","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.15278"}},"official":{"repos":["afeng-x/pixwizard"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"42815c6e3541a350699ab0f5fffdb53ff83e518297ac003c14f8e22e4492b988","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}