{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-generation/papers/2","list_of":"/task/video-generation","task":"Video Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":15,"rows_per_page":100,"rows":[101,200],"of":1466,"counts":{"archive_papers_tagged":1466,"with_a_code_link":609,"where_syntology_ran_a_sample":257,"not_listed_spam_title":0,"listed":1466,"listed_where_code_ran":257,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":221,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":221,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-generation","prev":"/task/video-generation","next":"/task/video-generation/papers/3","papers":[{"url":"/paper/mmgt-motion-mask-guided-two-stage-network-for","slug":"mmgt-motion-mask-guided-two-stage-network-for","title":"MMGT: Motion Mask Guided Two-Stage Network for Co-Speech Gesture Video Generation","date":"2025-05-29","arxiv_id":"2505.23120","repositories_listed":1,"syntology":null},{"url":"/paper/vcapsbench-a-large-scale-fine-grained","slug":"vcapsbench-a-large-scale-fine-grained","title":"VCapsBench: A Large-scale Fine-grained Benchmark for Video Caption Quality Evaluation","date":"2025-05-29","arxiv_id":"2505.23484","repositories_listed":1,"syntology":null},{"url":"/paper/vf-eval-evaluating-multimodal-llms-for","slug":"vf-eval-evaluating-multimodal-llms-for","title":"VF-Eval: Evaluating Multimodal LLMs for Generating Feedback on AIGC Videos","date":"2025-05-29","arxiv_id":"2505.23693","repositories_listed":1,"syntology":null},{"url":"/paper/videorepa-learning-physics-for-video","slug":"videorepa-learning-physics-for-video","title":"VideoREPA: Learning Physics for Video Generation through Relational Alignment with Foundation Models","date":"2025-05-29","arxiv_id":"2505.23656","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videorepa-learning-physics-for-video#ran","syntology_url":"https://syntology.ai/paper/2505.23656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23656"}},"official":{"repos":["aHapBean/VideoREPA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/let-them-talk-audio-driven-multi-person","slug":"let-them-talk-audio-driven-multi-person","title":"Let Them Talk: Audio-Driven Multi-Person Conversational Video Generation","date":"2025-05-28","arxiv_id":"2505.22647","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/let-them-talk-audio-driven-multi-person#ran","syntology_url":"https://syntology.ai/paper/2505.22647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22647"}},"official":{"repos":["meigen-ai/multitalk"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/minute-long-videos-with-dual-parallelisms","slug":"minute-long-videos-with-dual-parallelisms","title":"Minute-Long Videos with Dual Parallelisms","date":"2025-05-27","arxiv_id":"2505.21070","repositories_listed":1,"syntology":null},{"url":"/paper/drivecamsim-generalizable-camera-simulation","slug":"drivecamsim-generalizable-camera-simulation","title":"DriveCamSim: Generalizable Camera Simulation via Explicit Camera Modeling for Autonomous Driving","date":"2025-05-26","arxiv_id":"2505.19692","repositories_listed":1,"syntology":null},{"url":"/paper/dvd-quant-data-free-video-diffusion","slug":"dvd-quant-data-free-video-diffusion","title":"DVD-Quant: Data-free Video Diffusion Transformers Quantization","date":"2025-05-24","arxiv_id":"2505.18663","repositories_listed":1,"syntology":null},{"url":"/paper/vorta-efficient-video-diffusion-via-routing","slug":"vorta-efficient-video-diffusion-via-routing","title":"VORTA: Efficient Video Diffusion via Routing Sparse Attention","date":"2025-05-24","arxiv_id":"2505.18809","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vorta-efficient-video-diffusion-via-routing#ran","syntology_url":"https://syntology.ai/paper/2505.18809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18809"}},"official":{"repos":["wenhao728/vorta"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inflvg-reinforce-inference-time-consistent","slug":"inflvg-reinforce-inference-time-consistent","title":"InfLVG: Reinforce Inference-Time Consistent Long Video Generation with GRPO","date":"2025-05-23","arxiv_id":"2505.17574","repositories_listed":1,"syntology":null},{"url":"/paper/training-free-efficient-video-generation-via","slug":"training-free-efficient-video-generation-via","title":"Training-Free Efficient Video Generation via Dynamic Token Carving","date":"2025-05-22","arxiv_id":"2505.16864","repositories_listed":1,"syntology":null},{"url":"/paper/avatarshield-visual-reinforcement-learning","slug":"avatarshield-visual-reinforcement-learning","title":"AvatarShield: Visual Reinforcement Learning for Human-Centric Video Forgery Detection","date":"2025-05-21","arxiv_id":"2505.15173","repositories_listed":1,"syntology":null},{"url":"/paper/cinetechbench-a-benchmark-for-cinematographic","slug":"cinetechbench-a-benchmark-for-cinematographic","title":"CineTechBench: A Benchmark for Cinematographic Technique Understanding and Generation","date":"2025-05-21","arxiv_id":"2505.15145","repositories_listed":1,"syntology":null},{"url":"/paper/grouping-first-attending-smartly-training","slug":"grouping-first-attending-smartly-training","title":"Grouping First, Attending Smartly: Training-Free Acceleration for Diffusion Transformers","date":"2025-05-20","arxiv_id":"2505.14687","repositories_listed":1,"syntology":null},{"url":"/paper/programmatic-video-prediction-using-large","slug":"programmatic-video-prediction-using-large","title":"Programmatic Video Prediction Using Large Language Models","date":"2025-05-20","arxiv_id":"2505.14948","repositories_listed":1,"syntology":null},{"url":"/paper/busterx-mllm-powered-ai-generated-video","slug":"busterx-mllm-powered-ai-generated-video","title":"BusterX: MLLM-Powered AI-Generated Video Forgery Detection and Explanation","date":"2025-05-19","arxiv_id":"2505.12620","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/busterx-mllm-powered-ai-generated-video#ran","syntology_url":"https://syntology.ai/paper/2505.12620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12620"}},"official":{"repos":["l8cv/busterx"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dreamgen-unlocking-generalization-in-robot","slug":"dreamgen-unlocking-generalization-in-robot","title":"DreamGen: Unlocking Generalization in Robot Learning through Video World Models","date":"2025-05-19","arxiv_id":"2505.12705","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dreamgen-unlocking-generalization-in-robot#ran","syntology_url":"https://syntology.ai/paper/2505.12705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12705"}},"official":{"repos":["nvidia/gr00t-dreams"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-gpt-via-next-clip-diffusion","slug":"video-gpt-via-next-clip-diffusion","title":"Video-GPT via Next Clip Diffusion","date":"2025-05-18","arxiv_id":"2505.12489","repositories_listed":1,"syntology":null},{"url":"/paper/draftattention-fast-video-diffusion-via-low","slug":"draftattention-fast-video-diffusion-via-low","title":"DraftAttention: Fast Video Diffusion via Low-Resolution Attention Guidance","date":"2025-05-17","arxiv_id":"2505.14708","repositories_listed":1,"syntology":null},{"url":"/paper/fastcar-cache-attentive-replay-for-fast-auto","slug":"fastcar-cache-attentive-replay-for-fast-auto","title":"FastCar: Cache Attentive Replay for Fast Auto-Regressive Video Generation on the Edge","date":"2025-05-17","arxiv_id":"2505.14709","repositories_listed":1,"syntology":null},{"url":"/paper/love-benchmarking-and-evaluating-text-to","slug":"love-benchmarking-and-evaluating-text-to","title":"LOVE: Benchmarking and Evaluating Text-to-Video Generation and Video-to-Text Interpretation","date":"2025-05-17","arxiv_id":"2505.12098","repositories_listed":1,"syntology":null},{"url":"/paper/mtvcrafter-4d-motion-tokenization-for-open","slug":"mtvcrafter-4d-motion-tokenization-for-open","title":"MTVCrafter: 4D Motion Tokenization for Open-World Human Image Animation","date":"2025-05-15","arxiv_id":"2505.10238","repositories_listed":1,"syntology":null},{"url":"/paper/generative-ai-for-autonomous-driving","slug":"generative-ai-for-autonomous-driving","title":"Generative AI for Autonomous Driving: Frontiers and Opportunities","date":"2025-05-13","arxiv_id":"2505.08854","repositories_listed":1,"syntology":null},{"url":"/paper/dancegrpo-unleashing-grpo-on-visual","slug":"dancegrpo-unleashing-grpo-on-visual","title":"DanceGRPO: Unleashing GRPO on Visual Generation","date":"2025-05-12","arxiv_id":"2505.07818","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dancegrpo-unleashing-grpo-on-visual#ran","syntology_url":"https://syntology.ai/paper/2505.07818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07818"}},"official":null}},{"url":"/paper/ophora-a-large-scale-data-driven-text-guided","slug":"ophora-a-large-scale-data-driven-text-guided","title":"Ophora: A Large-Scale Data-Driven Text-Guided Ophthalmic Surgical Video Generation Model","date":"2025-05-12","arxiv_id":"2505.07449","repositories_listed":1,"syntology":null},{"url":"/paper/hunyuancustom-a-multimodal-driven","slug":"hunyuancustom-a-multimodal-driven","title":"HunyuanCustom: A Multimodal-Driven Architecture for Customized Video Generation","date":"2025-05-07","arxiv_id":"2505.04512","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hunyuancustom-a-multimodal-driven#ran","syntology_url":"https://syntology.ai/paper/2505.04512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.04512"}},"official":null}},{"url":"/paper/real-time-person-image-synthesis-using-a-flow","slug":"real-time-person-image-synthesis-using-a-flow","title":"Real-Time Person Image Synthesis Using a Flow Matching Model","date":"2025-05-06","arxiv_id":"2505.03562","repositories_listed":1,"syntology":null},{"url":"/paper/freepca-integrating-consistency-information","slug":"freepca-integrating-consistency-information","title":"FreePCA: Integrating Consistency Information across Long-short Frames in Training-free Long Video Generation via Principal Component Analysis","date":"2025-05-02","arxiv_id":"2505.01172","repositories_listed":1,"syntology":null},{"url":"/paper/videohallu-evaluating-and-mitigating-multi","slug":"videohallu-evaluating-and-mitigating-multi","title":"VideoHallu: Evaluating and Mitigating Multi-modal Hallucinations on Synthetic Video Understanding","date":"2025-05-02","arxiv_id":"2505.01481","repositories_listed":1,"syntology":null},{"url":"/paper/holotime-taming-video-diffusion-models-for","slug":"holotime-taming-video-diffusion-models-for","title":"HoloTime: Taming Video Diffusion Models for Panoramic 4D Scene Generation","date":"2025-04-30","arxiv_id":"2504.21650","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/holotime-taming-video-diffusion-models-for#ran","syntology_url":"https://syntology.ai/paper/2504.21650","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.21650"}},"official":{"repos":["pku-yuangroup/holotime"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/survey-of-video-diffusion-models-foundations","slug":"survey-of-video-diffusion-models-foundations","title":"Survey of Video Diffusion Models: Foundations, Implementations, and Applications","date":"2025-04-22","arxiv_id":"2504.16081","repositories_listed":1,"syntology":null},{"url":"/paper/uni3c-unifying-precisely-3d-enhanced-camera","slug":"uni3c-unifying-precisely-3d-enhanced-camera","title":"Uni3C: Unifying Precisely 3D-Enhanced Camera and Human Motion Controls for Video Generation","date":"2025-04-21","arxiv_id":"2504.14899","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/uni3c-unifying-precisely-3d-enhanced-camera#ran","syntology_url":"https://syntology.ai/paper/2504.14899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.14899"}},"official":{"repos":["ewrfcas/uni3c"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/spherediff-tuning-free-omnidirectional","slug":"spherediff-tuning-free-omnidirectional","title":"SphereDiff: Tuning-free Omnidirectional Panoramic Image and Video Generation via Spherical Latent Representation","date":"2025-04-19","arxiv_id":"2504.14396","repositories_listed":1,"syntology":null},{"url":"/paper/skyreels-v2-infinite-length-film-generative","slug":"skyreels-v2-infinite-length-film-generative","title":"SkyReels-V2: Infinite-length Film Generative Model","date":"2025-04-17","arxiv_id":"2504.13074","repositories_listed":1,"syntology":null},{"url":"/paper/vgdfr-diffusion-based-video-generation-with","slug":"vgdfr-diffusion-based-video-generation-with","title":"VGDFR: Diffusion-based Video Generation with Dynamic Latent Frame Rate","date":"2025-04-16","arxiv_id":"2504.12259","repositories_listed":1,"syntology":null},{"url":"/paper/aligning-anime-video-generation-with-human","slug":"aligning-anime-video-generation-with-human","title":"Aligning Anime Video Generation with Human Feedback","date":"2025-04-14","arxiv_id":"2504.10044","repositories_listed":1,"syntology":null},{"url":"/paper/h-more-learning-human-centric-motion","slug":"h-more-learning-human-centric-motion","title":"H-MoRe: Learning Human-centric Motion Representation for Action Analysis","date":"2025-04-14","arxiv_id":"2504.10676","repositories_listed":1,"syntology":null},{"url":"/paper/realcam-vid-high-resolution-video-dataset","slug":"realcam-vid-high-resolution-video-dataset","title":"RealCam-Vid: High-resolution Video Dataset with Dynamic Scenes and Metric-scale Camera Movements","date":"2025-04-11","arxiv_id":"2504.08212","repositories_listed":1,"syntology":null},{"url":"/paper/diffusion-transformers-for-tabular-data-time","slug":"diffusion-transformers-for-tabular-data-time","title":"Diffusion Transformers for Tabular Data Time Series Generation","date":"2025-04-10","arxiv_id":"2504.07566","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/diffusion-transformers-for-tabular-data-time#ran","syntology_url":"https://syntology.ai/paper/2504.07566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07566"}},"official":{"repos":["fabriziogaruti/TabDiT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/dydit-dynamic-diffusion-transformers-for","slug":"dydit-dynamic-diffusion-transformers-for","title":"DyDiT++: Dynamic Diffusion Transformers for Efficient Visual Generation","date":"2025-04-09","arxiv_id":"2504.06803","repositories_listed":1,"syntology":null},{"url":"/paper/camcontexti2v-context-aware-controllable","slug":"camcontexti2v-context-aware-controllable","title":"CamContextI2V: Context-aware Controllable Video Generation","date":"2025-04-08","arxiv_id":"2504.06022","repositories_listed":1,"syntology":null},{"url":"/paper/video4dgen-enhancing-video-and-4d-generation","slug":"video4dgen-enhancing-video-and-4d-generation","title":"Video4DGen: Enhancing Video and 4D Generation through Mutual Optimization","date":"2025-04-05","arxiv_id":"2504.04153","repositories_listed":1,"syntology":null},{"url":"/paper/model-reveals-what-to-cache-profiling-based","slug":"model-reveals-what-to-cache-profiling-based","title":"Model Reveals What to Cache: Profiling-Based Feature Reuse for Video Diffusion Models","date":"2025-04-04","arxiv_id":"2504.03140","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/model-reveals-what-to-cache-profiling-based#ran","syntology_url":"https://syntology.ai/paper/2504.03140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.03140"}},"official":{"repos":["geekguru123/profilingdit"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/audio-visual-controlled-video-diffusion-with","slug":"audio-visual-controlled-video-diffusion-with","title":"Audio-visual Controlled Video Diffusion with Masked Selective State Spaces Modeling for Natural Talking Head Generation","date":"2025-04-03","arxiv_id":"2504.02542","repositories_listed":1,"syntology":null},{"url":"/paper/conmo-controllable-motion-disentanglement-and","slug":"conmo-controllable-motion-disentanglement-and","title":"ConMo: Controllable Motion Disentanglement and Recomposition for Zero-Shot Motion Transfer","date":"2025-04-03","arxiv_id":"2504.02451","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/conmo-controllable-motion-disentanglement-and#ran","syntology_url":"https://syntology.ai/paper/2504.02451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02451"}},"official":{"repos":["andyplus1/conmo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/skyreels-a2-compose-anything-in-video","slug":"skyreels-a2-compose-anything-in-video","title":"SkyReels-A2: Compose Anything in Video Diffusion Transformers","date":"2025-04-03","arxiv_id":"2504.02436","repositories_listed":1,"syntology":null},{"url":"/paper/on-device-sora-enabling-training-free","slug":"on-device-sora-enabling-training-free","title":"On-device Sora: Enabling Training-Free Diffusion-based Text-to-Video Generation for Mobile Devices","date":"2025-03-31","arxiv_id":"2503.23796","repositories_listed":1,"syntology":null},{"url":"/paper/videogen-eval-agent-based-system-for-video","slug":"videogen-eval-agent-based-system-for-video","title":"VideoGen-Eval: Agent-based System for Video Generation Evaluation","date":"2025-03-30","arxiv_id":"2503.23452","repositories_listed":1,"syntology":null},{"url":"/paper/dynamictrl-rethinking-the-basic-structure-and","slug":"dynamictrl-rethinking-the-basic-structure-and","title":"DynamiCtrl: Rethinking the Basic Structure and the Role of Text for High-quality Human Image Animation","date":"2025-03-27","arxiv_id":"2503.21246","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-evolution-of-physics-cognition","slug":"exploring-the-evolution-of-physics-cognition","title":"Exploring the Evolution of Physics Cognition in Video Generation: A Survey","date":"2025-03-27","arxiv_id":"2503.21765","repositories_listed":1,"syntology":null},{"url":"/paper/vbench-2-0-advancing-video-generation","slug":"vbench-2-0-advancing-video-generation","title":"VBench-2.0: Advancing Video Generation Benchmark Suite for Intrinsic Faithfulness","date":"2025-03-27","arxiv_id":"2503.21755","repositories_listed":1,"syntology":null},{"url":"/paper/protecting-your-video-content-disrupting","slug":"protecting-your-video-content-disrupting","title":"Protecting Your Video Content: Disrupting Automated Video-based LLM Annotations","date":"2025-03-26","arxiv_id":"2503.21824","repositories_listed":1,"syntology":null},{"url":"/paper/rectable-fast-modeling-tabular-data-with","slug":"rectable-fast-modeling-tabular-data-with","title":"RecTable: Fast Modeling Tabular Data with Rectified Flow","date":"2025-03-26","arxiv_id":"2503.20731","repositories_listed":1,"syntology":null},{"url":"/paper/vpo-aligning-text-to-video-generation-models","slug":"vpo-aligning-text-to-video-generation-models","title":"VPO: Aligning Text-to-Video Generation Models with Prompt Optimization","date":"2025-03-26","arxiv_id":"2503.20491","repositories_listed":1,"syntology":null},{"url":"/paper/wan-open-and-advanced-large-scale-video","slug":"wan-open-and-advanced-large-scale-video","title":"Wan: Open and Advanced Large-Scale Video Generative Models","date":"2025-03-26","arxiv_id":"2503.20314","repositories_listed":1,"syntology":null},{"url":"/paper/efficientmt-efficient-temporal-adaptation-for","slug":"efficientmt-efficient-temporal-adaptation-for","title":"EfficientMT: Efficient Temporal Adaptation for Motion Transfer in Text-to-Video Diffusion Models","date":"2025-03-25","arxiv_id":"2503.19369","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/efficientmt-efficient-temporal-adaptation-for#ran","syntology_url":"https://syntology.ai/paper/2503.19369","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.19369"}},"official":{"repos":["prototypenx/efficientmt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/long-context-autoregressive-video-modeling-1","slug":"long-context-autoregressive-video-modeling-1","title":"Long-Context Autoregressive Video Modeling with Next-Frame Prediction","date":"2025-03-25","arxiv_id":"2503.19325","repositories_listed":1,"syntology":null},{"url":"/paper/amd-hummingbird-towards-an-efficient-text-to","slug":"amd-hummingbird-towards-an-efficient-text-to","title":"AMD-Hummingbird: Towards an Efficient Text-to-Video Model","date":"2025-03-24","arxiv_id":"2503.18559","repositories_listed":1,"syntology":null},{"url":"/paper/syncvp-joint-diffusion-for-synchronous-multi","slug":"syncvp-joint-diffusion-for-synchronous-multi","title":"SyncVP: Joint Diffusion for Synchronous Multi-Modal Video Prediction","date":"2025-03-24","arxiv_id":"2503.18933","repositories_listed":1,"syntology":null},{"url":"/paper/decouple-and-track-benchmarking-and-improving","slug":"decouple-and-track-benchmarking-and-improving","title":"Decouple and Track: Benchmarking and Improving Video Diffusion Transformers for Motion Transfer","date":"2025-03-21","arxiv_id":"2503.17350","repositories_listed":1,"syntology":null},{"url":"/paper/enabling-versatile-controls-for-video","slug":"enabling-versatile-controls-for-video","title":"Enabling Versatile Controls for Video Diffusion Models","date":"2025-03-21","arxiv_id":"2503.16983","repositories_listed":1,"syntology":null},{"url":"/paper/mila-multi-view-intensive-fidelity-long-term","slug":"mila-multi-view-intensive-fidelity-long-term","title":"MiLA: Multi-view Intensive-fidelity Long-term Video Generation World Model for Autonomous Driving","date":"2025-03-20","arxiv_id":"2503.15875","repositories_listed":1,"syntology":null},{"url":"/paper/xattention-block-sparse-attention-with","slug":"xattention-block-sparse-attention-with","title":"XAttention: Block Sparse Attention with Antidiagonal Scoring","date":"2025-03-20","arxiv_id":"2503.16428","repositories_listed":1,"syntology":null},{"url":"/paper/aigve-tool-ai-generated-video-evaluation","slug":"aigve-tool-ai-generated-video-evaluation","title":"AIGVE-Tool: AI-Generated Video Evaluation Toolkit with Multifaceted Benchmark","date":"2025-03-18","arxiv_id":"2503.14064","repositories_listed":1,"syntology":null},{"url":"/paper/concat-id-towards-universal-identity","slug":"concat-id-towards-universal-identity","title":"Concat-ID: Towards Universal Identity-Preserving Video Synthesis","date":"2025-03-18","arxiv_id":"2503.14151","repositories_listed":1,"syntology":null},{"url":"/paper/leanvae-an-ultra-efficient-reconstruction-vae","slug":"leanvae-an-ultra-efficient-reconstruction-vae","title":"LeanVAE: An Ultra-Efficient Reconstruction VAE for Video Diffusion Models","date":"2025-03-18","arxiv_id":"2503.14325","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/leanvae-an-ultra-efficient-reconstruction-vae#ran","syntology_url":"https://syntology.ai/paper/2503.14325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.14325"}},"official":{"repos":["westlake-repl/leanvae"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/steerx-creating-any-camera-free-3d-and-4d","slug":"steerx-creating-any-camera-free-3d-and-4d","title":"SteerX: Creating Any Camera-Free 3D and 4D Scenes with Geometric Steering","date":"2025-03-15","arxiv_id":"2503.12024","repositories_listed":1,"syntology":null},{"url":"/paper/step-video-ti2v-technical-report-a-state-of","slug":"step-video-ti2v-technical-report-a-state-of","title":"Step-Video-TI2V Technical Report: A State-of-the-Art Text-Driven Image-to-Video Generation Model","date":"2025-03-14","arxiv_id":"2503.11251","repositories_listed":1,"syntology":null},{"url":"/paper/vmbench-a-benchmark-for-perception-aligned","slug":"vmbench-a-benchmark-for-perception-aligned","title":"VMBench: A Benchmark for Perception-Aligned Video Motion Generation","date":"2025-03-13","arxiv_id":"2503.10076","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vmbench-a-benchmark-for-perception-aligned#ran","syntology_url":"https://syntology.ai/paper/2503.10076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10076"}},"official":{"repos":["gd-aigc/vmbench"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/neighboring-autoregressive-modeling-for","slug":"neighboring-autoregressive-modeling-for","title":"Neighboring Autoregressive Modeling for Efficient Visual Generation","date":"2025-03-12","arxiv_id":"2503.10696","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/neighboring-autoregressive-modeling-for#ran","syntology_url":"https://syntology.ai/paper/2503.10696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10696"}},"official":{"repos":["thisisbillhe/nar"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/open-sora-2-0-training-a-commercial-level","slug":"open-sora-2-0-training-a-commercial-level","title":"Open-Sora 2.0: Training a Commercial-Level Video Generation Model in $200k","date":"2025-03-12","arxiv_id":"2503.09642","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/open-sora-2-0-training-a-commercial-level#ran","syntology_url":"https://syntology.ai/paper/2503.09642","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.09642"}},"official":{"repos":["hpcaitech/open-sora"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/pisa-experiments-exploring-physics-post","slug":"pisa-experiments-exploring-physics-post","title":"PISA Experiments: Exploring Physics Post-Training for Video Diffusion Models by Watching Stuff Drop","date":"2025-03-12","arxiv_id":"2503.09595","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pisa-experiments-exploring-physics-post#ran","syntology_url":"https://syntology.ai/paper/2503.09595","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.09595"}},"official":{"repos":["vision-x-nyu/pisa-experiments"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/r-flav-rolling-flow-matching-for-infinite","slug":"r-flav-rolling-flow-matching-for-infinite","title":"$^R$FLAV: Rolling Flow matching for infinite Audio Video generation","date":"2025-03-11","arxiv_id":"2503.08307","repositories_listed":1,"syntology":null},{"url":"/paper/vrmdiff-text-guided-video-referring-matting","slug":"vrmdiff-text-guided-video-referring-matting","title":"VRMDiff: Text-Guided Video Referring Matting Generation of Diffusion","date":"2025-03-11","arxiv_id":"2503.10678","repositories_listed":1,"syntology":null},{"url":"/paper/ar-diffusion-asynchronous-video-generation","slug":"ar-diffusion-asynchronous-video-generation","title":"AR-Diffusion: Asynchronous Video Generation with Auto-Regressive Diffusion","date":"2025-03-10","arxiv_id":"2503.07418","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ar-diffusion-asynchronous-video-generation#ran","syntology_url":"https://syntology.ai/paper/2503.07418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07418"}},"official":null}},{"url":"/paper/automated-movie-generation-via-multi-agent","slug":"automated-movie-generation-via-multi-agent","title":"Automated Movie Generation via Multi-Agent CoT Planning","date":"2025-03-10","arxiv_id":"2503.07314","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/automated-movie-generation-via-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2503.07314","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07314"}},"official":{"repos":["showlab/movieagent"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/a-light-and-tuning-free-method-for-simulating","slug":"a-light-and-tuning-free-method-for-simulating","title":"A Light and Tuning-free Method for Simulating Camera Motion in Video Generation","date":"2025-03-09","arxiv_id":"2503.06508","repositories_listed":1,"syntology":null},{"url":"/paper/generative-video-bi-flow","slug":"generative-video-bi-flow","title":"Generative Video Bi-flow","date":"2025-03-09","arxiv_id":"2503.06364","repositories_listed":1,"syntology":null},{"url":"/paper/quantcache-adaptive-importance-guided","slug":"quantcache-adaptive-importance-guided","title":"QuantCache: Adaptive Importance-Guided Quantization with Hierarchical Latent and Layer Caching for Video Generation","date":"2025-03-09","arxiv_id":"2503.06545","repositories_listed":1,"syntology":null},{"url":"/paper/videophy-2-a-challenging-action-centric","slug":"videophy-2-a-challenging-action-centric","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","date":"2025-03-09","arxiv_id":"2503.06800","repositories_listed":1,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":12,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":2,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/videophy-2-a-challenging-action-centric#ran","syntology_url":"https://syntology.ai/paper/2503.06800","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06800"}},"official":null}},{"url":"/paper/dropletvideo-a-dataset-and-approach-to","slug":"dropletvideo-a-dataset-and-approach-to","title":"DropletVideo: A Dataset and Approach to Explore Integral Spatio-Temporal Consistent Video Generation","date":"2025-03-08","arxiv_id":"2503.06053","repositories_listed":1,"syntology":null},{"url":"/paper/mm-storyagent-immersive-narrated-storybook","slug":"mm-storyagent-immersive-narrated-storybook","title":"MM-StoryAgent: Immersive Narrated Storybook Video Generation with a Multi-Agent Paradigm across Text, Image and Audio","date":"2025-03-07","arxiv_id":"2503.05242","repositories_listed":1,"syntology":null},{"url":"/paper/unified-reward-model-for-multimodal","slug":"unified-reward-model-for-multimodal","title":"Unified Reward Model for Multimodal Understanding and Generation","date":"2025-03-07","arxiv_id":"2503.05236","repositories_listed":1,"syntology":null},{"url":"/paper/the-best-of-both-worlds-integrating-language","slug":"the-best-of-both-worlds-integrating-language","title":"The Best of Both Worlds: Integrating Language Models and Diffusion Models for Video Generation","date":"2025-03-06","arxiv_id":"2503.04606","repositories_listed":1,"syntology":null},{"url":"/paper/toward-lightweight-and-fast-decoders-for","slug":"toward-lightweight-and-fast-decoders-for","title":"Toward Lightweight and Fast Decoders for Diffusion Models in Image and Video Generation","date":"2025-03-06","arxiv_id":"2503.04871","repositories_listed":1,"syntology":null},{"url":"/paper/dualdiff-dual-branch-diffusion-for-high","slug":"dualdiff-dual-branch-diffusion-for-high","title":"DualDiff+: Dual-Branch Diffusion for High-Fidelity Video Generation with Reward Guidance","date":"2025-03-05","arxiv_id":"2503.03689","repositories_listed":1,"syntology":null},{"url":"/paper/gen3c-3d-informed-world-consistent-video","slug":"gen3c-3d-informed-world-consistent-video","title":"GEN3C: 3D-Informed World-Consistent Video Generation with Precise Camera Control","date":"2025-03-05","arxiv_id":"2503.03751","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gen3c-3d-informed-world-consistent-video#ran","syntology_url":"https://syntology.ai/paper/2503.03751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.03751"}},"official":{"repos":["nv-tlabs/GEN3C"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/high-quality-virtual-single-viewpoint","slug":"high-quality-virtual-single-viewpoint","title":"High-Quality Virtual Single-Viewpoint Surgical Video: Geometric Autocalibration of Multiple Cameras in Surgical Lights","date":"2025-03-05","arxiv_id":"2503.03558","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-video-tokenization-a-conditioned","slug":"rethinking-video-tokenization-a-conditioned","title":"Rethinking Video Tokenization: A Conditioned Diffusion-based Approach","date":"2025-03-05","arxiv_id":"2503.03708","repositories_listed":1,"syntology":null},{"url":"/paper/videoufo-a-million-scale-user-focused-dataset","slug":"videoufo-a-million-scale-user-focused-dataset","title":"VideoUFO: A Million-Scale User-Focused Dataset for Text-to-Video Generation","date":"2025-03-03","arxiv_id":"2503.01739","repositories_listed":1,"syntology":null},{"url":"/paper/extrapolating-and-decoupling-image-to-video","slug":"extrapolating-and-decoupling-image-to-video","title":"Extrapolating and Decoupling Image-to-Video Generation Models: Motion Modeling is Easier Than You Think","date":"2025-03-02","arxiv_id":"2503.00948","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/extrapolating-and-decoupling-image-to-video#ran","syntology_url":"https://syntology.ai/paper/2503.00948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00948"}},"official":{"repos":["Chuge0335/EDG"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/c-drag-chain-of-thought-driven-motion","slug":"c-drag-chain-of-thought-driven-motion","title":"C-Drag: Chain-of-Thought Driven Motion Controller for Video Generation","date":"2025-02-27","arxiv_id":"2502.19868","repositories_listed":1,"syntology":null},{"url":"/paper/mobius-text-to-seamless-looping-video","slug":"mobius-text-to-seamless-looping-video","title":"Mobius: Text to Seamless Looping Video Generation via Latent Shift","date":"2025-02-27","arxiv_id":"2502.20307","repositories_listed":1,"syntology":null},{"url":"/paper/spargeattn-accurate-sparse-attention","slug":"spargeattn-accurate-sparse-attention","title":"SpargeAttention: Accurate and Training-free Sparse Attention Accelerating Any Model Inference","date":"2025-02-25","arxiv_id":"2502.18137","repositories_listed":1,"syntology":null},{"url":"/paper/diffusion-models-for-tabular-data-challenges","slug":"diffusion-models-for-tabular-data-challenges","title":"Diffusion Models for Tabular Data: Challenges, Current Progress, and Future Directions","date":"2025-02-24","arxiv_id":"2502.17119","repositories_listed":1,"syntology":null},{"url":"/paper/vidcapbench-a-comprehensive-benchmark-of","slug":"vidcapbench-a-comprehensive-benchmark-of","title":"VidCapBench: A Comprehensive Benchmark of Video Captioning for Controllable Text-to-Video Generation","date":"2025-02-18","arxiv_id":"2502.12782","repositories_listed":1,"syntology":null},{"url":"/paper/dlfr-vae-dynamic-latent-frame-rate-vae-for","slug":"dlfr-vae-dynamic-latent-frame-rate-vae-for","title":"DLFR-VAE: Dynamic Latent Frame Rate VAE for Video Generation","date":"2025-02-17","arxiv_id":"2502.11897","repositories_listed":1,"syntology":null},{"url":"/paper/object-centric-image-to-video-generation-with","slug":"object-centric-image-to-video-generation-with","title":"Object-Centric Image to Video Generation with Language Guidance","date":"2025-02-17","arxiv_id":"2502.11655","repositories_listed":1,"syntology":null},{"url":"/paper/phantom-subject-consistent-video-generation","slug":"phantom-subject-consistent-video-generation","title":"Phantom: Subject-consistent video generation via cross-modal alignment","date":"2025-02-16","arxiv_id":"2502.11079","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/phantom-subject-consistent-video-generation#ran","syntology_url":"https://syntology.ai/paper/2502.11079","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11079"}},"official":null}},{"url":"/paper/enhance-a-video-better-generated-video-for","slug":"enhance-a-video-better-generated-video-for","title":"Enhance-A-Video: Better Generated Video for Free","date":"2025-02-11","arxiv_id":"2502.07508","repositories_listed":1,"syntology":null}],"record_sha256":"68e2d01c5775c702b3f1fe6a58e2bf3ecee1d018fc87b0a6ae77dc489deac92a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}