{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-generation/papers/8","list_of":"/task/video-generation","task":"Video Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":15,"rows_per_page":100,"rows":[701,800],"of":1466,"counts":{"archive_papers_tagged":1466,"with_a_code_link":609,"where_syntology_ran_a_sample":257,"not_listed_spam_title":0,"listed":1466,"listed_where_code_ran":257,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":221,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":221,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-generation","prev":"/task/video-generation/papers/7","next":"/task/video-generation/papers/9","papers":[{"url":null,"slug":"generating-time-consistent-dynamics-with","title":"Generating time-consistent dynamics with discriminator-guided image diffusion models","date":"2025-05-14","arxiv_id":"2505.09089","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-pre-trained-autoregressive","title":"Generative Pre-trained Autoregressive Diffusion Transformer","date":"2025-05-12","arxiv_id":"2505.07344","repositories_listed":0,"syntology":null},{"url":null,"slug":"shotadapter-text-to-multi-shot-video","title":"ShotAdapter: Text-to-Multi-Shot Video Generation with Diffusion Models","date":"2025-05-12","arxiv_id":"2505.07652","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridgeiv-bridging-customized-image-and-video","title":"BridgeIV: Bridging Customized Image and Video Generation through Test-Time Autoregressive Identity Propagation","date":"2025-05-11","arxiv_id":"2505.06985","repositories_listed":0,"syntology":null},{"url":null,"slug":"dape-dual-stage-parameter-efficient-fine","title":"DAPE: Dual-Stage Parameter-Efficient Fine-Tuning for Consistent Video Editing with Diffusion Models","date":"2025-05-11","arxiv_id":"2505.07057","repositories_listed":0,"syntology":null},{"url":null,"slug":"profashion-prototype-guided-fashion-video","title":"ProFashion: Prototype-guided Fashion Video Generation with Multiple Reference Images","date":"2025-05-10","arxiv_id":"2505.06537","repositories_listed":0,"syntology":null},{"url":null,"slug":"t2vtextbench-a-human-evaluation-benchmark-for","title":"T2VTextBench: A Human Evaluation Benchmark for Textual Control in Video Generation Models","date":"2025-05-08","arxiv_id":"2505.04946","repositories_listed":0,"syntology":null},{"url":null,"slug":"paha-parts-aware-audio-driven-human-animation","title":"A Unit Enhancement and Guidance Framework for Audio-Driven Avatar Video Generation","date":"2025-05-06","arxiv_id":"2505.03603","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformers-for-learning-on-noisy-and-task","title":"Transformers for Learning on Noisy and Task-Level Manifolds: Approximation and Generalization Insights","date":"2025-05-06","arxiv_id":"2505.03205","repositories_listed":0,"syntology":null},{"url":null,"slug":"dualreal-adaptive-joint-training-for-lossless","title":"DualReal: Adaptive Joint Training for Lossless Identity-Motion Fusion in Video Customization","date":"2025-05-04","arxiv_id":"2505.02192","repositories_listed":0,"syntology":null},{"url":null,"slug":"posepilot-steering-camera-pose-for-generative","title":"PosePilot: Steering Camera Pose for Generative World Models with Self-supervised Depth","date":"2025-05-03","arxiv_id":"2505.01729","repositories_listed":0,"syntology":null},{"url":null,"slug":"t2vphysbench-a-first-principles-benchmark-for","title":"T2VPhysBench: A First-Principles Benchmark for Physical Consistency in Text-to-Video Generation","date":"2025-05-01","arxiv_id":"2505.00337","repositories_listed":0,"syntology":null},{"url":null,"slug":"capturing-conditional-dependence-via-auto","title":"Capturing Conditional Dependence via Auto-regressive Diffusion Models","date":"2025-04-30","arxiv_id":"2504.21314","repositories_listed":0,"syntology":null},{"url":null,"slug":"eye2eye-a-simple-approach-for-monocular-to","title":"Eye2Eye: A Simple Approach for Monocular-to-Stereo Video Synthesis","date":"2025-04-30","arxiv_id":"2505.00135","repositories_listed":0,"syntology":null},{"url":null,"slug":"revision-high-quality-low-cost-video","title":"ReVision: High-Quality, Low-Cost Video Generation with Explicit 3D Physics Modeling for Complex Motion and Interaction","date":"2025-04-30","arxiv_id":"2504.21855","repositories_listed":0,"syntology":null},{"url":null,"slug":"tesseract-learning-4d-embodied-world-models","title":"TesserAct: Learning 4D Embodied World Models","date":"2025-04-29","arxiv_id":"2504.20995","repositories_listed":0,"syntology":null},{"url":null,"slug":"dive-efficient-multi-view-driving-scenes","title":"DiVE: Efficient Multi-View Driving Scenes Generation Based on Video Diffusion Transformer","date":"2025-04-28","arxiv_id":"2504.19614","repositories_listed":0,"syntology":null},{"url":null,"slug":"stealing-creator-s-workflow-a-creator","title":"Stealing Creator's Workflow: A Creator-Inspired Agentic Framework with Iterative Feedback Loop for Improved Scientific Short-form Generation","date":"2025-04-26","arxiv_id":"2504.18805","repositories_listed":0,"syntology":null},{"url":null,"slug":"we-ll-fix-it-in-post-improving-text-to-video","title":"We'll Fix it in Post: Improving Text-to-Video Generation with Neuro-Symbolic Feedback","date":"2025-04-24","arxiv_id":"2504.17180","repositories_listed":0,"syntology":null},{"url":null,"slug":"manipdreamer-boosting-robotic-manipulation","title":"ManipDreamer: Boosting Robotic Manipulation World Model with Action Tree and Visual Guidance","date":"2025-04-23","arxiv_id":"2504.16464","repositories_listed":0,"syntology":null},{"url":null,"slug":"subject-driven-video-generation-via","title":"Subject-driven Video Generation via Disentangled Identity and Motion","date":"2025-04-23","arxiv_id":"2504.17816","repositories_listed":0,"syntology":null},{"url":null,"slug":"ditpainter-efficient-video-inpainting-with","title":"DiTPainter: Efficient Video Inpainting with Diffusion Transformers","date":"2025-04-22","arxiv_id":"2504.15661","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-physical-video-generation-with","title":"Reasoning Physical Video Generation with Diffusion Timestep Tokens via Reinforcement Learning","date":"2025-04-22","arxiv_id":"2504.15932","repositories_listed":0,"syntology":null},{"url":null,"slug":"dyst-xl-dynamic-layout-planning-and-content","title":"DyST-XL: Dynamic Layout Planning and Content Control for Compositional Text-to-Video Generation","date":"2025-04-21","arxiv_id":"2504.15032","repositories_listed":0,"syntology":null},{"url":null,"slug":"tiger200k-manually-curated-high-visual","title":"Tiger200K: Manually Curated High Visual Quality Video Dataset from UGC Platform","date":"2025-04-21","arxiv_id":"2504.15182","repositories_listed":0,"syntology":null},{"url":null,"slug":"turbo2k-towards-ultra-efficient-and-high","title":"Turbo2K: Towards Ultra-Efficient and High-Quality 2K Video Synthesis","date":"2025-04-20","arxiv_id":"2504.14470","repositories_listed":0,"syntology":null},{"url":null,"slug":"modular-cam-modular-dynamic-camera-view-video","title":"Modular-Cam: Modular Dynamic Camera-view Video Generation with LLM","date":"2025-04-16","arxiv_id":"2504.12048","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-devil-is-in-the-prompts-retrieval","title":"The Devil is in the Prompts: Retrieval-Augmented Prompt Optimization for Text-to-Video Generation","date":"2025-04-16","arxiv_id":"2504.11739","repositories_listed":0,"syntology":null},{"url":null,"slug":"interanimate-taming-region-aware-diffusion","title":"InterAnimate: Taming Region-aware Diffusion Model for Realistic Human Interaction Animation","date":"2025-04-15","arxiv_id":"2504.10905","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnivdiff-omni-controllable-video-diffusion","title":"OmniVDiff: Omni Controllable Video Diffusion for Generation and Understanding","date":"2025-04-15","arxiv_id":"2504.10825","repositories_listed":0,"syntology":null},{"url":null,"slug":"videopanda-video-panoramic-diffusion-with","title":"VideoPanda: Video Panoramic Diffusion with Multi-view Attention","date":"2025-04-15","arxiv_id":"2504.11389","repositories_listed":0,"syntology":null},{"url":null,"slug":"finger-content-aware-fine-grained-evaluation","title":"FingER: Content Aware Fine-grained Evaluation with Reasoning for AI-Generated Videos","date":"2025-04-14","arxiv_id":"2504.10358","repositories_listed":0,"syntology":null},{"url":null,"slug":"h3ae-high-compression-high-speed-and-high","title":"H3AE: High Compression, High Speed, and High Quality AutoEncoder for Video Diffusion Models","date":"2025-04-14","arxiv_id":"2504.10567","repositories_listed":0,"syntology":null},{"url":null,"slug":"cammimic-zero-shot-image-to-camera-motion","title":"CamMimic: Zero-Shot Image To Camera Motion Personalized Video Generation Using Diffusion Models","date":"2025-04-13","arxiv_id":"2504.09472","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-models-for-robotic-manipulation-a","title":"Diffusion Models for Robotic Manipulation: A Survey","date":"2025-04-11","arxiv_id":"2504.08438","repositories_listed":0,"syntology":null},{"url":null,"slug":"easygennet-an-efficient-framework-for-audio","title":"EasyGenNet: An Efficient Framework for Audio-Driven Gesture Video Generation Based on Diffusion Model","date":"2025-04-11","arxiv_id":"2504.08344","repositories_listed":0,"syntology":null},{"url":null,"slug":"seaweed-7b-cost-effective-training-of-video","title":"Seaweed-7B: Cost-Effective Training of Video Generation Foundation Model","date":"2025-04-11","arxiv_id":"2504.08685","repositories_listed":0,"syntology":null},{"url":null,"slug":"tokenmotion-decoupled-motion-control-via","title":"TokenMotion: Decoupled Motion Control via Token Disentanglement for Human-centric Video Generation","date":"2025-04-11","arxiv_id":"2504.08181","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-free-guidance-in-text-to-video","title":"Training-free Guidance in Text-to-Video Generation via Multimodal Planning and Structured Noise Initialization","date":"2025-04-11","arxiv_id":"2504.08641","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-the-frame-generating-360deg-panoramic","title":"Beyond the Frame: Generating 360° Panoramic Videos from Perspective Videos","date":"2025-04-10","arxiv_id":"2504.07940","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-minute-video-generation-with-test-time","title":"One-Minute Video Generation with Test-Time Training","date":"2025-04-07","arxiv_id":"2504.05298","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-i-warped-your-noise-a-temporally","title":"How I Warped Your Noise: a Temporally-Correlated Noise Prior for Diffusion Models","date":"2025-04-03","arxiv_id":"2504.03072","repositories_listed":0,"syntology":null},{"url":null,"slug":"mg-gen-single-image-to-motion-graphics","title":"MG-Gen: Single Image to Motion Graphics Generation with Layer Decomposition","date":"2025-04-03","arxiv_id":"2504.02361","repositories_listed":0,"syntology":null},{"url":null,"slug":"morpheus-benchmarking-physical-reasoning-of","title":"Morpheus: Benchmarking Physical Reasoning of Video Generative Models with Real Physical Experiments","date":"2025-04-03","arxiv_id":"2504.02918","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnicam-unified-multimodal-video-generation","title":"OmniCam: Unified Multimodal Video Generation via Camera Control","date":"2025-04-03","arxiv_id":"2504.02312","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-splatter-momentum-3d-scene-generation","title":"Scene Splatter: Momentum 3D Scene Generation from Single Image with Video Diffusion Model","date":"2025-04-03","arxiv_id":"2504.02764","repositories_listed":0,"syntology":null},{"url":null,"slug":"worldprompter-traversable-text-to-scene","title":"WorldPrompter: Traversable Text-to-Scene Generation","date":"2025-04-02","arxiv_id":"2504.02045","repositories_listed":0,"syntology":null},{"url":null,"slug":"worldscore-a-unified-evaluation-benchmark-for","title":"WorldScore: A Unified Evaluation Benchmark for World Generation","date":"2025-04-01","arxiv_id":"2504.00983","repositories_listed":0,"syntology":null},{"url":null,"slug":"any2caption-interpreting-any-condition-to","title":"Any2Caption:Interpreting Any Condition to Caption for Controllable Video Generation","date":"2025-03-31","arxiv_id":"2503.24379","repositories_listed":0,"syntology":null},{"url":"/paper/hoigen-1m-a-large-scale-dataset-for-human","slug":"hoigen-1m-a-large-scale-dataset-for-human","title":"HOIGen-1M: A Large-scale Dataset for Human-Object Interaction Video Generation","date":"2025-03-31","arxiv_id":"2503.23715","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hoigen-1m-a-large-scale-dataset-for-human#ran","syntology_url":"https://syntology.ai/paper/2503.23715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23715"}},"official":null}},{"url":null,"slug":"humandreamer-generating-controllable-human","title":"HumanDreamer: Generating Controllable Human-Motion Videos via Decoupled Generation","date":"2025-03-31","arxiv_id":"2503.24026","repositories_listed":0,"syntology":null},{"url":null,"slug":"jointtuner-appearance-motion-adaptive-joint","title":"JointTuner: Appearance-Motion Adaptive Joint Training for Customized Video Generation","date":"2025-03-31","arxiv_id":"2503.23951","repositories_listed":0,"syntology":null},{"url":null,"slug":"javisdit-joint-audio-video-diffusion","title":"JavisDiT: Joint Audio-Video Diffusion Transformer with Hierarchical Spatio-Temporal Prior Synchronization","date":"2025-03-30","arxiv_id":"2503.23377","repositories_listed":0,"syntology":null},{"url":null,"slug":"mocha-towards-movie-grade-talking-character","title":"MoCha: Towards Movie-Grade Talking Character Synthesis","date":"2025-03-30","arxiv_id":"2503.23307","repositories_listed":0,"syntology":null},{"url":null,"slug":"sketchvideo-sketch-based-video-generation-and","title":"SketchVideo: Sketch-based Video Generation and Editing","date":"2025-03-30","arxiv_id":"2503.23284","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-physically-plausible-video-generation","title":"Towards Physically Plausible Video Generation via VLM Planning","date":"2025-03-30","arxiv_id":"2503.23368","repositories_listed":0,"syntology":null},{"url":null,"slug":"cogen-3d-consistent-video-generation-via","title":"CoGen: 3D Consistent Video Generation via Adaptive Conditioning for Autonomous Driving","date":"2025-03-28","arxiv_id":"2503.22231","repositories_listed":0,"syntology":null},{"url":null,"slug":"echoflow-a-foundation-model-for-cardiac","title":"EchoFlow: A Foundation Model for Cardiac Ultrasound Image and Video Generation","date":"2025-03-28","arxiv_id":"2503.22357","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero4d-training-free-4d-video-generation-from","title":"Zero4D: Training-Free 4D Video Generation From Single Video Using Off-the-Shelf Video Diffusion Model","date":"2025-03-28","arxiv_id":"2503.22622","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-driven-gesture-generation-via-deviation","title":"Audio-driven Gesture Generation via Deviation Feature in the Latent Space","date":"2025-03-27","arxiv_id":"2503.21616","repositories_listed":0,"syntology":null},{"url":null,"slug":"chatanyone-stylized-real-time-portrait-video","title":"ChatAnyone: Stylized Real-time Portrait Video Generation with Hierarchical Motion Diffusion Model","date":"2025-03-27","arxiv_id":"2503.21144","repositories_listed":0,"syntology":null},{"url":null,"slug":"videomage-multi-subject-and-motion","title":"VideoMage: Multi-Subject and Motion Customization of Text-to-Video Diffusion Models","date":"2025-03-27","arxiv_id":"2503.21781","repositories_listed":0,"syntology":null},{"url":null,"slug":"accidentsim-generating-physically-realistic","title":"AccidentSim: Generating Physically Realistic Vehicle Collision Videos from Real-World Accident Reports","date":"2025-03-26","arxiv_id":"2503.20654","repositories_listed":0,"syntology":null},{"url":null,"slug":"gaia-2-a-controllable-multi-view-generative","title":"GAIA-2: A Controllable Multi-View Generative World Model for Autonomous Driving","date":"2025-03-26","arxiv_id":"2503.20523","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthetic-video-enhances-physical-fidelity-in","title":"Synthetic Video Enhances Physical Fidelity in Video Synthesis","date":"2025-03-26","arxiv_id":"2503.20822","repositories_listed":0,"syntology":null},{"url":null,"slug":"unconditional-priors-matter-improving","title":"Unconditional Priors Matter! Improving Conditional Generation of Fine-Tuned Diffusion Models","date":"2025-03-26","arxiv_id":"2503.20240","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-motion-graphs","title":"Video Motion Graphs","date":"2025-03-26","arxiv_id":"2503.20218","repositories_listed":0,"syntology":null},{"url":null,"slug":"audcast-audio-driven-human-video-generation","title":"AudCast: Audio-Driven Human Video Generation by Cascaded Diffusion Transformers","date":"2025-03-25","arxiv_id":"2503.19824","repositories_listed":0,"syntology":null},{"url":null,"slug":"fulldit-multi-task-video-generative","title":"FullDiT: Multi-Task Video Generative Foundation Model with Full Attention","date":"2025-03-25","arxiv_id":"2503.19907","repositories_listed":0,"syntology":null},{"url":null,"slug":"fuxi-rtm-a-physics-guided-prediction","title":"FuXi-RTM: A Physics-Guided Prediction Framework with Radiative Transfer Modeling","date":"2025-03-25","arxiv_id":"2503.19940","repositories_listed":0,"syntology":null},{"url":null,"slug":"mask-2-dit-dual-mask-based-diffusion","title":"Mask$^2$DiT: Dual Mask-based Diffusion Transformer for Multi-Scene Long Video Generation","date":"2025-03-25","arxiv_id":"2503.19881","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-learning-of-motion-concepts","title":"Self-Supervised Learning of Motion Concepts by Optimizing Counterfactuals","date":"2025-03-25","arxiv_id":"2503.19953","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-human-object-interaction-synthesis","title":"Zero-Shot Human-Object Interaction Synthesis with Multimodal Priors","date":"2025-03-25","arxiv_id":"2503.20118","repositories_listed":0,"syntology":null},{"url":null,"slug":"aether-geometric-aware-unified-world-modeling","title":"Aether: Geometric-Aware Unified World Modeling","date":"2025-03-24","arxiv_id":"2503.18945","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-text-to-video-generation-help-video","title":"Can Text-to-Video Generation help Video-Language Alignment?","date":"2025-03-24","arxiv_id":"2503.18507","repositories_listed":0,"syntology":null},{"url":null,"slug":"evanimate-event-conditioned-image-to-video","title":"EvAnimate: Event-conditioned Image-to-Video Generation for Human Animation","date":"2025-03-24","arxiv_id":"2503.18552","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-efficient-motion-control-for-video","title":"Resource-Efficient Motion Control for Video Generation via Dynamic Mask Guidance","date":"2025-03-24","arxiv_id":"2503.18386","repositories_listed":0,"syntology":null},{"url":null,"slug":"teller-real-time-streaming-audio-driven","title":"Teller: Real-Time Streaming Audio-Driven Portrait Animation with Autoregressive Motion Generation","date":"2025-03-24","arxiv_id":"2503.18429","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-free-diffusion-acceleration-with","title":"Training-free Diffusion Acceleration with Bottleneck Sampling","date":"2025-03-24","arxiv_id":"2503.18940","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-t1-test-time-scaling-for-video","title":"Video-T1: Test-Time Scaling for Video Generation","date":"2025-03-24","arxiv_id":"2503.18942","repositories_listed":0,"syntology":null},{"url":null,"slug":"longdiff-training-free-long-video-generation","title":"LongDiff: Training-Free Long Video Generation in One Go","date":"2025-03-23","arxiv_id":"2503.18150","repositories_listed":0,"syntology":null},{"url":null,"slug":"transanimate-taming-layer-diffusion-to","title":"TransAnimate: Taming Layer Diffusion to Generate RGBA Video","date":"2025-03-23","arxiv_id":"2503.17934","repositories_listed":0,"syntology":null},{"url":null,"slug":"rdtf-resource-efficient-dual-mask-training","title":"RDTF: Resource-efficient Dual-mask Training Framework for Multi-frame Animated Sticker Generation","date":"2025-03-22","arxiv_id":"2503.17735","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-fast-and-slow-scalable-parallel","title":"Generating, Fast and Slow: Scalable Parallel Video Generation with Video Interface Networks","date":"2025-03-21","arxiv_id":"2503.17539","repositories_listed":0,"syntology":null},{"url":null,"slug":"position-interactive-generative-video-as-next","title":"Position: Interactive Generative Video as Next-Generation Game Engine","date":"2025-03-21","arxiv_id":"2503.17359","repositories_listed":0,"syntology":null},{"url":null,"slug":"re-hold-video-hand-object-interaction","title":"Re-HOLD: Video Hand Object Interaction Reenactment via adaptive Layout-instructed Diffusion Model","date":"2025-03-21","arxiv_id":"2503.16942","repositories_listed":0,"syntology":null},{"url":null,"slug":"magicmotion-controllable-video-generation","title":"MagicMotion: Controllable Video Generation with Dense-to-Sparse Trajectory Guidance","date":"2025-03-20","arxiv_id":"2503.16421","repositories_listed":0,"syntology":null},{"url":null,"slug":"posetraj-pose-aware-trajectory-control-in","title":"PoseTraj: Pose-Aware Trajectory Control in Video Diffusion","date":"2025-03-20","arxiv_id":"2503.16068","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalingnoise-scaling-inference-time-search","title":"ScalingNoise: Scaling Inference-Time Search for Generating Infinite Videos","date":"2025-03-20","arxiv_id":"2503.16400","repositories_listed":0,"syntology":null},{"url":null,"slug":"videorfsplat-direct-scene-level-text-to-3d","title":"VideoRFSplat: Direct Scene-Level Text-to-3D Gaussian Splatting Generation with Flexible Pose and Multi-View Joint Modeling","date":"2025-03-20","arxiv_id":"2503.15855","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-regularization-makes-your-video","title":"Temporal Regularization Makes Your Video Generator Stronger","date":"2025-03-19","arxiv_id":"2503.15417","repositories_listed":0,"syntology":null},{"url":null,"slug":"videogen-of-thought-step-by-step-generating","title":"VideoGen-of-Thought: Step-by-step generating multi-shot video with minimal manual intervention","date":"2025-03-19","arxiv_id":"2503.15138","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-autoregressive-video-generation-with","title":"Fast Autoregressive Video Generation with Diagonal Decoding","date":"2025-03-18","arxiv_id":"2503.14070","repositories_listed":0,"syntology":null},{"url":null,"slug":"impossible-videos","title":"Impossible Videos","date":"2025-03-18","arxiv_id":"2503.14378","repositories_listed":0,"syntology":null},{"url":null,"slug":"magiccomp-training-free-dual-phase-refinement","title":"MagicComp: Training-free Dual-Phase Refinement for Compositional Video Generation","date":"2025-03-18","arxiv_id":"2503.14428","repositories_listed":0,"syntology":null},{"url":null,"slug":"musicinfuser-making-video-diffusion-listen","title":"MusicInfuser: Making Video Diffusion Listen and Dance","date":"2025-03-18","arxiv_id":"2503.14505","repositories_listed":0,"syntology":null},{"url":null,"slug":"autv-creating-underwater-video-datasets-with","title":"AUTV: Creating Underwater Video Datasets with Pixel-wise Annotations","date":"2025-03-17","arxiv_id":"2503.12828","repositories_listed":0,"syntology":null},{"url":null,"slug":"frame-wise-conditioning-adaptation-for-fine","title":"Frame-wise Conditioning Adaptation for Fine-Tuning Diffusion Models in Text-to-Video Prediction","date":"2025-03-17","arxiv_id":"2503.12953","repositories_listed":0,"syntology":null},{"url":null,"slug":"eq-taa-equivariant-traffic-accident","title":"EQ-TAA: Equivariant Traffic Accident Anticipation via Diffusion-Based Accident Video Synthesis","date":"2025-03-16","arxiv_id":"2506.10002","repositories_listed":0,"syntology":null},{"url":null,"slug":"spc-gs-gaussian-splatting-with-semantic","title":"SPC-GS: Gaussian Splatting with Semantic-Prompt Consistency for Indoor Open-World Free-view Synthesis from Sparse Inputs","date":"2025-03-16","arxiv_id":"2503.12535","repositories_listed":0,"syntology":null}],"record_sha256":"5254c32a41a4bc8c01c01ada24e4131625f2a9b011e88aa5cb43d4403c280939","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}