{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-generation/papers/11","list_of":"/task/video-generation","task":"Video Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":15,"rows_per_page":100,"rows":[1001,1100],"of":1466,"counts":{"archive_papers_tagged":1466,"with_a_code_link":609,"where_syntology_ran_a_sample":257,"not_listed_spam_title":0,"listed":1466,"listed_where_code_ran":257,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":221,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":221,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-generation","prev":"/task/video-generation/papers/10","next":"/task/video-generation/papers/12","papers":[{"url":null,"slug":"infinitydrive-breaking-time-limits-in-driving","title":"InfinityDrive: Breaking Time Limits in Driving World Models","date":"2024-12-02","arxiv_id":"2412.01522","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-video-diffusion-generation-with","title":"Long Video Diffusion Generation with Segmented Cross-Attention and Content-Rich Video Data Curation","date":"2024-12-02","arxiv_id":"2412.01316","repositories_listed":0,"syntology":null},{"url":null,"slug":"motrans-customized-motion-transfer-with-text","title":"MoTrans: Customized Motion Transfer with Text-driven Video Diffusion Models","date":"2024-12-02","arxiv_id":"2412.01343","repositories_listed":0,"syntology":null},{"url":null,"slug":"world-consistent-video-diffusion-with","title":"World-consistent Video Diffusion with Explicit 3D Modeling","date":"2024-12-02","arxiv_id":"2412.01821","repositories_listed":0,"syntology":null},{"url":null,"slug":"divd-deblurring-with-improved-video-diffusion","title":"DIVD: Deblurring with Improved Video Diffusion Model","date":"2024-12-01","arxiv_id":"2412.00773","repositories_listed":0,"syntology":null},{"url":null,"slug":"synergizing-motion-and-appearance-multi-scale","title":"Synergizing Motion and Appearance: Multi-Scale Compensatory Codebooks for Talking Head Video Generation","date":"2024-12-01","arxiv_id":"2412.00719","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-action-clips-detecting-ai-generated","title":"Human Action CLIPs: Detecting AI-generated Human Motion","date":"2024-11-30","arxiv_id":"2412.00526","repositories_listed":0,"syntology":null},{"url":null,"slug":"fleximo-towards-flexible-text-to-human-motion","title":"Fleximo: Towards Flexible Text-to-Human Motion Video Generation","date":"2024-11-29","arxiv_id":"2411.19459","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-modes-what-could-happen-next","title":"Motion Modes: What Could Happen Next?","date":"2024-11-29","arxiv_id":"2412.00148","repositories_listed":0,"syntology":null},{"url":null,"slug":"msg-score-a-comprehensive-evaluation-for","title":"MSG score: A Comprehensive Evaluation for Multi-Scene Video Generation","date":"2024-11-28","arxiv_id":"2411.19121","repositories_listed":0,"syntology":null},{"url":null,"slug":"openhumanvid-a-large-scale-high-quality","title":"OpenHumanVid: A Large-Scale High-Quality Dataset for Enhancing Human-Centric Video Generation","date":"2024-11-28","arxiv_id":"2412.00115","repositories_listed":0,"syntology":null},{"url":null,"slug":"spagent-adaptive-task-decomposition-and-model","title":"SPAgent: Adaptive Task Decomposition and Model Selection for General Video Generation and Editing","date":"2024-11-28","arxiv_id":"2411.18983","repositories_listed":0,"syntology":null},{"url":null,"slug":"trajectory-attention-for-fine-grained-video","title":"Trajectory Attention for Fine-grained Video Motion Control","date":"2024-11-28","arxiv_id":"2411.19324","repositories_listed":0,"syntology":null},{"url":null,"slug":"ac3d-analyzing-and-improving-3d-camera","title":"AC3D: Analyzing and Improving 3D Camera Control in Video Diffusion Transformers","date":"2024-11-27","arxiv_id":"2411.18673","repositories_listed":0,"syntology":null},{"url":null,"slug":"individual-content-and-motion-dynamics","title":"Individual Content and Motion Dynamics Preserved Pruning for Video Diffusion Models","date":"2024-11-27","arxiv_id":"2411.18375","repositories_listed":0,"syntology":null},{"url":null,"slug":"motioncharacter-identity-preserving-and","title":"MotionCharacter: Identity-Preserving and Motion Controllable Human Video Generation","date":"2024-11-27","arxiv_id":"2411.18281","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-chunk-wise-generation-for-long-videos","title":"Towards Chunk-Wise Generation for Long Videos","date":"2024-11-27","arxiv_id":"2411.18668","repositories_listed":0,"syntology":null},{"url":null,"slug":"anchorcrafter-animate-cyberanchors-saling","title":"AnchorCrafter: Animate CyberAnchors Saling Your Products via Human-Object Interacting Video Generation","date":"2024-11-26","arxiv_id":"2411.17383","repositories_listed":0,"syntology":null},{"url":null,"slug":"free-2-guide-gradient-free-path-integral","title":"Free$^2$Guide: Gradient-Free Path Integral Control for Enhancing Text-to-Video Generation with Large Vision-Language Models","date":"2024-11-26","arxiv_id":"2411.17041","repositories_listed":0,"syntology":null},{"url":null,"slug":"passive-deepfake-detection-across-multi","title":"Passive Deepfake Detection Across Multi-modalities: A Comprehensive Survey","date":"2024-11-26","arxiv_id":"2411.17911","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalvideo-high-id-fidelity-video","title":"PersonalVideo: High ID-Fidelity Video Customization without Dynamic and Semantic Degradation","date":"2024-11-26","arxiv_id":"2411.17048","repositories_listed":0,"syntology":null},{"url":null,"slug":"physmotion-physics-grounded-dynamics-from-a","title":"PhysMotion: Physics-Grounded Dynamics From a Single Image","date":"2024-11-26","arxiv_id":"2411.17189","repositories_listed":0,"syntology":null},{"url":null,"slug":"dreamrunner-fine-grained-storytelling-video","title":"DreamRunner: Fine-Grained Storytelling Video Generation with Retrieval-Augmented Motion Adaptation","date":"2024-11-25","arxiv_id":"2411.16657","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-activity-agv-quality-assessment-a","title":"Human-Activity AGV Quality Assessment: A Benchmark Dataset and an Objective Evaluation Metric","date":"2024-11-25","arxiv_id":"2411.16619","repositories_listed":0,"syntology":null},{"url":null,"slug":"pathways-on-the-image-manifold-image-editing","title":"Pathways on the Image Manifold: Image Editing via Video Generation","date":"2024-11-25","arxiv_id":"2411.16819","repositories_listed":0,"syntology":null},{"url":null,"slug":"letstalk-latent-diffusion-transformer-for","title":"LetsTalk: Latent Diffusion Transformer for Talking Video Synthesis","date":"2024-11-24","arxiv_id":"2411.16748","repositories_listed":0,"syntology":null},{"url":null,"slug":"importance-based-token-merging-for-efficient","title":"Importance-Based Token Merging for Efficient Image and Video Generation","date":"2024-11-23","arxiv_id":"2411.16720","repositories_listed":0,"syntology":null},{"url":null,"slug":"optical-flow-guided-prompt-optimization-for","title":"Optical-Flow Guided Prompt Optimization for Coherent Video Generation","date":"2024-11-23","arxiv_id":"2411.15540","repositories_listed":0,"syntology":null},{"url":null,"slug":"videorepair-improving-text-to-video","title":"VideoRepair: Improving Text-to-Video Generation via Misalignment Evaluation and Localized Refinement","date":"2024-11-22","arxiv_id":"2411.15115","repositories_listed":0,"syntology":null},{"url":null,"slug":"magicdrivedit-high-resolution-long-video","title":"MagicDriveDiT: High-Resolution Long Video Generation for Autonomous Driving with Adaptive Control","date":"2024-11-21","arxiv_id":"2411.13807","repositories_listed":0,"syntology":null},{"url":"/paper/taq-dit-time-aware-quantization-for-diffusion","slug":"taq-dit-time-aware-quantization-for-diffusion","title":"TaQ-DiT: Time-aware Quantization for Diffusion Transformers","date":"2024-11-21","arxiv_id":"2411.14172","repositories_listed":0,"syntology":{"n":7,"n_ran":5,"n_constructed":2,"n_ran_checked":2,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/taq-dit-time-aware-quantization-for-diffusion#ran","syntology_url":"https://syntology.ai/paper/2411.14172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14172"}},"official":null}},{"url":null,"slug":"understanding-world-or-predicting-future-a","title":"Understanding World or Predicting Future? A Comprehensive Survey of World Models","date":"2024-11-21","arxiv_id":"2411.14499","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-you-see-is-what-matters-a-novel-visual","title":"What You See Is What Matters: A Novel Visual and Physics-Based Metric for Evaluating Video Generation Quality","date":"2024-11-20","arxiv_id":"2411.13609","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-3d-physical-simulation-of-open","title":"Automated 3D Physical Simulation of Open-world Scene with Gaussian Splatting","date":"2024-11-19","arxiv_id":"2411.12789","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-motion-from-video-diffusion-models","title":"Towards motion from video diffusion models","date":"2024-11-19","arxiv_id":"2411.12831","repositories_listed":0,"syntology":null},{"url":null,"slug":"medical-video-generation-for-disease","title":"Medical Video Generation for Disease Progression Simulation","date":"2024-11-18","arxiv_id":"2411.11943","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatialdreamer-self-supervised-stereo-video","title":"SpatialDreamer: Self-supervised Stereo Video Synthesis from Monocular Input","date":"2024-11-18","arxiv_id":"2411.11934","repositories_listed":0,"syntology":null},{"url":null,"slug":"teaching-video-diffusion-model-with-latent","title":"Teaching Video Diffusion Model with Latent Physical Phenomenon Knowledge","date":"2024-11-18","arxiv_id":"2411.11343","repositories_listed":0,"syntology":null},{"url":null,"slug":"animateanything-consistent-and-controllable","title":"AnimateAnything: Consistent and Controllable Animation for Video Generation","date":"2024-11-16","arxiv_id":"2411.10836","repositories_listed":0,"syntology":null},{"url":null,"slug":"vibe-a-text-to-video-benchmark-for-evaluating","title":"ViBe: A Text-to-Video Benchmark for Evaluating Hallucination in Large Multimodal Models","date":"2024-11-16","arxiv_id":"2411.10867","repositories_listed":0,"syntology":null},{"url":"/paper/vidman-exploiting-implicit-dynamics-from","slug":"vidman-exploiting-implicit-dynamics-from","title":"VidMan: Exploiting Implicit Dynamics from Video Diffusion Model for Effective Robot Manipulation","date":"2024-11-14","arxiv_id":"2411.09153","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-vision-autoregressive-model","title":"A Survey on Vision Autoregressive Model","date":"2024-11-13","arxiv_id":"2411.08666","repositories_listed":0,"syntology":null},{"url":null,"slug":"egovid-5m-a-large-scale-video-action-dataset","title":"EgoVid-5M: A Large-Scale Video-Action Dataset for Egocentric Video Generation","date":"2024-11-13","arxiv_id":"2411.08380","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-control-for-enhanced-complex-action","title":"Motion Control for Enhanced Complex Action Video Generation","date":"2024-11-13","arxiv_id":"2411.08328","repositories_listed":0,"syntology":null},{"url":null,"slug":"artificial-intelligence-for-biomedical-video","title":"Artificial Intelligence for Biomedical Video Generation","date":"2024-11-12","arxiv_id":"2411.07619","repositories_listed":0,"syntology":null},{"url":null,"slug":"i2vcontrol-camera-precise-video-camera","title":"I2VControl-Camera: Precise Video Camera Control with Adjustable Motion Strength","date":"2024-11-10","arxiv_id":"2411.06525","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-emerging-approaches-and-advances","title":"A Survey of Emerging Approaches and Advances in Video Generation","date":"2024-11-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"whale-towards-generalizable-and-scalable","title":"WHALE: Towards Generalizable and Scalable World Models for Embodied Decision-making","date":"2024-11-08","arxiv_id":"2411.05619","repositories_listed":0,"syntology":null},{"url":null,"slug":"dimensionx-create-any-3d-and-4d-scenes-from-a","title":"DimensionX: Create Any 3D and 4D Scenes from a Single Image with Controllable Video Diffusion","date":"2024-11-07","arxiv_id":"2411.04928","repositories_listed":0,"syntology":null},{"url":null,"slug":"sg-i2v-self-guided-trajectory-control-in","title":"SG-I2V: Self-Guided Trajectory Control in Image-to-Video Generation","date":"2024-11-07","arxiv_id":"2411.04989","repositories_listed":0,"syntology":null},{"url":null,"slug":"storyagent-customized-storytelling-video","title":"StoryAgent: Customized Storytelling Video Generation via Multi-Agent Collaboration","date":"2024-11-07","arxiv_id":"2411.04925","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-interplay-between-video","title":"Exploring the Interplay Between Video Generation and World Models in Autonomous Driving: A Survey","date":"2024-11-05","arxiv_id":"2411.02914","repositories_listed":0,"syntology":null},{"url":null,"slug":"tip-i2v-a-million-scale-real-text-and-image","title":"TIP-I2V: A Million-Scale Real Text and Image Prompt Dataset for Image-to-Video Generation","date":"2024-11-05","arxiv_id":"2411.04709","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-caching-for-faster-video-generation","title":"Adaptive Caching for Faster Video Generation with Diffusion Transformers","date":"2024-11-04","arxiv_id":"2411.02397","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-far-is-video-generation-from-world-model","title":"How Far is Video Generation from World Model: A Physical Law Perspective","date":"2024-11-04","arxiv_id":"2411.02385","repositories_listed":0,"syntology":null},{"url":null,"slug":"optical-flow-representation-alignment-mamba","title":"Optical Flow Representation Alignment Mamba Diffusion Model for Medical Video Generation","date":"2024-11-03","arxiv_id":"2411.01647","repositories_listed":0,"syntology":null},{"url":"/paper/fashion-vdm-video-diffusion-model-for-virtual-1","slug":"fashion-vdm-video-diffusion-model-for-virtual-1","title":"Fashion-VDM: Video Diffusion Model for Virtual Try-On","date":"2024-10-31","arxiv_id":"2411.00225","repositories_listed":0,"syntology":null},{"url":null,"slug":"stereo-talker-audio-driven-3d-human-synthesis","title":"Stereo-Talker: Audio-driven 3D Human Synthesis with Prior-Guided Mixture-of-Experts","date":"2024-10-31","arxiv_id":"2410.23836","repositories_listed":0,"syntology":null},{"url":null,"slug":"lumisculpt-a-consistency-lighting-control","title":"LumiSculpt: A Consistency Lighting Control Network for Video Generation","date":"2024-10-30","arxiv_id":"2410.22979","repositories_listed":0,"syntology":null},{"url":null,"slug":"slowfast-vgen-slow-fast-learning-for-action","title":"SlowFast-VGen: Slow-Fast Learning for Action-Driven Long Video Generation","date":"2024-10-30","arxiv_id":"2410.23277","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-memorization-in-video-diffusion","title":"Investigating Memorization in Video Diffusion Models","date":"2024-10-29","arxiv_id":"2410.21669","repositories_listed":0,"syntology":null},{"url":null,"slug":"arlon-boosting-diffusion-transformers-with","title":"ARLON: Boosting Diffusion Transformers with Autoregressive Models for Long Video Generation","date":"2024-10-27","arxiv_id":"2410.20502","repositories_listed":0,"syntology":null},{"url":null,"slug":"give-guiding-visual-encoder-to-perceive","title":"GiVE: Guiding Visual Encoder to Perceive Overlooked Information","date":"2024-10-26","arxiv_id":"2410.20109","repositories_listed":0,"syntology":null},{"url":null,"slug":"mardini-masked-autoregressive-diffusion-for","title":"MarDini: Masked Autoregressive Diffusion for Video Generation at Scale","date":"2024-10-26","arxiv_id":"2410.20280","repositories_listed":0,"syntology":null},{"url":null,"slug":"fastercache-training-free-video-diffusion","title":"FasterCache: Training-Free Video Diffusion Model Acceleration with High Quality","date":"2024-10-25","arxiv_id":"2410.19355","repositories_listed":0,"syntology":null},{"url":null,"slug":"framer-interactive-frame-interpolation","title":"Framer: Interactive Frame Interpolation","date":"2024-10-24","arxiv_id":"2410.18978","repositories_listed":0,"syntology":null},{"url":null,"slug":"visage-video-synthesis-using-action-graphs","title":"VISAGE: Video Synthesis using Action Graphs for Surgery","date":"2024-10-23","arxiv_id":"2410.17751","repositories_listed":0,"syntology":null},{"url":null,"slug":"worldsimbench-towards-video-generation-models","title":"WorldSimBench: Towards Video Generation Models as World Simulators","date":"2024-10-23","arxiv_id":"2410.18072","repositories_listed":0,"syntology":null},{"url":null,"slug":"3dgs-enhancer-enhancing-unbounded-3d-gaussian","title":"3DGS-Enhancer: Enhancing Unbounded 3D Gaussian Splatting with View-consistent 2D Diffusion Priors","date":"2024-10-21","arxiv_id":"2410.16266","repositories_listed":0,"syntology":null},{"url":null,"slug":"eva-an-embodied-world-model-for-future-video","title":"EVA: An Embodied World Model for Future Video Anticipation","date":"2024-10-20","arxiv_id":"2410.15461","repositories_listed":0,"syntology":null},{"url":null,"slug":"framebridge-improving-image-to-video","title":"FrameBridge: Improving Image-to-Video Generation with Bridge Models","date":"2024-10-20","arxiv_id":"2410.15371","repositories_listed":0,"syntology":null},{"url":null,"slug":"asymkv-enabling-1-bit-quantization-of-kv","title":"AsymKV: Enabling 1-Bit Quantization of KV Cache with Layer-Wise Asymmetric Quantization Configurations","date":"2024-10-17","arxiv_id":"2410.13212","repositories_listed":0,"syntology":null},{"url":null,"slug":"dreamvideo-2-zero-shot-subject-driven-video","title":"DreamVideo-2: Zero-Shot Subject-Driven Video Customization with Precise Motion Control","date":"2024-10-17","arxiv_id":"2410.13830","repositories_listed":0,"syntology":null},{"url":null,"slug":"drivedreamer4d-world-models-are-effective","title":"DriveDreamer4D: World Models Are Effective Data Machines for 4D Driving Scene Representation","date":"2024-10-17","arxiv_id":"2410.13571","repositories_listed":0,"syntology":null},{"url":null,"slug":"fundus-to-fluorescein-angiography-video","title":"Fundus to Fluorescein Angiography Video Generation as a Retinal Generative Foundation Model","date":"2024-10-17","arxiv_id":"2410.13242","repositories_listed":0,"syntology":null},{"url":null,"slug":"vidpanos-generative-panoramic-videos-from","title":"VidPanos: Generative Panoramic Videos from Casual Panning Videos","date":"2024-10-17","arxiv_id":"2410.13832","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-camera-motion-control-for-video","title":"Boosting Camera Motion Control for Video Diffusion Transformers","date":"2024-10-14","arxiv_id":"2410.10802","repositories_listed":0,"syntology":null},{"url":null,"slug":"cavia-camera-controllable-multi-view-video","title":"Cavia: Camera-controllable Multi-view Video Diffusion with View-Integrated Attention","date":"2024-10-14","arxiv_id":"2410.10774","repositories_listed":0,"syntology":null},{"url":null,"slug":"dragentity-trajectory-guided-video-generation","title":"DragEntity: Trajectory Guided Video Generation using Entity and Positional Relationships","date":"2024-10-14","arxiv_id":"2410.10751","repositories_listed":0,"syntology":null},{"url":null,"slug":"quality-prediction-of-ai-generated-images-and","title":"Quality Prediction of AI Generated Images and Videos: Emerging Trends and Opportunities","date":"2024-10-11","arxiv_id":"2410.08534","repositories_listed":0,"syntology":null},{"url":null,"slug":"animating-the-past-reconstruct-trilobite-via","title":"Animating the Past: Reconstruct Trilobite via Video Generation","date":"2024-10-10","arxiv_id":"2410.14715","repositories_listed":0,"syntology":null},{"url":null,"slug":"harivo-harnessing-text-to-image-models-for","title":"HARIVO: Harnessing Text-to-Image Models for Video Generation","date":"2024-10-10","arxiv_id":"2410.07763","repositories_listed":0,"syntology":null},{"url":null,"slug":"koala-36m-a-large-scale-video-dataset","title":"Koala-36M: A Large-scale Video Dataset Improving Consistency between Fine-grained Conditions and Video Content","date":"2024-10-10","arxiv_id":"2410.08260","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-laws-for-diffusion-transformers","title":"Scaling Laws For Diffusion Transformers","date":"2024-10-10","arxiv_id":"2410.08184","repositories_listed":0,"syntology":null},{"url":null,"slug":"broadway-boost-your-text-to-video-generation","title":"BroadWay: Boost Your Text-to-Video Generation Model in a Training-free Way","date":"2024-10-08","arxiv_id":"2410.06241","repositories_listed":0,"syntology":null},{"url":null,"slug":"gr-2-a-generative-video-language-action-model","title":"GR-2: A Generative Video-Language-Action Model with Web-Scale Knowledge for Robot Manipulation","date":"2024-10-08","arxiv_id":"2410.06158","repositories_listed":0,"syntology":null},{"url":null,"slug":"vibidsampler-enhancing-video-interpolation","title":"ViBiDSampler: Enhancing Video Interpolation Using Bidirectional Diffusion Sampler","date":"2024-10-08","arxiv_id":"2410.05651","repositories_listed":0,"syntology":null},{"url":null,"slug":"acdc-autoregressive-coherent-multimodal","title":"ACDC: Autoregressive Coherent Multimodal Generation using Diffusion Correction","date":"2024-10-07","arxiv_id":"2410.04721","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-dawn-of-video-generation-preliminary","title":"The Dawn of Video Generation: Preliminary Explorations with SORA-like Models","date":"2024-10-07","arxiv_id":"2410.05227","repositories_listed":0,"syntology":null},{"url":null,"slug":"realizing-video-summarization-from-the-path","title":"Realizing Video Summarization from the Path of Language-based Semantic Understanding","date":"2024-10-06","arxiv_id":"2410.04511","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-crystallization-and-liquid-noise-zero","title":"Noise Crystallization and Liquid Noise: Zero-shot Video Generation using Image Diffusion Models","date":"2024-10-05","arxiv_id":"2410.05322","repositories_listed":0,"syntology":null},{"url":null,"slug":"loong-generating-minute-level-long-videos","title":"Loong: Generating Minute-level Long Videos with Autoregressive Language Models","date":"2024-10-03","arxiv_id":"2410.02757","repositories_listed":0,"syntology":null},{"url":null,"slug":"people-are-poorly-equipped-to-detect-ai","title":"People are poorly equipped to detect AI-powered voice clones","date":"2024-10-03","arxiv_id":"2410.03791","repositories_listed":0,"syntology":null},{"url":null,"slug":"comuni-decomposing-common-and-unique-video","title":"COMUNI: Decomposing Common and Unique Video Signals for Diffusion-based Video Generation","date":"2024-10-02","arxiv_id":"2410.01718","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-bench-video-benchmarking-the-video-quality","title":"Q-Bench-Video: Benchmarking the Video Quality Understanding of LMMs","date":"2024-09-30","arxiv_id":"2409.20063","repositories_listed":0,"syntology":null},{"url":null,"slug":"convergence-of-diffusion-models-under-the","title":"Convergence of Diffusion Models Under the Manifold Hypothesis in High-Dimensions","date":"2024-09-27","arxiv_id":"2409.18804","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-learning-of-deviation-in","title":"Self-Supervised Learning of Deviation in Latent Representation for Co-speech Gesture Video Generation","date":"2024-09-26","arxiv_id":"2409.17674","repositories_listed":0,"syntology":null},{"url":null,"slug":"pose-guided-fine-grained-sign-language-video","title":"Pose-Guided Fine-Grained Sign Language Video Generation","date":"2024-09-25","arxiv_id":"2409.16709","repositories_listed":0,"syntology":null},{"url":null,"slug":"gen2act-human-video-generation-in-novel","title":"Gen2Act: Human Video Generation in Novel Scenarios enables Generalizable Robot Manipulation","date":"2024-09-24","arxiv_id":"2409.16283","repositories_listed":0,"syntology":null},{"url":null,"slug":"technical-report-competition-solution-for-2","title":"Technical Report: Competition Solution For Modelscope-Sora","date":"2024-09-24","arxiv_id":"2410.07194","repositories_listed":0,"syntology":null}],"record_sha256":"2b54f3a5e0c0f462b9330eee86573d424e765056ca9471c2d57a35e3e51febb6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}