{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-generation/papers/9","list_of":"/task/video-generation","task":"Video Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":15,"rows_per_page":100,"rows":[801,900],"of":1466,"counts":{"archive_papers_tagged":1466,"with_a_code_link":609,"where_syntology_ran_a_sample":257,"not_listed_spam_title":0,"listed":1466,"listed_where_code_ran":257,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":221,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":221,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-generation","prev":"/task/video-generation/papers/8","next":"/task/video-generation/papers/10","papers":[{"url":null,"slug":"cross-modal-learning-for-music-to-music-video","title":"Cross-Modal Learning for Music-to-Music-Video Description Generation","date":"2025-03-14","arxiv_id":"2503.11190","repositories_listed":0,"syntology":null},{"url":null,"slug":"hitvideo-hierarchical-tokenizers-for","title":"HiTVideo: Hierarchical Tokenizers for Enhancing Text-to-Video Generation with Autoregressive Large Language Models","date":"2025-03-14","arxiv_id":"2503.11513","repositories_listed":0,"syntology":null},{"url":"/paper/recammaster-camera-controlled-generative","slug":"recammaster-camera-controlled-generative","title":"ReCamMaster: Camera-Controlled Generative Rendering from A Single Video","date":"2025-03-14","arxiv_id":"2503.11647","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recammaster-camera-controlled-generative#ran","syntology_url":"https://syntology.ai/paper/2503.11647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.11647"}},"official":null}},{"url":null,"slug":"taste-rob-advancing-video-generation-of-task","title":"TASTE-Rob: Advancing Video Generation of Task-Oriented Hand-Object Interaction for Generalizable Robotic Manipulation","date":"2025-03-14","arxiv_id":"2503.11423","repositories_listed":0,"syntology":null},{"url":null,"slug":"cinema-coherent-multi-subject-video","title":"CINEMA: Coherent Multi-Subject Video Generation via MLLM-Based Guidance","date":"2025-03-13","arxiv_id":"2503.10391","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-context-tuning-for-video-generation","title":"Long Context Tuning for Video Generation","date":"2025-03-13","arxiv_id":"2503.10589","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-latent-motion-for-portrait-video","title":"Semantic Latent Motion for Portrait Video Generation","date":"2025-03-13","arxiv_id":"2503.10096","repositories_listed":0,"syntology":null},{"url":null,"slug":"videomerge-towards-training-free-long-video","title":"VideoMerge: Towards Training-free Long Video Generation","date":"2025-03-13","arxiv_id":"2503.09926","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-diffusion-sampling-via","title":"Accelerating Diffusion Sampling via Exploiting Local Transition Coherence","date":"2025-03-12","arxiv_id":"2503.09675","repositories_listed":0,"syntology":null},{"url":null,"slug":"i2v3d-controllable-image-to-video-generation","title":"I2V3D: Controllable image-to-video generation with 3D guidance","date":"2025-03-12","arxiv_id":"2503.09733","repositories_listed":0,"syntology":null},{"url":null,"slug":"lucibot-automated-robot-policy-learning-from","title":"LuciBot: Automated Robot Policy Learning from Generated Videos","date":"2025-03-12","arxiv_id":"2503.09871","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-limitations-of-vision-language-models","title":"On the Limitations of Vision-Language Models in Understanding Image Transforms","date":"2025-03-12","arxiv_id":"2503.09837","repositories_listed":0,"syntology":null},{"url":null,"slug":"other-vehicle-trajectories-are-also-needed-a","title":"Other Vehicle Trajectories Are Also Needed: A Driving World Model Unifies Ego-Other Vehicle Trajectories in Video Latant Space","date":"2025-03-12","arxiv_id":"2503.09215","repositories_listed":0,"syntology":null},{"url":null,"slug":"reangle-a-video-4d-video-generation-as-video","title":"Reangle-A-Video: 4D Video Generation as Video-to-Video Translation","date":"2025-03-12","arxiv_id":"2503.09151","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-dense-prediction-of-video-diffusion","title":"Unified Dense Prediction of Video Diffusion","date":"2025-03-12","arxiv_id":"2503.09344","repositories_listed":0,"syntology":null},{"url":null,"slug":"objectmover-generative-object-movement-with","title":"ObjectMover: Generative Object Movement with Video Prior","date":"2025-03-11","arxiv_id":"2503.08037","repositories_listed":0,"syntology":null},{"url":null,"slug":"wisa-world-simulator-assistant-for-physics","title":"WISA: World Simulator Assistant for Physics-Aware Text-to-Video Generation","date":"2025-03-11","arxiv_id":"2503.08153","repositories_listed":0,"syntology":null},{"url":null,"slug":"dreamrelation-relation-centric-video","title":"DreamRelation: Relation-Centric Video Customization","date":"2025-03-10","arxiv_id":"2503.07602","repositories_listed":0,"syntology":null},{"url":null,"slug":"tr-dq-time-rotation-diffusion-quantization","title":"TR-DQ: Time-Rotation Diffusion Quantization","date":"2025-03-09","arxiv_id":"2503.06564","repositories_listed":0,"syntology":null},{"url":null,"slug":"gsv3d-gaussian-splatting-based-geometric","title":"GSV3D: Gaussian Splatting-based Geometric Distillation with Stable Video Diffusion for Single-Image 3D Object Generation","date":"2025-03-08","arxiv_id":"2503.06136","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-centric-world-model-for-language","title":"Object-Centric World Model for Language-Guided Manipulation","date":"2025-03-08","arxiv_id":"2503.06170","repositories_listed":0,"syntology":null},{"url":null,"slug":"text2story-advancing-video-storytelling-with","title":"Text2Story: Advancing Video Storytelling with Text Guidance","date":"2025-03-08","arxiv_id":"2503.06310","repositories_listed":0,"syntology":null},{"url":null,"slug":"vact-a-video-automatic-causal-testing-system","title":"VACT: A Video Automatic Causal Testing System and a Benchmark","date":"2025-03-08","arxiv_id":"2503.06163","repositories_listed":0,"syntology":null},{"url":null,"slug":"magicinfinite-generating-infinite-talking","title":"MagicInfinite: Generating Infinite Talking Videos with Your Words and Voice","date":"2025-03-07","arxiv_id":"2503.05978","repositories_listed":0,"syntology":null},{"url":null,"slug":"fluidnexus-3d-fluid-reconstruction-and","title":"FluidNexus: 3D Fluid Reconstruction and Prediction from a Single Video","date":"2025-03-06","arxiv_id":"2503.04720","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-super-resolution-all-you-need-is-a","title":"Rethinking Video Super-Resolution: Towards Diffusion-Based Methods without Motion Alignment","date":"2025-03-05","arxiv_id":"2503.03355","repositories_listed":0,"syntology":null},{"url":null,"slug":"haic-improving-human-action-understanding-and","title":"HAIC: Improving Human Action Understanding and Generation with Better Captions for Multi-modal Large Language Models","date":"2025-02-28","arxiv_id":"2502.20811","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-video-action-model","title":"Unified Video Action Model","date":"2025-02-28","arxiv_id":"2503.00200","repositories_listed":0,"syntology":null},{"url":null,"slug":"flexidit-your-diffusion-transformer-can","title":"FlexiDiT: Your Diffusion Transformer Can Easily Generate High-Quality Samples with Less Compute","date":"2025-02-27","arxiv_id":"2502.20126","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-pseudo-average-shifting-attention-pasa","title":"Online Pseudo-average Shifting Attention(PASA) for Robust Low-precision LLM Inference: Algorithms and Numerical Analysis","date":"2025-02-26","arxiv_id":"2503.01873","repositories_listed":0,"syntology":null},{"url":null,"slug":"asurvey-spatiotemporal-consistency-in-video","title":"ASurvey: Spatiotemporal Consistency in Video Generation","date":"2025-02-25","arxiv_id":"2502.17863","repositories_listed":0,"syntology":null},{"url":null,"slug":"videograin-modulating-space-time-attention","title":"VideoGrain: Modulating Space-Time Attention for Multi-grained Video Editing","date":"2025-02-24","arxiv_id":"2502.17258","repositories_listed":0,"syntology":null},{"url":null,"slug":"x-dancer-expressive-music-to-human-dance","title":"X-Dancer: Expressive Music to Human Dance Video Generation","date":"2025-02-24","arxiv_id":"2502.17414","repositories_listed":0,"syntology":null},{"url":"/paper/riflex-a-free-lunch-for-length-extrapolation","slug":"riflex-a-free-lunch-for-length-extrapolation","title":"RIFLEx: A Free Lunch for Length Extrapolation in Video Diffusion Transformers","date":"2025-02-21","arxiv_id":"2502.15894","repositories_listed":0,"syntology":{"n":7,"n_ran":6,"n_constructed":1,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":1,"n_no_contract":1,"n_pointer_only":4,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 1 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/riflex-a-free-lunch-for-length-extrapolation#ran","syntology_url":"https://syntology.ai/paper/2502.15894","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15894"}},"official":null}},{"url":null,"slug":"designing-parameter-and-compute-efficient","title":"Designing Parameter and Compute Efficient Diffusion Transformers using Distillation","date":"2025-02-20","arxiv_id":"2502.14226","repositories_listed":0,"syntology":null},{"url":null,"slug":"hardware-friendly-static-quantization-method","title":"Hardware-Friendly Static Quantization Method for Video Diffusion Transformers","date":"2025-02-20","arxiv_id":"2502.15077","repositories_listed":0,"syntology":null},{"url":"/paper/improving-the-diffusability-of-autoencoders","slug":"improving-the-diffusability-of-autoencoders","title":"Improving the Diffusability of Autoencoders","date":"2025-02-20","arxiv_id":"2502.14831","repositories_listed":0,"syntology":null},{"url":null,"slug":"relactrl-relevance-guided-efficient-control","title":"RelaCtrl: Relevance-Guided Efficient Control for Diffusion Transformers","date":"2025-02-20","arxiv_id":"2502.14377","repositories_listed":0,"syntology":null},{"url":null,"slug":"llmpopcorn-an-empirical-study-of-llms-as","title":"LLMPopcorn: An Empirical Study of LLMs as Assistants for Popular Micro-video Generation","date":"2025-02-18","arxiv_id":"2502.12945","repositories_listed":0,"syntology":null},{"url":null,"slug":"malt-diffusion-memory-augmented-latent","title":"MALT Diffusion: Memory-Augmented Latent Transformers for Any-Length Video Generation","date":"2025-02-18","arxiv_id":"2502.12632","repositories_listed":0,"syntology":null},{"url":null,"slug":"maskflow-discrete-flows-for-flexible-and","title":"MaskFlow: Discrete Flows For Flexible and Efficient Long Video Generation","date":"2025-02-16","arxiv_id":"2502.11234","repositories_listed":0,"syntology":null},{"url":null,"slug":"realcam-i2v-real-world-image-to-video","title":"RealCam-I2V: Real-World Image-to-Video Generation with Interactive Complex Camera Control","date":"2025-02-14","arxiv_id":"2502.10059","repositories_listed":0,"syntology":null},{"url":null,"slug":"gevrm-goal-expressive-video-generation-model","title":"GEVRM: Goal-Expressive Video Generation Model For Robust Visual Manipulation","date":"2025-02-13","arxiv_id":"2502.09268","repositories_listed":0,"syntology":null},{"url":null,"slug":"anycharv-bootstrap-controllable-character","title":"AnyCharV: Bootstrap Controllable Character Video Generation with Fine-to-Coarse Guidance","date":"2025-02-12","arxiv_id":"2502.08189","repositories_listed":0,"syntology":null},{"url":null,"slug":"cinemaster-a-3d-aware-and-controllable","title":"CineMaster: A 3D-Aware and Controllable Framework for Cinematic Text-to-Video Generation","date":"2025-02-12","arxiv_id":"2502.08639","repositories_listed":0,"syntology":null},{"url":null,"slug":"flovd-optical-flow-meets-video-diffusion","title":"FloVD: Optical Flow Meets Video Diffusion Model for Enhanced Camera-Controlled Video Synthesis","date":"2025-02-12","arxiv_id":"2502.08244","repositories_listed":0,"syntology":null},{"url":null,"slug":"articulate-that-object-part-atop-3d-part","title":"Articulate That Object Part (ATOP): 3D Part Articulation from Text and Motion Personalization","date":"2025-02-11","arxiv_id":"2502.07278","repositories_listed":0,"syntology":null},{"url":"/paper/contextual-gesture-co-speech-gesture-video","slug":"contextual-gesture-co-speech-gesture-video","title":"Contextual Gesture: Co-Speech Gesture Video Generation through Context-aware Gesture Representation","date":"2025-02-11","arxiv_id":"2502.07239","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-ghost-investigating-ranking-bias","title":"Generative Ghost: Investigating Ranking Bias Hidden in AI-Generated Videos","date":"2025-02-11","arxiv_id":"2502.07327","repositories_listed":0,"syntology":null},{"url":null,"slug":"next-block-prediction-video-generation-via","title":"Next Block Prediction: Video Generation via Semi-Auto-Regressive Modeling","date":"2025-02-11","arxiv_id":"2502.07737","repositories_listed":0,"syntology":null},{"url":null,"slug":"vidcraft3-camera-object-and-lighting-control","title":"VidCRAFT3: Camera, Object, and Lighting Control for Image-to-Video Generation","date":"2025-02-11","arxiv_id":"2502.07531","repositories_listed":0,"syntology":null},{"url":null,"slug":"customvideox-3d-reference-attention-driven","title":"CustomVideoX: 3D Reference Attention Driven Dynamic Adaptation for Zero-Shot Customized Video Diffusion Transformers","date":"2025-02-10","arxiv_id":"2502.06527","repositories_listed":0,"syntology":null},{"url":null,"slug":"senorita-2m-a-high-quality-instruction-based","title":"Señorita-2M: A High-Quality Instruction-based Dataset for General Video Editing by Video Specialists","date":"2025-02-10","arxiv_id":"2502.06734","repositories_listed":0,"syntology":null},{"url":null,"slug":"humandit-pose-guided-diffusion-transformer","title":"HumanDiT: Pose-Guided Diffusion Transformer for Long-form Human Motion Video Generation","date":"2025-02-07","arxiv_id":"2502.04847","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-real-world-action-video-dynamics","title":"Learning Real-World Action-Video Dynamics with Heterogeneous Masked Autoregression","date":"2025-02-06","arxiv_id":"2502.04296","repositories_listed":0,"syntology":null},{"url":null,"slug":"motioncanvas-cinematic-shot-design-with","title":"MotionCanvas: Cinematic Shot Design with Controllable Image-to-Video Generation","date":"2025-02-06","arxiv_id":"2502.04299","repositories_listed":0,"syntology":null},{"url":null,"slug":"unicp-a-unified-caching-and-pruning-framework","title":"UniCP: A Unified Caching and Pruning Framework for Efficient Video Generation","date":"2025-02-06","arxiv_id":"2502.04393","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniform-a-unified-diffusion-transformer-for","title":"UniForm: A Unified Multi-Task Diffusion Transformer for Audio-Video Generation","date":"2025-02-06","arxiv_id":"2502.03897","repositories_listed":0,"syntology":null},{"url":null,"slug":"freqprior-improving-video-diffusion-models","title":"FreqPrior: Improving Video Diffusion Models with Frequency Filtering Gaussian Noise","date":"2025-02-05","arxiv_id":"2502.03496","repositories_listed":0,"syntology":null},{"url":null,"slug":"motionagent-fine-grained-controllable-video","title":"MotionAgent: Fine-grained Controllable Video Generation via Motion Field Agent","date":"2025-02-05","arxiv_id":"2502.03207","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-physical-understanding-in-video","title":"Towards Physical Understanding in Video Generation: A 3D Point Regularization Approach","date":"2025-02-05","arxiv_id":"2502.03639","repositories_listed":0,"syntology":null},{"url":null,"slug":"harness-local-rewards-for-global-benefits","title":"Harness Local Rewards for Global Benefits: Effective Text-to-Video Generation Alignment with Patch-level Reward Models","date":"2025-02-04","arxiv_id":"2502.06812","repositories_listed":0,"syntology":null},{"url":null,"slug":"ipo-iterative-preference-optimization-for","title":"IPO: Iterative Preference Optimization for Text-to-Video Generation","date":"2025-02-04","arxiv_id":"2502.02088","repositories_listed":0,"syntology":null},{"url":null,"slug":"videojam-joint-appearance-motion","title":"VideoJAM: Joint Appearance-Motion Representations for Enhanced Motion Generation in Video Models","date":"2025-02-04","arxiv_id":"2502.02492","repositories_listed":0,"syntology":null},{"url":null,"slug":"mj-video-fine-grained-benchmarking-and","title":"MJ-VIDEO: Fine-Grained Benchmarking and Rewarding Video Preferences in Video Generation","date":"2025-02-03","arxiv_id":"2502.01719","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnihuman-1-rethinking-the-scaling-up-of-one","title":"OmniHuman-1: Rethinking the Scaling-Up of One-Stage Conditioned Human Animation Models","date":"2025-02-03","arxiv_id":"2502.01061","repositories_listed":0,"syntology":null},{"url":null,"slug":"secure-personalized-music-to-video-generation","title":"Secure & Personalized Music-to-Video Generation via CHARCHA","date":"2025-02-03","arxiv_id":"2502.02610","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-videogen-accelerating-video-diffusion","title":"Sparse VideoGen: Accelerating Video Diffusion Transformers with Spatial-Temporal Sparsity","date":"2025-02-03","arxiv_id":"2502.01776","repositories_listed":0,"syntology":null},{"url":null,"slug":"huvidpo-enhancing-video-generation-through","title":"HuViDPO:Enhancing Video Generation through Direct Preference Optimization for Human-Centric Alignment","date":"2025-02-02","arxiv_id":"2502.01690","repositories_listed":0,"syntology":null},{"url":null,"slug":"zeroth-order-informed-fine-tuning-for","title":"Zeroth-order Informed Fine-Tuning for Diffusion Model: A Recursive Likelihood Ratio Optimizer","date":"2025-02-02","arxiv_id":"2502.00639","repositories_listed":0,"syntology":null},{"url":null,"slug":"shape-from-semantics-3d-shape-generation-from","title":"Shape from Semantics: 3D Shape Generation from Multi-View Semantics","date":"2025-02-01","arxiv_id":"2502.00360","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-latent-flow-matching-optimal-polynomial","title":"Video Latent Flow Matching: Optimal Polynomial Projections for Video Interpolation and Extrapolation","date":"2025-02-01","arxiv_id":"2502.00500","repositories_listed":0,"syntology":null},{"url":null,"slug":"every-image-listens-every-image-dances-music","title":"Every Image Listens, Every Image Dances: Music-Driven Image Animation","date":"2025-01-30","arxiv_id":"2501.18801","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-video-generation-with-human","title":"Improving Video Generation with Human Feedback","date":"2025-01-23","arxiv_id":"2501.13918","repositories_listed":0,"syntology":null},{"url":null,"slug":"taming-teacher-forcing-for-masked","title":"Taming Teacher Forcing for Masked Autoregressive Video Generation","date":"2025-01-21","arxiv_id":"2501.12389","repositories_listed":0,"syntology":null},{"url":null,"slug":"genvidbench-a-challenging-benchmark-for","title":"GenVidBench: A Challenging Benchmark for Detecting AI-Generated Video","date":"2025-01-20","arxiv_id":"2501.11340","repositories_listed":0,"syntology":null},{"url":null,"slug":"emo2-end-effector-guided-audio-driven-avatar","title":"EMO2: End-Effector Guided Audio-Driven Avatar Video Generation","date":"2025-01-18","arxiv_id":"2501.10687","repositories_listed":0,"syntology":null},{"url":null,"slug":"richspace-enriching-text-to-video-prompt","title":"RichSpace: Enriching Text-to-Video Prompt Space via Text Embedding Interpolation","date":"2025-01-17","arxiv_id":"2501.09982","repositories_listed":0,"syntology":null},{"url":null,"slug":"learnings-from-scaling-visual-tokenizers-for","title":"Learnings from Scaling Visual Tokenizers for Reconstruction and Generation","date":"2025-01-16","arxiv_id":"2501.09755","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoworld-exploring-knowledge-learning-from","title":"VideoWorld: Exploring Knowledge Learning from Unlabeled Videos","date":"2025-01-16","arxiv_id":"2501.09781","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-subjective-and-objective","title":"Comprehensive Subjective and Objective Evaluation Method for Text-generated Video","date":"2025-01-15","arxiv_id":"2501.08545","repositories_listed":0,"syntology":null},{"url":null,"slug":"ouroboros-diffusion-exploring-consistent","title":"Ouroboros-Diffusion: Exploring Consistent Content Generation in Tuning-free Long Video Diffusion","date":"2025-01-15","arxiv_id":"2501.09019","repositories_listed":0,"syntology":null},{"url":null,"slug":"repvideo-rethinking-cross-layer","title":"RepVideo: Rethinking Cross-Layer Representation for Video Generation","date":"2025-01-15","arxiv_id":"2501.08994","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-gaussian-splatting-with-normal-information","title":"3D Gaussian Splatting with Normal Information for Mesh Extraction and Improved Rendering","date":"2025-01-14","arxiv_id":"2501.08370","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-adversarial-post-training-for-one","title":"Diffusion Adversarial Post-Training for One-Step Video Generation","date":"2025-01-14","arxiv_id":"2501.08316","repositories_listed":0,"syntology":null},{"url":null,"slug":"gamefactory-creating-new-games-with","title":"GameFactory: Creating New Games with Generative Interactive Videos","date":"2025-01-14","arxiv_id":"2501.08325","repositories_listed":0,"syntology":null},{"url":null,"slug":"layeranimate-layer-specific-control-for","title":"LayerAnimate: Layer-specific Control for Animation","date":"2025-01-14","arxiv_id":"2501.08295","repositories_listed":0,"syntology":null},{"url":null,"slug":"blobgen-vid-compositional-text-to-video","title":"BlobGEN-Vid: Compositional Text-to-Video Generation with Blob Video Representations","date":"2025-01-13","arxiv_id":"2501.07647","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-free-motion-guided-video-generation","title":"Training-Free Motion-Guided Video Generation with Enhanced Temporal Consistency Using Motion Consistency Loss","date":"2025-01-13","arxiv_id":"2501.07563","repositories_listed":0,"syntology":null},{"url":null,"slug":"heterollm-accelerating-large-language-model","title":"HeteroLLM: Accelerating Large Language Model Inference on Mobile SoCs platform with Heterogeneous AI Accelerators","date":"2025-01-11","arxiv_id":"2501.14794","repositories_listed":0,"syntology":null},{"url":null,"slug":"qffusion-controllable-portrait-video-editing","title":"Qffusion: Controllable Portrait Video Editing via Quadrant-Grid Attention Learning","date":"2025-01-11","arxiv_id":"2501.06438","repositories_listed":0,"syntology":null},{"url":null,"slug":"met3r-measuring-multi-view-consistency-in","title":"MEt3R: Measuring Multi-View Consistency in Generated Images","date":"2025-01-10","arxiv_id":"2501.06336","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-subject-open-set-personalization-in","title":"Multi-subject Open-set Personalization in Video Generation","date":"2025-01-10","arxiv_id":"2501.06187","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoauteur-towards-long-narrative-video","title":"VideoAuteur: Towards Long Narrative Video Generation","date":"2025-01-10","arxiv_id":"2501.06173","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-growing-of-video-tokenizers-for","title":"Progressive Growing of Video Tokenizers for Highly Compressed Latent Spaces","date":"2025-01-09","arxiv_id":"2501.05442","repositories_listed":0,"syntology":null},{"url":null,"slug":"conceptmaster-multi-concept-video","title":"ConceptMaster: Multi-Concept Video Customization on Diffusion Transformer Models Without Test-Time Tuning","date":"2025-01-08","arxiv_id":"2501.04698","repositories_listed":0,"syntology":null},{"url":null,"slug":"lipgen-viseme-guided-lip-video-generation-for","title":"LipGen: Viseme-Guided Lip Video Generation for Enhancing Visual Speech Recognition","date":"2025-01-08","arxiv_id":"2501.04204","repositories_listed":0,"syntology":null},{"url":null,"slug":"tuning-free-long-video-generation-via-global","title":"Tuning-Free Long Video Generation via Global-Local Collaborative Diffusion","date":"2025-01-08","arxiv_id":"2501.05484","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-aware-generative-frame-interpolation","title":"Motion-Aware Generative Frame Interpolation","date":"2025-01-07","arxiv_id":"2501.03699","repositories_listed":0,"syntology":null},{"url":null,"slug":"brick-diffusion-generating-long-videos-with","title":"Brick-Diffusion: Generating Long Videos with Brick-to-Wall Denoising","date":"2025-01-06","arxiv_id":"2501.02741","repositories_listed":0,"syntology":null}],"record_sha256":"3275b3ddcf214638270d148967123d92b55bc89aa333b04d5f1f5ec047332c2d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}