{"url":"/task/video-editing","name":"Video Editing","slug":"video-editing","description_markdown":null,"categories":[{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Graphs","url":"/area/graphs"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":346,"papers_with_code":133,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":1,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/v2vbench","name":"V2VBench","full_name":"","num_papers_in_archive":2},{"url":"/dataset/five","name":"FiVE","full_name":"A Fine-grained Video Editing Benchmark","num_papers_in_archive":1},{"url":"/dataset/vpbench","name":"VPBench","full_name":"","num_papers_in_archive":1},{"url":"/dataset/vpdata","name":"VPData","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/video-temporal-consistency","name":"Video Temporal Consistency"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":133,"tagged_in_all":346,"items":[{"url":"/paper/leo-generative-latent-image-animator-for","title":"LEO: Generative Latent Image Animator for Human Video Synthesis","date":"2023-05-06","arxiv_id":"2305.03989","repositories_listed":5,"syntology":null},{"url":"/paper/soccernet-v2-a-dataset-and-benchmarks-for","title":"SoccerNet-v2: A Dataset and Benchmarks for Holistic Understanding of Broadcast Soccer Videos","date":"2020-11-26","arxiv_id":"2011.13367","repositories_listed":4,"syntology":null},{"url":"/paper/rave-randomized-noise-shuffling-for-fast-and","title":"RAVE: Randomized Noise Shuffling for Fast and Consistent Video Editing with Diffusion Models","date":"2023-12-07","arxiv_id":"2312.04524","repositories_listed":3,"syntology":{"n":9,"n_ran":8,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/point-to-point-video-generation","title":"Point-to-Point Video Generation","date":"2019-04-05","arxiv_id":"1904.02912","repositories_listed":3,"syntology":null},{"url":"/paper/vace-all-in-one-video-creation-and-editing","title":"VACE: All-in-One Video Creation and Editing","date":"2025-03-10","arxiv_id":"2503.07598","repositories_listed":2,"syntology":{"n":19,"n_ran":8,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/re-attentional-controllable-video-diffusion","title":"Re-Attentional Controllable Video Diffusion Editing","date":"2024-12-16","arxiv_id":"2412.11710","repositories_listed":2,"syntology":null},{"url":"/paper/movie-gen-a-cast-of-media-foundation-models","title":"Movie Gen: A Cast of Media Foundation Models","date":"2024-10-17","arxiv_id":"2410.13720","repositories_listed":2,"syntology":null},{"url":"/paper/e-bench-subjective-aligned-benchmark-suite","title":"VE-Bench: Subjective-Aligned Benchmark Suite for Text-Driven Video Editing Quality Assessment","date":"2024-08-21","arxiv_id":"2408.11481","repositories_listed":2,"syntology":null},{"url":"/paper/segment-anything-for-videos-a-systematic","title":"Segment Anything for Videos: A Systematic Survey","date":"2024-07-31","arxiv_id":"2408.08315","repositories_listed":2,"syntology":null},{"url":"/paper/raccoon-remove-add-and-change-video-content","title":"RACCooN: A Versatile Instructional Video Editing Framework with Auto-Generated Narratives","date":"2024-05-28","arxiv_id":"2405.18406","repositories_listed":2,"syntology":{"n":11,"n_ran":5,"n_unverified":6,"n_pointer_only":5}},{"url":"/paper/moonshot-towards-controllable-video","title":"Moonshot: Towards Controllable Video Generation and Editing with Multimodal Conditions","date":"2024-01-03","arxiv_id":"2401.01827","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/fdnerf-semantics-driven-face-reconstruction","title":"FaceDNeRF: Semantics-Driven Face Reconstruction, Prompt Editing and Relighting with Diffusion Models","date":"2023-06-01","arxiv_id":"2306.00783","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/online-misinformation-video-detection-a","title":"Combating Online Misinformation Videos: Characterization, Detection, and Future Directions","date":"2023-02-07","arxiv_id":"2302.03242","repositories_listed":2,"syntology":null},{"url":"/paper/layered-neural-atlases-for-consistent-video","title":"Layered Neural Atlases for Consistent Video Editing","date":"2021-09-23","arxiv_id":"2109.11418","repositories_listed":2,"syntology":{"n":17,"n_ran":2,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/feature-combination-meets-attention-baidu","title":"Feature Combination Meets Attention: Baidu Soccer Embeddings and Transformer based Temporal Detection","date":"2021-06-28","arxiv_id":"2106.14447","repositories_listed":2,"syntology":null},{"url":"/paper/non-rigid-neural-radiance-fields","title":"Non-Rigid Neural Radiance Fields: Reconstruction and Novel View Synthesis of a Dynamic Scene From Monocular Video","date":"2020-12-22","arxiv_id":"2012.12247","repositories_listed":2,"syntology":null},{"url":"/paper/free-form-video-inpainting-with-3d-gated","title":"Free-form Video Inpainting with 3D Gated Convolution and Temporal PatchGAN","date":"2019-04-23","arxiv_id":"1904.10247","repositories_listed":2,"syntology":null},{"url":"/paper/bodynet-volumetric-inference-of-3d-human-body","title":"BodyNet: Volumetric Inference of 3D Human Body Shapes","date":"2018-04-13","arxiv_id":"1804.04875","repositories_listed":2,"syntology":null},{"url":"/paper/causally-steered-diffusion-for-automated","title":"Causally Steered Diffusion for Automated Video Counterfactual Generation","date":"2025-06-17","arxiv_id":"2506.14404","repositories_listed":1,"syntology":null},{"url":"/paper/fade-frequency-aware-diffusion-model-1","title":"FADE: Frequency-Aware Diffusion Model Factorization for Video Editing","date":"2025-06-06","arxiv_id":"2506.05934","repositories_listed":1,"syntology":null},{"url":"/paper/zero-to-hero-zero-shot-initialization","title":"Zero-to-Hero: Zero-Shot Initialization Empowering Reference-Based Video Appearance Editing","date":"2025-05-29","arxiv_id":"2505.23134","repositories_listed":1,"syntology":null},{"url":"/paper/video-editing-for-audio-visual-dubbing","title":"Video Editing for Audio-Visual Dubbing","date":"2025-05-29","arxiv_id":"2505.23406","repositories_listed":1,"syntology":null},{"url":"/paper/dc-sam-in-context-segment-anything-in-images","title":"DC-SAM: In-Context Segment Anything in Images and Videos via Dual Consistency","date":"2025-04-16","arxiv_id":"2504.12080","repositories_listed":1,"syntology":null},{"url":"/paper/flashdepth-real-time-streaming-video-depth","title":"FlashDepth: Real-time Streaming Video Depth Estimation at 2K Resolution","date":"2025-04-09","arxiv_id":"2504.07093","repositories_listed":1,"syntology":null},{"url":"/paper/referdino-plus-2nd-solution-for-4th-pvuw","title":"ReferDINO-Plus: 2nd Solution for 4th PVUW MeViS Challenge at CVPR 2025","date":"2025-03-30","arxiv_id":"2503.23509","repositories_listed":1,"syntology":null},{"url":"/paper/insvie-1m-effective-instruction-based-video","title":"InsViE-1M: Effective Instruction-based Video Editing with Elaborate Dataset Construction","date":"2025-03-26","arxiv_id":"2503.20287","repositories_listed":1,"syntology":{"n":18,"n_ran":1,"n_unverified":17,"n_pointer_only":0}},{"url":"/paper/wan-open-and-advanced-large-scale-video","title":"Wan: Open and Advanced Large-Scale Video Generative Models","date":"2025-03-26","arxiv_id":"2503.20314","repositories_listed":1,"syntology":null},{"url":"/paper/alias-free-latent-diffusion-models-improving","title":"Alias-Free Latent Diffusion Models:Improving Fractional Shift Equivariance of Diffusion Latent Space","date":"2025-03-12","arxiv_id":"2503.09419","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/videopainter-any-length-video-inpainting-and","title":"VideoPainter: Any-length Video Inpainting and Editing with Plug-and-Play Context Control","date":"2025-03-07","arxiv_id":"2503.05639","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-temporally-aware-features-for-point","title":"Exploring Temporally-Aware Features for Point Tracking","date":"2025-01-21","arxiv_id":"2501.12218","repositories_listed":1,"syntology":null}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}