{"url":"/task/video-alignment","name":"Video Alignment","slug":"video-alignment","description_markdown":null,"categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":83,"papers_with_code":43,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/video-alignment-on-msu-video-alignment-and","slug":"video-alignment-on-msu-video-alignment-and","dataset":"MSU Video Alignment and Retrieval Benchmark Suite","dataset_url":"/dataset/msu-video-alignment-and-retrieval-benchmark","rows_in_archive":4,"metrics":["Accuracy w/ 3 frames error (Light)","Accuracy w/ 3 frames error (Medium geometric)","Accuracy w/ 3 frames error (Medium color)","Accuracy w/ 3 frames error (Hard)"],"first_row_in_archive_order":{"model":"VQMT3D","paper_title":"ACCURATE METHOD OF TEMPORAL-SHIFT ESTIMATION FOR 3D VIDEO","paper_url":"/paper/accurate-method-of-temporal-shift-estimation","paper_date":"2018-06-03","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/video-alignment-on-upenn-action","slug":"video-alignment-on-upenn-action","dataset":"UPenn Action","dataset_url":"/dataset/penn-action","rows_in_archive":4,"metrics":["Kendall's Tau"],"first_row_in_archive_order":{"model":"TCC + TCN","paper_title":"Temporal Cycle-Consistency Learning","paper_url":"/paper/temporal-cycle-consistency-learning","paper_date":"2019-04-16","arxiv_id":"1904.07846","code_links":[{"title":"google-research/google-research","url":"https://github.com/google-research/google-research/tree/master/tcc"},{"title":"June01/tcc_Temporal_Cycle_Consistency_Loss.pytorch","url":"https://github.com/June01/tcc_Temporal_Cycle_Consistency_Loss.pytorch"}],"syntology":{"n":12,"n_ran":0,"n_unverified":12,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/penn-action","name":"Penn Action","full_name":"","num_papers_in_archive":110},{"url":"/dataset/n-digit-mnist","name":"N-Digit MNIST","full_name":"","num_papers_in_archive":5},{"url":"/dataset/msu-video-alignment-and-retrieval-benchmark","name":"MSU Video Alignment and Retrieval Benchmark Suite","full_name":"MSU Video Alignment and Retrieval Benchmark Suite","num_papers_in_archive":3},{"url":"/dataset/ikea-assembly-in-the-wild-dataset","name":"IAW Dataset","full_name":"Ikea Assembly In The Wild Dataset","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/video-understanding","name":"Video Understanding"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":43,"tagged_in_all":83,"items":[{"url":"/paper/time-contrastive-networks-self-supervised","title":"Time-Contrastive Networks: Self-Supervised Learning from Video","date":"2017-04-23","arxiv_id":"1704.06888","repositories_listed":7,"syntology":null},{"url":"/paper/hunyuanvideo-a-systematic-framework-for-large","title":"HunyuanVideo: A Systematic Framework For Large Video Generative Models","date":"2024-12-03","arxiv_id":"2412.03603","repositories_listed":2,"syntology":{"n":27,"n_ran":18,"n_unverified":9,"n_pointer_only":27}},{"url":"/paper/e-bench-subjective-aligned-benchmark-suite","title":"VE-Bench: Subjective-Aligned Benchmark Suite for Text-Driven Video Editing Quality Assessment","date":"2024-08-21","arxiv_id":"2408.11481","repositories_listed":2,"syntology":null},{"url":"/paper/cogvideox-text-to-video-diffusion-models-with","title":"CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer","date":"2024-08-12","arxiv_id":"2408.06072","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/miradata-a-large-scale-video-dataset-with","title":"MiraData: A Large-Scale Video Dataset with Long Durations and Structured Captions","date":"2024-07-08","arxiv_id":"2407.06358","repositories_listed":2,"syntology":null},{"url":"/paper/aigcbench-comprehensive-evaluation-of-image","title":"AIGCBench: Comprehensive Evaluation of Image-to-Video Content Generated by AI","date":"2024-01-03","arxiv_id":"2401.01651","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/view-invariant-occlusion-robust-probabilistic","title":"View-Invariant, Occlusion-Robust Probabilistic Embedding for Human Pose","date":"2020-10-23","arxiv_id":"2010.13321","repositories_listed":2,"syntology":null},{"url":"/paper/view-invariant-probabilistic-embedding-for","title":"View-Invariant Probabilistic Embedding for Human Pose","date":"2019-12-02","arxiv_id":"1912.01001","repositories_listed":2,"syntology":null},{"url":"/paper/temporal-cycle-consistency-learning","title":"Temporal Cycle-Consistency Learning","date":"2019-04-16","arxiv_id":"1904.07846","repositories_listed":2,"syntology":{"n":12,"n_ran":0,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/learning-from-video-and-text-via-large-scale","title":"Learning from Video and Text via Large-Scale Discriminative Clustering","date":"2017-07-27","arxiv_id":"1707.09074","repositories_listed":2,"syntology":null},{"url":"/paper/discovla-discrepancy-reduction-in-vision-1","title":"DiscoVLA: Discrepancy Reduction in Vision, Language, and Alignment for Parameter-Efficient Video-Text Retrieval","date":"2025-06-10","arxiv_id":"2506.08887","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/hallo4-high-fidelity-dynamic-portrait","title":"Hallo4: High-Fidelity Dynamic Portrait Animation via Direct Preference Optimization and Temporal Motion Modulation","date":"2025-05-29","arxiv_id":"2505.23525","repositories_listed":1,"syntology":null},{"url":"/paper/love-benchmarking-and-evaluating-text-to","title":"LOVE: Benchmarking and Evaluating Text-to-Video Generation and Video-to-Text Interpretation","date":"2025-05-17","arxiv_id":"2505.12098","repositories_listed":1,"syntology":null},{"url":"/paper/hunyuancustom-a-multimodal-driven","title":"HunyuanCustom: A Multimodal-Driven Architecture for Customized Video Generation","date":"2025-05-07","arxiv_id":"2505.04512","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/video4dgen-enhancing-video-and-4d-generation","title":"Video4DGen: Enhancing Video and 4D Generation through Mutual Optimization","date":"2025-04-05","arxiv_id":"2504.04153","repositories_listed":1,"syntology":null},{"url":"/paper/vrmdiff-text-guided-video-referring-matting","title":"VRMDiff: Text-Guided Video Referring Matting Generation of Diffusion","date":"2025-03-11","arxiv_id":"2503.10678","repositories_listed":1,"syntology":null},{"url":"/paper/deep-understanding-of-sign-language-for-sign","title":"Deep Understanding of Sign Language for Sign to Subtitle Alignment","date":"2025-03-05","arxiv_id":"2503.03287","repositories_listed":1,"syntology":null},{"url":"/paper/inference-time-text-to-video-alignment-with","title":"Inference-Time Text-to-Video Alignment with Diffusion Latent Beam Search","date":"2025-01-31","arxiv_id":"2501.19252","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/sound-bridge-associating-egocentric-and","title":"Sound Bridge: Associating Egocentric and Exocentric Videos via Audio Cues","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/neuro-symbolic-evaluation-of-text-to-video","title":"Neuro-Symbolic Evaluation of Text-to-Video Models using Formal Verification","date":"2024-11-22","arxiv_id":"2411.16718","repositories_listed":1,"syntology":null},{"url":"/paper/t2v-turbo-v2-enhancing-video-generation-model","title":"T2V-Turbo-v2: Enhancing Video Generation Model Post-Training through Data, Reward, and Conditional Guidance Design","date":"2024-10-08","arxiv_id":"2410.05677","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_unverified":0,"n_pointer_only":8}},{"url":"/paper/mamba-enhanced-text-audio-video-alignment","title":"Mamba-Enhanced Text-Audio-Video Alignment Network for Emotion Recognition in Conversations","date":"2024-09-08","arxiv_id":"2409.05243","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-contrastive-learning-for-9","title":"Self-Supervised Contrastive Learning for Videos using Differentiable Local Alignment","date":"2024-09-06","arxiv_id":"2409.04607","repositories_listed":1,"syntology":null},{"url":"/paper/foleycrafter-bring-silent-videos-to-life-with","title":"FoleyCrafter: Bring Silent Videos to Life with Lifelike and Synchronized Sounds","date":"2024-07-01","arxiv_id":"2407.01494","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/safesora-towards-safety-alignment-of","title":"SafeSora: Towards Safety Alignment of Text2Video Generation via a Human Preference Dataset","date":"2024-06-20","arxiv_id":"2406.14477","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":2}},{"url":"/paper/listen-then-see-video-alignment-with-speaker","title":"Listen Then See: Video Alignment with Speaker Attention","date":"2024-04-21","arxiv_id":"2404.13530","repositories_listed":1,"syntology":null},{"url":"/paper/subjective-aligned-dateset-and-metric-for","title":"Subjective-Aligned Dataset and Metric for Text-to-Video Quality Assessment","date":"2024-03-18","arxiv_id":"2403.11956","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":5}},{"url":"/paper/cococo-improving-text-guided-video-inpainting","title":"CoCoCo: Improving Text-Guided Video Inpainting for Better Consistency, Controllability and Compatibility","date":"2024-03-18","arxiv_id":"2403.12035","repositories_listed":1,"syntology":null},{"url":"/paper/evalcrafter-benchmarking-and-evaluating-large","title":"EvalCrafter: Benchmarking and Evaluating Large Video Generation Models","date":"2023-10-17","arxiv_id":"2310.11440","repositories_listed":1,"syntology":{"n":11,"n_ran":5,"n_unverified":6,"n_pointer_only":11}},{"url":"/paper/show-1-marrying-pixel-and-latent-diffusion","title":"Show-1: Marrying Pixel and Latent Diffusion Models for Text-to-Video Generation","date":"2023-09-27","arxiv_id":"2309.15818","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}}],"syntology_records":13,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}