{"url":"/dataset/opens2v-eval","name":"OpenS2V-Eval","full_name":null,"description_markdown":"**OpenS2V-Eval** introduces 180 prompts from seven major categories of S2V, which incorporate both real and synthetic test data. Furthermore, \r\nto accurately align human preferences with S2V benchmarks, we propose three automatic metrics: **NexusScore**, **NaturalScore**, **GmeScore**\r\nto separately quantify subject consistency, naturalness, and text relevance in generated videos. Building on this, we conduct a comprehensive \r\nevaluation of 14 representative S2V models, highlighting their strengths and weaknesses across different content.","description_withheld":null,"homepage":"https://pku-yuangroup.github.io/OpenS2V-Nexus/","introduced_date":"2025-05-26","introduced_date_note":null,"introduced_by":{"paper":"/paper/opens2v-nexus-a-detailed-benchmark-and","title":"OpenS2V-Nexus: A Detailed Benchmark and Million-Scale Dataset for Subject-to-Video Generation","first_author":"Shenghai Yuan","url":null},"license":{"name":"CC-BY-4.0","url":"https://github.com/PKU-YuanGroup/OpenS2V-Nexus/blob/main/LICENSE"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Video Generation","url":"/task/video-generation","datasets_with_task":"/datasets/task/video-generation"},{"name":"Text-to-Video Generation","url":"/task/text-to-video-generation","datasets_with_task":"/datasets/task/text-to-video-generation"},{"name":"Image to Video Generation","url":"/task/image-to-video","datasets_with_task":"/datasets/task/image-to-video"},{"name":"Open-Domain Subject-to-Video","url":"/task/open-domain-subject-to-video","datasets_with_task":"/datasets/task/open-domain-subject-to-video"},{"name":"Human-Domain Subject-to-Video","url":"/task/human-domain-subject-to-video","datasets_with_task":"/datasets/task/human-domain-subject-to-video"},{"name":"Single-Domain Subject-to-Video","url":"/task/single-domain-subject-to-video","datasets_with_task":"/datasets/task/single-domain-subject-to-video"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["OpenS2V-Eval"],"data_loaders":[],"num_papers_in_archive":8,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/open-domain-subject-to-video-on-opens2v-eval","task":"Open-Domain Subject-to-Video","dataset_variant":"OpenS2V-Eval","rows":10,"metrics":["Total Score","Venue","Aesthetics","Motion","FaceSim","GmeScore","NexusScore","NaturalScore"],"first_row_in_archive_order":{"model":"Kling 1.6","paper":"/paper/identity-preserving-text-to-video-generation","metrics":{"Aesthetics":"0.4460","FaceSim":"0.4010","GmeScore":"0.6620","Motion":"0.4160","NaturalScore":"0.7906","NexusScore":"0.4592","Total Score":"0.5446","Venue":"Close-Source"},"code_links":[{"title":"PKU-YuanGroup/ConsisID","url":"https://github.com/PKU-YuanGroup/ConsisID"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/magref-masked-guidance-for-any-reference","title":"MAGREF: Masked Guidance for Any-Reference Video Generation","date":"2025-05-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":5,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/skyreels-a2-compose-anything-in-video","title":"SkyReels-A2: Compose Anything in Video Diffusion Transformers","date":"2025-04-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vace-all-in-one-video-creation-and-editing","title":"VACE: All-in-One Video Creation and Editing","date":"2025-03-10","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":8,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/phantom-subject-consistent-video-generation","title":"Phantom: Subject-consistent video generation via cross-modal alignment","date":"2025-02-16","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":3,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/identity-preserving-text-to-video-generation","title":"Identity-Preserving Text-to-Video Generation by Frequency Decomposition","date":"2024-11-26","rows_on_this_dataset":3,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":45,"samples_ran":16,"samples_unverified":29,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}