{"url":"/dataset/opens2v-5m","name":"OpenS2V-5M","full_name":null,"description_markdown":"We create the first open-source large-scale S2V generation dataset **OpenS2V-5M**, which consists of five million high-quality \r\n720P subject-text-video triples. To ensure subject-information diversity in our dataset by, we **(1)** segmenting subjects \r\nand building pairing information via cross-video associations and **(2)** prompting GPT-Image on raw frames to synthesize multi-view representations. `The dataset supports both Subject-to-Video and Text-to-Video generation tasks.`","description_withheld":null,"homepage":"https://pku-yuangroup.github.io/OpenS2V-Nexus/","introduced_date":"2025-05-26","introduced_date_note":null,"introduced_by":{"paper":"/paper/opens2v-nexus-a-detailed-benchmark-and","title":"OpenS2V-Nexus: A Detailed Benchmark and Million-Scale Dataset for Subject-to-Video Generation","first_author":"Shenghai Yuan","url":null},"license":{"name":"CC-BY-4.0","url":"https://github.com/PKU-YuanGroup/OpenS2V-Nexus/blob/main/LICENSE"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Video Generation","url":"/task/video-generation","datasets_with_task":"/datasets/task/video-generation"},{"name":"Text-to-Video Generation","url":"/task/text-to-video-generation","datasets_with_task":"/datasets/task/text-to-video-generation"},{"name":"Image to Video Generation","url":"/task/image-to-video","datasets_with_task":"/datasets/task/image-to-video"},{"name":"Open-Domain Subject-to-Video","url":"/task/open-domain-subject-to-video","datasets_with_task":"/datasets/task/open-domain-subject-to-video"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["OpenS2V-5M"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/open-domain-subject-to-video-on-opens2v-5m","task":"Open-Domain Subject-to-Video","dataset_variant":"OpenS2V-5M","rows":0,"metrics":["Total Score"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}