{"url":"/dataset/vlep","name":"VLEP","full_name":"Video-and-Language Event Prediction","description_markdown":"VLEP contains 28,726 future event prediction examples (along with their rationales) from 10,234 diverse TV Show and YouTube Lifestyle Vlog video clips. Each example (see Figure 1) consists of a Premise Event (a short video clip with dialogue), a Premise Summary (a text summary of the premise event), and two potential natural language Future Events (along with Rationales) written by people. These clips are on average 6.1 seconds long and are harvested from diverse event-rich sources, i.e., TV show and YouTube Lifestyle Vlog videos.\r\n\r\nSource: [What is More Likely to Happen Next? Video-and-Language Future Event Prediction](https://arxiv.org/pdf/2010.07999.pdf)","description_withheld":null,"homepage":"https://github.com/jayleicn/VideoLanguageFuturePred/blob/main/data/README.md","introduced_date":"2020-10-15","introduced_date_note":null,"introduced_by":{"paper":"/paper/what-is-more-likely-to-happen-next-video-and","title":"What is More Likely to Happen Next? Video-and-Language Future Event Prediction","first_author":"Jie Lei","url":null},"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Video Question Answering","url":"/task/video-question-answering","datasets_with_task":"/datasets/task/video-question-answering"}],"languages":[],"variants":["VLEP"],"data_loaders":[{"repo":"https://github.com/jayleicn/VideoLanguageFuturePred","url":"https://github.com/jayleicn/VideoLanguageFuturePred","frameworks":[]}],"num_papers_in_archive":11,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-question-answering-on-vlep","task":"Video Question Answering","dataset_variant":"VLEP","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"LLaMA-VQA","paper":"/paper/large-language-models-are-temporal-and-causal","metrics":{"Accuracy":"71.0"},"code_links":[{"title":"mlvlab/Flipped-VQA","url":"https://github.com/mlvlab/Flipped-VQA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/large-language-models-are-temporal-and-causal","title":"Large Language Models are Temporal and Causal Reasoners for Video Question Answering","date":"2023-10-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}