{"url":"/dataset/vinoground","name":"Vinoground","full_name":null,"description_markdown":"A temporal counterfactual dataset composing of 1000 short and natural video-caption pairs.","description_withheld":null,"homepage":"https://huggingface.co/datasets/HanSolo9682/Vinoground","introduced_date":"2024-10-03","introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Temporal Relation Extraction","url":"/task/temporal-relation-extraction","datasets_with_task":"/datasets/task/temporal-relation-extraction"},{"name":"Video Compression","url":"/task/video-compression","datasets_with_task":"/datasets/task/video-compression"},{"name":"Counterfactual Reasoning","url":"/task/counterfactual-reasoning","datasets_with_task":"/datasets/task/counterfactual-reasoning"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Vinoground"],"data_loaders":[],"num_papers_in_archive":17,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/temporal-relation-extraction-on-vinoground","task":"Temporal Relation Extraction","dataset_variant":"Vinoground","rows":24,"metrics":["Text Score","Video Score","Group Score"],"first_row_in_archive_order":{"model":"GPT-4o (CoT)","paper":null,"metrics":{"Group Score":"35","Text Score":"59.2","Video Score":"51"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/qwen2-vl-enhancing-vision-language-model-s","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","date":"2024-09-18","rows_on_this_dataset":2,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":8,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llava-onevision-easy-visual-task-transfer","title":"LLaVA-OneVision: Easy Visual Task Transfer","date":"2024-08-06","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/2408-01800","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","date":"2024-08-03","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":9,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/internlm-xcomposer-2-5-a-versatile-large","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","date":"2024-07-03","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videollama-2-advancing-spatial-temporal","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","date":"2024-06-11","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":7,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ma-lmm-memory-augmented-large-multimodal","title":"MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding","date":"2024-04-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":7,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gemini-1-5-unlocking-multimodal-understanding","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","date":"2024-03-08","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/vtimellm-empower-llm-to-grasp-video-moments","title":"VTimeLLM: Empower LLM to Grasp Video Moments","date":"2023-11-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-llava-learning-united-visual-1","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","date":"2023-11-16","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":4,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/languagebind-extending-video-language","title":"LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment","date":"2023-10-03","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":7,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/imagebind-one-embedding-space-to-bind-them","title":"ImageBind: One Embedding Space To Bind Them All","date":"2023-05-09","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":34,"samples_ran":24,"samples_unverified":10,"pointer_only_for_licence":32,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videoclip-contrastive-pre-training-for-zero","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","date":"2021-09-28","rows_on_this_dataset":1,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":9,"samples_harvested":120,"samples_ran":73,"samples_unverified":47,"pointer_only_for_licence":46,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}