{"url":"/dataset/situated-reasoning-star","name":"STAR Benchmark","full_name":"Situated Reasoning","description_markdown":"How to capture the present knowledge from surrounding situations and perform reasoning accordingly is crucial and challenging for machine intelligence. \r\n[STAR Benchmark](http://star.csail.mit.edu) is a novel benchmark for Situated Reasoning, which provides 60K challenging situated questions in four types of tasks, 140K situated hypergraphs, symbolic situation programs, and logic-grounded diagnosis for real-world video situations. \r\n([Data Download](https://github.com/csbobby/STAR_Benchmark), \r\n[STAR Leaderboard](https://eval.ai/web/challenges/challenge-page/1325/overview))","description_withheld":null,"homepage":"https://star.csail.mit.edu/","introduced_date":"2021-12-06","introduced_date_note":null,"introduced_by":null,"license":{"name":"Apache License 2.0","url":"https://github.com/csbobby/STAR/blob/main/LICENSE"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Video Question Answering","url":"/task/video-question-answering","datasets_with_task":"/datasets/task/video-question-answering"},{"name":"Video Understanding","url":"/task/video-understanding","datasets_with_task":"/datasets/task/video-understanding"},{"name":"Zero-Shot Video Question Answer","url":"/task/zeroshot-video-question-answer","datasets_with_task":"/datasets/task/zeroshot-video-question-answer"},{"name":"Video Grounding","url":"/task/video-grounding","datasets_with_task":"/datasets/task/video-grounding"},{"name":"Video scene graph generation","url":"/task/video-scene-graph-generation","datasets_with_task":"/datasets/task/video-scene-graph-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["STAR Benchmark"],"data_loaders":[],"num_papers_in_archive":17,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-question-answering-on-situated","task":"Video Question Answering","dataset_variant":"STAR Benchmark","rows":17,"metrics":["Average Accuracy"],"first_row_in_archive_order":{"model":"VLAP (4 frames)","paper":"/paper/vlap-efficient-video-language-alignment-via","metrics":{"Average Accuracy":"67.1"},"code_links":[{"title":"xijun-cs/vila","url":"https://github.com/xijun-cs/vila"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-video-question-answer-on-star","task":"Zero-Shot Video Question Answer","dataset_variant":"STAR Benchmark","rows":4,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"VideoChat2","paper":"/paper/mvbench-a-comprehensive-multi-modal-video","metrics":{"Accuracy":"59.0"},"code_links":[{"title":"opengvlab/ask-anything","url":"https://github.com/opengvlab/ask-anything"},{"title":"magic-research/PLLaVA","url":"https://github.com/magic-research/PLLaVA"},{"title":"bytedance/tarsier","url":"https://github.com/bytedance/tarsier"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/vidctx-context-aware-video-question-answering","title":"VidCtx: Context-aware Video Question Answering with Image Models","date":"2024-12-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/traveler-a-multi-lmm-agent-framework-for","title":"TraveLER: A Modular Multi-LMM Agent Framework for Video Question-Answering","date":"2024-04-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/glance-and-focus-memory-prompting-for-multi-1","title":"Glance and Focus: Memory Prompting for Multi-Event Video Question Answering","date":"2024-01-03","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":9,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vlap-efficient-video-language-alignment-via","title":"ViLA: Efficient Video-Language Alignment for Video Question Answering","date":"2023-12-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mvbench-a-comprehensive-multi-modal-video","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","date":"2023-11-28","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":7,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/large-language-models-are-temporal-and-causal","title":"Large Language Models are Temporal and Causal Reasoners for Video Question Answering","date":"2023-10-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/anymal-an-efficient-and-scalable-any-modality","title":"AnyMAL: An Efficient and Scalable Any-Modality Augmented Language Model","date":"2023-09-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/self-chained-image-language-model-for-video-1","title":"Self-Chained Image-Language Model for Video Localization and Question Answering","date":"2023-05-11","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-situation-hyper-graphs-for-video","title":"Learning Situation Hyper-Graphs for Video Question Answering","date":"2023-04-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":8,"samples_unverified":7,"pointer_only_for_licence":15,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mist-multi-modal-iterative-spatial-temporal","title":"MIST: Multi-modal Iterative Spatial-Temporal Transformer for Long-form Video Question Answering","date":"2022-12-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/revisiting-the-video-in-video-language","title":"Revisiting the \"Video\" in Video-Language Understanding","date":"2022-06-03","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/flamingo-a-visual-language-model-for-few-shot-1","title":"Flamingo: a Visual Language Model for Few-Shot Learning","date":"2022-04-29","rows_on_this_dataset":5,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":18,"samples_unverified":6,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/all-in-one-exploring-unified-video-language","title":"All in One: Exploring Unified Video-Language Pre-training","date":"2022-03-14","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":9,"samples_harvested":88,"samples_ran":56,"samples_unverified":32,"pointer_only_for_licence":29,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}