{"url":"/sota/video-based-generative-performance-1","task":{"name":"Video-based Generative Performance Benchmarking (Correctness of Information)","url":"/task/video-based-generative-performance-1","note":null},"dataset":{"name":"VideoInstruct","url":"/dataset/videoinstruct"},"category":"Reasoning","categories":["Reasoning"],"category_note":null,"description":"The benchmark evaluates a generative Video Conversational Model with respect to Correctness of Information.\r\n\r\nWe curate a test set based on the ActivityNet-200 dataset, featuring videos with rich, dense descriptive captions and associated question-answer pairs from human annotations. We develop an evaluation pipeline using the GPT-3.5 model that assigns a relative score to the generated predictions on a scale of 1-5.","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["gpt-score"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"gpt-score":"higher"}},"counts":{"rows":18,"rows_with_code":18,"rows_with_paper_page":18,"rows_dated":18,"rows_using_additional_data":0},"rows":[{"rank_in_archive_order":1,"model":"PPLLaVA-7B","metrics":{"gpt-score":"3.85"},"uses_additional_data":false,"paper_date":"2024-11-04","paper":"/paper/ppllava-varied-video-sequence-understanding","paper_url":"https://arxiv.org/abs/2411.02327v2","paper_title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","code":"https://github.com/farewellthree/ppllava","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":7,"n_samples":9,"n_pointer_only_licence":3}},{"rank_in_archive_order":2,"model":"PLLaVA-34B","metrics":{"gpt-score":"3.60"},"uses_additional_data":false,"paper_date":"2024-04-25","paper":"/paper/pllava-parameter-free-llava-extension-from-1","paper_url":"https://arxiv.org/abs/2404.16994v2","paper_title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","code":"https://github.com/magic-research/PLLaVA","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":2,"n_samples":2,"n_pointer_only_licence":2}},{"rank_in_archive_order":3,"model":"TS-LLaVA-34B","metrics":{"gpt-score":"3.55"},"uses_additional_data":false,"paper_date":"2024-11-17","paper":"/paper/ts-llava-constructing-visual-tokens-through","paper_url":"https://arxiv.org/abs/2411.11066v1","paper_title":"TS-LLaVA: Constructing Visual Tokens through Thumbnail-and-Sampling for Training-Free Video Large Language Models","code":"https://github.com/tingyu215/ts-llava","n_code_links":1,"syntology":{"n_ran":5,"n_unverified":6,"n_samples":11,"n_pointer_only_licence":0}},{"rank_in_archive_order":4,"model":"SlowFast-LLaVA-34B","metrics":{"gpt-score":"3.48"},"uses_additional_data":false,"paper_date":"2024-07-22","paper":"/paper/slowfast-llava-a-strong-training-free","paper_url":"https://arxiv.org/abs/2407.15841v2","paper_title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","code":"https://github.com/apple/ml-slowfast-llava","n_code_links":1,"syntology":{"n_ran":4,"n_unverified":2,"n_samples":6,"n_pointer_only_licence":6}},{"rank_in_archive_order":5,"model":"VideoChat2_HD_mistral","metrics":{"gpt-score":"3.40"},"uses_additional_data":false,"paper_date":"2023-11-28","paper":"/paper/mvbench-a-comprehensive-multi-modal-video","paper_url":"https://arxiv.org/abs/2311.17005v4","paper_title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","code":"https://github.com/opengvlab/ask-anything","n_code_links":3,"syntology":{"n_ran":7,"n_unverified":3,"n_samples":10,"n_pointer_only_licence":0}},{"rank_in_archive_order":6,"model":"VideoGPT+","metrics":{"gpt-score":"3.27"},"uses_additional_data":false,"paper_date":"2024-06-13","paper":"/paper/videogpt-integrating-image-and-video-encoders","paper_url":"https://arxiv.org/abs/2406.09418v1","paper_title":"VideoGPT+: Integrating Image and Video Encoders for Enhanced Video Understanding","code":"https://github.com/mbzuai-oryx/videogpt-plus","n_code_links":1,"syntology":{"n_ran":6,"n_unverified":2,"n_samples":8,"n_pointer_only_licence":8}},{"rank_in_archive_order":7,"model":"ST-LLM","metrics":{"gpt-score":"3.23"},"uses_additional_data":false,"paper_date":"2024-03-30","paper":"/paper/st-llm-large-language-models-are-effective-1","paper_url":"https://arxiv.org/abs/2404.00308v1","paper_title":"ST-LLM: Large Language Models Are Effective Temporal Learners","code":"https://github.com/TencentARC/ST-LLM","n_code_links":1,"syntology":{"n_ran":7,"n_unverified":4,"n_samples":11,"n_pointer_only_licence":3}},{"rank_in_archive_order":8,"model":"MiniGPT4-video-7B","metrics":{"gpt-score":"3.08"},"uses_additional_data":false,"paper_date":"2024-04-04","paper":"/paper/minigpt4-video-advancing-multimodal-llms-for","paper_url":"https://arxiv.org/abs/2404.03413v1","paper_title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","code":"https://github.com/Vision-CAIR/MiniGPT4-video","n_code_links":2,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":0}},{"rank_in_archive_order":9,"model":"VideoChat2","metrics":{"gpt-score":"3.02"},"uses_additional_data":false,"paper_date":"2023-11-28","paper":"/paper/mvbench-a-comprehensive-multi-modal-video","paper_url":"https://arxiv.org/abs/2311.17005v4","paper_title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","code":"https://github.com/opengvlab/ask-anything","n_code_links":3,"syntology":{"n_ran":7,"n_unverified":3,"n_samples":10,"n_pointer_only_licence":0}},{"rank_in_archive_order":10,"model":"Chat-UniVi","metrics":{"gpt-score":"2.89"},"uses_additional_data":false,"paper_date":"2023-11-14","paper":"/paper/chat-univi-unified-visual-representation","paper_url":"https://arxiv.org/abs/2311.08046v3","paper_title":"Chat-UniVi: Unified Visual Representation Empowers Large Language Models with Image and Video Understanding","code":"https://github.com/pku-yuangroup/chat-univi","n_code_links":4,"syntology":null},{"rank_in_archive_order":11,"model":"VTimeLLM","metrics":{"gpt-score":"2.78"},"uses_additional_data":false,"paper_date":"2023-11-30","paper":"/paper/vtimellm-empower-llm-to-grasp-video-moments","paper_url":"https://arxiv.org/abs/2311.18445v1","paper_title":"VTimeLLM: Empower LLM to Grasp Video Moments","code":"https://github.com/huangb23/vtimellm","n_code_links":1,"syntology":{"n_ran":5,"n_unverified":6,"n_samples":11,"n_pointer_only_licence":11}},{"rank_in_archive_order":12,"model":"MovieChat","metrics":{"gpt-score":"2.76"},"uses_additional_data":false,"paper_date":"2023-07-31","paper":"/paper/moviechat-from-dense-token-to-sparse-memory","paper_url":"https://arxiv.org/abs/2307.16449v4","paper_title":"MovieChat: From Dense Token to Sparse Memory for Long Video Understanding","code":"https://github.com/rese1f/MovieChat","n_code_links":1,"syntology":null},{"rank_in_archive_order":13,"model":"BT-Adapter","metrics":{"gpt-score":"2.68"},"uses_additional_data":false,"paper_date":"2023-09-27","paper":"/paper/one-for-all-video-conversation-is-feasible","paper_url":"https://arxiv.org/abs/2309.15785v2","paper_title":"BT-Adapter: Video Conversation is Feasible Without Video Instruction Tuning","code":"https://github.com/farewellthree/BT-Adapter","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":3,"n_samples":4,"n_pointer_only_licence":0}},{"rank_in_archive_order":14,"model":"Video-ChatGPT","metrics":{"gpt-score":"2.40"},"uses_additional_data":false,"paper_date":"2023-06-08","paper":"/paper/video-chatgpt-towards-detailed-video","paper_url":"https://arxiv.org/abs/2306.05424v2","paper_title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","code":"https://github.com/mbzuai-oryx/video-chatgpt","n_code_links":2,"syntology":null},{"rank_in_archive_order":15,"model":"Video Chat","metrics":{"gpt-score":"2.32"},"uses_additional_data":false,"paper_date":"2023-05-10","paper":"/paper/videochat-chat-centric-video-understanding","paper_url":"https://arxiv.org/abs/2305.06355v2","paper_title":"VideoChat: Chat-Centric Video Understanding","code":"https://github.com/opengvlab/ask-anything","n_code_links":1,"syntology":null},{"rank_in_archive_order":16,"model":"BT-Adapter (zero-shot)","metrics":{"gpt-score":"2.16"},"uses_additional_data":false,"paper_date":"2023-09-27","paper":"/paper/one-for-all-video-conversation-is-feasible","paper_url":"https://arxiv.org/abs/2309.15785v2","paper_title":"BT-Adapter: Video Conversation is Feasible Without Video Instruction Tuning","code":"https://github.com/farewellthree/BT-Adapter","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":3,"n_samples":4,"n_pointer_only_licence":0}},{"rank_in_archive_order":17,"model":"LLaMA Adapter","metrics":{"gpt-score":"2.03"},"uses_additional_data":false,"paper_date":"2023-04-28","paper":"/paper/llama-adapter-v2-parameter-efficient-visual","paper_url":"https://arxiv.org/abs/2304.15010v1","paper_title":"LLaMA-Adapter V2: Parameter-Efficient Visual Instruction Model","code":"https://github.com/opengvlab/llama-adapter","n_code_links":3,"syntology":{"n_ran":0,"n_unverified":1,"n_samples":1,"n_pointer_only_licence":1}},{"rank_in_archive_order":18,"model":"Video LLaMA","metrics":{"gpt-score":"1.96"},"uses_additional_data":false,"paper_date":"2023-06-05","paper":"/paper/video-llama-an-instruction-tuned-audio-visual","paper_url":"https://arxiv.org/abs/2306.02858v4","paper_title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","code":"https://github.com/damo-nlp-sg/video-llama","n_code_links":4,"syntology":{"n_ran":18,"n_unverified":7,"n_samples":25,"n_pointer_only_licence":9}}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":14,"rows_with_any_sample_ran":12,"distinct_papers_with_graph_line":12,"distinct_papers_with_any_sample_ran":10,"samples_over_distinct_papers":{"n_ran":57,"n_unverified":43,"n_samples":100,"n_pointer_only_licence":43,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":65,"n_unverified":49,"n_samples":114,"n_pointer_only_licence":43,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}