{"url":"/task/video-based-generative-performance-2","name":"Video-based Generative Performance Benchmarking (Detail Orientation))","slug":"video-based-generative-performance-2","description_markdown":"The benchmark evaluates a generative Video Conversational Model with respect to Detail Orientation.\r\n\r\nWe curate a test set based on the ActivityNet-200 dataset, featuring videos with rich, dense descriptive captions and associated question-answer pairs from human annotations. We develop an evaluation pipeline using the GPT-3.5 model that assigns a relative score to the generated predictions on a scale of 1-5.","categories":[{"name":"Reasoning","url":"/area/reasoning"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":15,"papers_with_code":15,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/video-based-generative-performance-4","slug":"video-based-generative-performance-4","dataset":"VideoInstruct","dataset_url":"/dataset/videoinstruct","rows_in_archive":18,"metrics":["gpt-score"],"first_row_in_archive_order":{"model":"PPLLaVA-7B","paper_title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","paper_url":"/paper/ppllava-varied-video-sequence-understanding","paper_date":"2024-11-04","arxiv_id":"2411.02327","code_links":[{"title":"farewellthree/ppllava","url":"https://github.com/farewellthree/ppllava"}],"syntology":{"n":9,"n_ran":2,"n_unverified":7,"n_pointer_only":3}}}],"datasets":[{"url":"/dataset/videoinstruct","name":"VideoInstruct","full_name":"Video Instruction Dataset","num_papers_in_archive":30}],"subtasks":[],"parent_tasks":[{"url":"/task/video-based-generative-performance","name":"Video-based Generative Performance Benchmarking"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":15,"of":15,"tagged_in_all":15,"items":[{"url":"/paper/chat-univi-unified-visual-representation","title":"Chat-UniVi: Unified Visual Representation Empowers Large Language Models with Image and Video Understanding","date":"2023-11-14","arxiv_id":"2311.08046","repositories_listed":4,"syntology":null},{"url":"/paper/video-llama-an-instruction-tuned-audio-visual","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","date":"2023-06-05","arxiv_id":"2306.02858","repositories_listed":4,"syntology":{"n":25,"n_ran":18,"n_unverified":7,"n_pointer_only":9}},{"url":"/paper/mvbench-a-comprehensive-multi-modal-video","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","date":"2023-11-28","arxiv_id":"2311.17005","repositories_listed":3,"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/llama-adapter-v2-parameter-efficient-visual","title":"LLaMA-Adapter V2: Parameter-Efficient Visual Instruction Model","date":"2023-04-28","arxiv_id":"2304.15010","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/minigpt4-video-advancing-multimodal-llms-for","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","date":"2024-04-04","arxiv_id":"2404.03413","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/video-chatgpt-towards-detailed-video","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","date":"2023-06-08","arxiv_id":"2306.05424","repositories_listed":2,"syntology":null},{"url":"/paper/ts-llava-constructing-visual-tokens-through","title":"TS-LLaVA: Constructing Visual Tokens through Thumbnail-and-Sampling for Training-Free Video Large Language Models","date":"2024-11-17","arxiv_id":"2411.11066","repositories_listed":1,"syntology":{"n":11,"n_ran":5,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/ppllava-varied-video-sequence-understanding","title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","date":"2024-11-04","arxiv_id":"2411.02327","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_unverified":7,"n_pointer_only":3}},{"url":"/paper/slowfast-llava-a-strong-training-free","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","date":"2024-07-22","arxiv_id":"2407.15841","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_unverified":2,"n_pointer_only":6}},{"url":"/paper/videogpt-integrating-image-and-video-encoders","title":"VideoGPT+: Integrating Image and Video Encoders for Enhanced Video Understanding","date":"2024-06-13","arxiv_id":"2406.09418","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_unverified":2,"n_pointer_only":8}},{"url":"/paper/pllava-parameter-free-llava-extension-from-1","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","date":"2024-04-25","arxiv_id":"2404.16994","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":2}},{"url":"/paper/vtimellm-empower-llm-to-grasp-video-moments","title":"VTimeLLM: Empower LLM to Grasp Video Moments","date":"2023-11-30","arxiv_id":"2311.18445","repositories_listed":1,"syntology":{"n":11,"n_ran":5,"n_unverified":6,"n_pointer_only":11}},{"url":"/paper/one-for-all-video-conversation-is-feasible","title":"BT-Adapter: Video Conversation is Feasible Without Video Instruction Tuning","date":"2023-09-27","arxiv_id":"2309.15785","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/moviechat-from-dense-token-to-sparse-memory","title":"MovieChat: From Dense Token to Sparse Memory for Long Video Understanding","date":"2023-07-31","arxiv_id":"2307.16449","repositories_listed":1,"syntology":null},{"url":"/paper/videochat-chat-centric-video-understanding","title":"VideoChat: Chat-Centric Video Understanding","date":"2023-05-10","arxiv_id":"2305.06355","repositories_listed":1,"syntology":null}],"syntology_records":11,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}