{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/enhancing-temporal-modeling-of-video-llms-via","title":"Enhancing Temporal Modeling of Video LLMs via Time Gating","arxiv_id":"2410.05714","date":"2024-10-08","proceeding":null,"authors":["Zi-Yuan Hu","Yiwu Zhong","Shijia Huang","Michael R. Lyu","LiWei Wang"],"abstract":"Video Large Language Models (Video LLMs) have achieved impressive performance on video-and-language tasks, such as video question answering. However, most existing Video LLMs neglect temporal information in video data, leading to struggles with temporal-aware video understanding. To address this gap, we propose a Time Gating Video LLM (TG-Vid) designed to enhance temporal modeling through a novel Time Gating module (TG). The TG module employs a time gating mechanism on its sub-modules, comprising gating spatial attention, gating temporal attention, and gating MLP. This architecture enables our model to achieve a robust understanding of temporal information within videos. Extensive evaluation of temporal-sensitive video benchmarks (i.e., MVBench, TempCompass, and NExT-QA) demonstrates that our TG-Vid model significantly outperforms the existing Video LLMs. Further, comprehensive ablation studies validate that the performance gains are attributed to the designs of our TG module. Our code is available at https://github.com/LaVi-Lab/TG-Vid.","url_abs":"https://arxiv.org/abs/2410.05714v1","url_pdf":"https://arxiv.org/pdf/2410.05714v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"enhancing-temporal-modeling-of-video-llms-via","repo_url":"https://github.com/lavi-lab/tg-vid","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}}],"tasks":[{"task_slug":null,"task_name":"MVBench"},{"task_slug":"question-answering","task_name":"Question Answering"},{"task_slug":"video-question-answering","task_name":"Video Question Answering"},{"task_slug":"video-understanding","task_name":"Video Understanding"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/2410.05714","atlas_url":"https://app.syntology.ai/?focus=2410.05714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05714"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/lavi-lab/tg-vid","reach":{"status":"ok"}}],"summary":{"ran":2,"ran_fixture":1,"ran_violates":1,"ran_honours":1,"ran_draft_wrong":1,"unverified":3},"by_repo_kind":{"official":{"samples":9,"ran":6,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":9,"samples":[{"code_sha256_prefix":"8bd2d829faa2f7b2","entry":"RandomMaskingGenerator","repo":"lavi-lab/tg-vid","repo_kind":"official","path":"stllm/models/utils.py","file_url":"https://github.com/lavi-lab/tg-vid/blob/HEAD/stllm/models/utils.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"8bd2d829faa2f7b2"}},{"code_sha256_prefix":"9b4dff79d5e6102c","entry":"apply_rotary_pos_emb","repo":"lavi-lab/tg-vid","repo_kind":"official","path":"stllm/models/modeling_llama_mem.py","file_url":"https://github.com/lavi-lab/tg-vid/blob/HEAD/stllm/models/modeling_llama_mem.py","link_basis":"plan_row","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"9b4dff79d5e6102c"}},{"code_sha256_prefix":"4cb732f513d69dfd","entry":"disabled_train","repo":"lavi-lab/tg-vid","repo_kind":"official","path":"stllm/models/blip2.py","file_url":"https://github.com/lavi-lab/tg-vid/blob/HEAD/stllm/models/blip2.py","link_basis":"harvester_set","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"code_sha256_prefix":"da651e3979a18f84","entry":"get_sinusoid_encoding_table","repo":"lavi-lab/tg-vid","repo_kind":"official","path":"stllm/models/utils.py","file_url":"https://github.com/lavi-lab/tg-vid/blob/HEAD/stllm/models/utils.py","link_basis":"plan_row","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"da651e3979a18f84"}},{"code_sha256_prefix":"b99eea6376d1e212","entry":"rotate_half","repo":"lavi-lab/tg-vid","repo_kind":"official","path":"stllm/models/modeling_llama_mem.py","file_url":"https://github.com/lavi-lab/tg-vid/blob/HEAD/stllm/models/modeling_llama_mem.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"b99eea6376d1e212"}},{"code_sha256_prefix":"cb33571427334815","entry":"tile","repo":"lavi-lab/tg-vid","repo_kind":"official","path":"stllm/models/base_model.py","file_url":"https://github.com/lavi-lab/tg-vid/blob/HEAD/stllm/models/base_model.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"cb33571427334815"}},{"code_sha256_prefix":"0ec9fc2025c16f65","entry":"all_gather_with_grad","repo":"lavi-lab/tg-vid","repo_kind":"official","path":"stllm/models/base_model.py","file_url":"https://github.com/lavi-lab/tg-vid/blob/HEAD/stllm/models/base_model.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"code_sha256_prefix":"ce86d2ef6c0e458f","entry":"create_eva_vit_g","repo":"lavi-lab/tg-vid","repo_kind":"official","path":"stllm/models/eva_vit.py","file_url":"https://github.com/lavi-lab/tg-vid/blob/HEAD/stllm/models/eva_vit.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"ce86d2ef6c0e458f"}},{"code_sha256_prefix":"515cc4dd73dc0caf","entry":"forward","repo":"lavi-lab/tg-vid","repo_kind":"official","path":"stllm/models/peft_model.py","file_url":"https://github.com/lavi-lab/tg-vid/blob/HEAD/stllm/models/peft_model.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"515cc4dd73dc0caf"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}