{"url":"/task/video-text-retrieval","name":"Video-Text Retrieval","slug":"video-text-retrieval","description_markdown":"Video-Text retrieval requires understanding of both video and language together. Therefore it's different to video retrieval task.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":111,"papers_with_code":59,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":6,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/video-text-retrieval-on-test-of-time","slug":"video-text-retrieval-on-test-of-time","dataset":"Test-of-Time","dataset_url":"/dataset/test-of-time","rows_in_archive":4,"metrics":["2-Class Accuracy"],"first_row_in_archive_order":{"model":"Video-LLAMA","paper_title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","paper_url":"/paper/video-llama-an-instruction-tuned-audio-visual","paper_date":"2023-06-05","arxiv_id":"2306.02858","code_links":[{"title":"damo-nlp-sg/video-llama","url":"https://github.com/damo-nlp-sg/video-llama"},{"title":"damo-nlp-sg/videollama2","url":"https://github.com/damo-nlp-sg/videollama2"},{"title":"damo-nlp-sg/videollama3","url":"https://github.com/damo-nlp-sg/videollama3"},{"title":"xinding-sys/StreamMind","url":"https://github.com/xinding-sys/StreamMind"}],"syntology":{"n":25,"n_ran":18,"n_unverified":7,"n_pointer_only":9}}}],"datasets":[{"url":"/dataset/webvid","name":"WebVid","full_name":"","num_papers_in_archive":257},{"url":"/dataset/test-of-time","name":"Test-of-Time","full_name":"Test of Time Synthetic Video Dataset","num_papers_in_archive":5},{"url":"/dataset/symon","name":"SYMON","full_name":"Synopses of Movie Narratives","num_papers_in_archive":4},{"url":"/dataset/youku-mplug","name":"Youku-mPLUG","full_name":"","num_papers_in_archive":4},{"url":"/dataset/vtc","name":"VTC","full_name":"Videos, Titles and Comments","num_papers_in_archive":2},{"url":"/dataset/wonderbread","name":"wonderbread","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/video-retrieval","name":"Video Retrieval"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":59,"tagged_in_all":111,"items":[{"url":"/paper/languagebind-extending-video-language","title":"LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment","date":"2023-10-03","arxiv_id":"2310.01852","repositories_listed":6,"syntology":{"n":14,"n_ran":7,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/clip4clip-an-empirical-study-of-clip-for-end","title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval","date":"2021-04-18","arxiv_id":"2104.08860","repositories_listed":5,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/frozen-in-time-a-joint-video-and-image","title":"Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval","date":"2021-04-01","arxiv_id":"2104.00650","repositories_listed":5,"syntology":{"n":11,"n_ran":3,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/video-llama-an-instruction-tuned-audio-visual","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","date":"2023-06-05","arxiv_id":"2306.02858","repositories_listed":4,"syntology":{"n":25,"n_ran":18,"n_unverified":7,"n_pointer_only":9}},{"url":"/paper/fine-grained-video-text-retrieval-with","title":"Fine-grained Video-Text Retrieval with Hierarchical Graph Reasoning","date":"2020-03-01","arxiv_id":"2003.00392","repositories_listed":4,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/x-clip-end-to-end-multi-grained-contrastive","title":"X-CLIP: End-to-End Multi-grained Contrastive Learning for Video-Text Retrieval","date":"2022-07-15","arxiv_id":"2207.07285","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/mplug-effective-and-efficient-vision-language","title":"mPLUG: Effective and Efficient Vision-Language Learning by Cross-modal Skip-connections","date":"2022-05-24","arxiv_id":"2205.12005","repositories_listed":3,"syntology":null},{"url":"/paper/internvl-scaling-up-vision-foundation-models","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","date":"2023-12-21","arxiv_id":"2312.14238","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/rgnet-a-unified-retrieval-and-grounding","title":"RGNet: A Unified Clip Retrieval and Grounding Network for Long Videos","date":"2023-12-11","arxiv_id":"2312.06729","repositories_listed":2,"syntology":null},{"url":"/paper/timechat-a-time-sensitive-multimodal-large","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","date":"2023-12-04","arxiv_id":"2312.02051","repositories_listed":2,"syntology":{"n":11,"n_ran":7,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/uniadapter-unified-parameter-efficient","title":"UniAdapter: Unified Parameter-Efficient Transfer Learning for Cross-modal Modeling","date":"2023-02-13","arxiv_id":"2302.06605","repositories_listed":2,"syntology":{"n":10,"n_ran":8,"n_unverified":2,"n_pointer_only":10}},{"url":"/paper/egocentric-video-language-pretraining","title":"Egocentric Video-Language Pretraining","date":"2022-06-03","arxiv_id":"2206.01670","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/bridgeformer-bridging-video-text-retrieval","title":"Bridging Video-text Retrieval with Multiple Choice Questions","date":"2022-01-13","arxiv_id":"2201.04850","repositories_listed":2,"syntology":{"n":24,"n_ran":13,"n_unverified":11,"n_pointer_only":6}},{"url":"/paper/improving-video-text-retrieval-by-multi","title":"Improving Video-Text Retrieval by Multi-Stream Corpus Alignment and Dual Softmax Loss","date":"2021-09-09","arxiv_id":"2109.04290","repositories_listed":2,"syntology":null},{"url":"/paper/discovla-discrepancy-reduction-in-vision-1","title":"DiscoVLA: Discrepancy Reduction in Vision, Language, and Alignment for Parameter-Efficient Video-Text Retrieval","date":"2025-06-10","arxiv_id":"2506.08887","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/one-trajectory-one-token-grounded-video","title":"One Trajectory, One Token: Grounded Video Tokenization via Panoptic Sub-object Trajectory","date":"2025-05-29","arxiv_id":"2505.23617","repositories_listed":1,"syntology":null},{"url":"/paper/lovr-a-benchmark-for-long-video-retrieval-in","title":"LoVR: A Benchmark for Long Video Retrieval in Multimodal Contexts","date":"2025-05-20","arxiv_id":"2505.13928","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-working-memory-query-guided-segment","title":"Temporal Working Memory: Query-Guided Segment Refinement for Enhanced Multimodal Understanding","date":"2025-02-09","arxiv_id":"2502.06020","repositories_listed":1,"syntology":null},{"url":"/paper/expertized-caption-auto-enhancement-for-video","title":"Expertized Caption Auto-Enhancement for Video-Text Retrieval","date":"2025-02-05","arxiv_id":"2502.02885","repositories_listed":1,"syntology":null},{"url":"/paper/reversed-in-time-a-novel-temporal-emphasized","title":"Reversed in Time: A Novel Temporal-Emphasized Benchmark for Cross-Modal Video-Text Retrieval","date":"2024-12-26","arxiv_id":"2412.19178","repositories_listed":1,"syntology":null},{"url":"/paper/carel-instruction-guided-reinforcement","title":"CAREL: Instruction-guided reinforcement learning with cross-modal auxiliary objectives","date":"2024-11-29","arxiv_id":"2411.19787","repositories_listed":1,"syntology":null},{"url":"/paper/decomposing-relationship-from-1-to-n-into-n-1","title":"Text Proxy: Decomposing Retrieval from a 1-to-N Relationship into N 1-to-1 Relationships for Text-Video Retrieval","date":"2024-10-09","arxiv_id":"2410.06618","repositories_listed":1,"syntology":null},{"url":"/paper/2407-21757","title":"Learning Video Context as Interleaved Multimodal Sequences","date":"2024-07-31","arxiv_id":"2407.21757","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/video-language-alignment-pre-training-via","title":"Video-Language Alignment via Spatio-Temporal Graph Transformer","date":"2024-07-16","arxiv_id":"2407.11677","repositories_listed":1,"syntology":null},{"url":"/paper/diving-deep-into-the-motion-representation-of","title":"Diving Deep into the Motion Representation of Video-Text Models","date":"2024-06-07","arxiv_id":"2406.05075","repositories_listed":1,"syntology":null},{"url":"/paper/vid-tldr-training-free-token-merging-for","title":"vid-TLDR: Training Free Token merging for Light-weight Video Transformer","date":"2024-03-20","arxiv_id":"2403.13347","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/m2-raap-a-multi-modal-recipe-for-advancing","title":"M2-RAAP: A Multi-Modal Recipe for Advancing Adaptation-based Pre-training towards Effective and Efficient Zero-shot Video-text Retrieval","date":"2024-01-31","arxiv_id":"2401.17797","repositories_listed":1,"syntology":null},{"url":"/paper/pros-prompting-to-simulate-generalized","title":"ProS: Prompting-to-simulate Generalized knowledge for Universal Cross-Domain Retrieval","date":"2023-12-19","arxiv_id":"2312.12478","repositories_listed":1,"syntology":null},{"url":"/paper/harvest-video-foundation-models-via-efficient","title":"Harvest Video Foundation Models via Efficient Post-Pretraining","date":"2023-10-30","arxiv_id":"2310.19554","repositories_listed":1,"syntology":null},{"url":"/paper/testa-temporal-spatial-token-aggregation-for","title":"TESTA: Temporal-Spatial Token Aggregation for Long-form Video-Language Understanding","date":"2023-10-29","arxiv_id":"2310.19060","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_unverified":4,"n_pointer_only":0}}],"syntology_records":15,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}