{"url":"/task/text-to-video-retrieval","name":"Text to Video Retrieval","slug":"text-to-video-retrieval","description_markdown":"She's gone \r\nI can't find her anywhere \r\nI'm looking everywhere for her\r\nEverywhere is dark","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":75,"papers_with_code":51,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":7,"subtasks":1,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/text-to-video-retrieval-on-kinetic-geb","slug":"text-to-video-retrieval-on-kinetic-geb","dataset":"Kinetics-GEB+","dataset_url":"/dataset/kinetic-gebc","rows_in_archive":2,"metrics":["mAP","text-to-video R@1","text-to-video R@10","text-to-video R@5","text-to-video R@50"],"first_row_in_archive_order":{"model":"FROZEN-revised","paper_title":"GEB+: A Benchmark for Generic Event Boundary Captioning, Grounding and Retrieval","paper_url":"/paper/generic-event-boundary-captioning-a-benchmark","paper_date":"2022-04-01","arxiv_id":"2204.00486","code_links":[{"title":"yuxuan-w/geb-plus","url":"https://github.com/yuxuan-w/geb-plus"}],"syntology":null}},{"leaderboard":"/sota/text-to-video-retrieval-on-msr-vtt","slug":"text-to-video-retrieval-on-msr-vtt","dataset":"MSR-VTT","dataset_url":"/dataset/msr-vtt","rows_in_archive":1,"metrics":["text-to-video R@1"],"first_row_in_archive_order":{"model":"CLIP4Clip","paper_title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval","paper_url":"/paper/clip4clip-an-empirical-study-of-clip-for-end","paper_date":"2021-04-18","arxiv_id":"2104.08860","code_links":[{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"ArrowLuo/CLIP4Clip","url":"https://github.com/ArrowLuo/CLIP4Clip"},{"title":"roudimit/AVLnet","url":"https://github.com/roudimit/AVLnet"},{"title":"facebookresearch/EgoTV","url":"https://github.com/facebookresearch/EgoTV"},{"title":"willard-yuan/video-text-retrieval-papers","url":"https://github.com/willard-yuan/video-text-retrieval-papers"}],"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":3}}},{"leaderboard":"/sota/text-to-video-retrieval-on-msvd-indonesian","slug":"text-to-video-retrieval-on-msvd-indonesian","dataset":"MSVD-Indonesian","dataset_url":"/dataset/msvd-indonesian","rows_in_archive":1,"metrics":["R@1","R@5","R@10","Median Rank","Mean Rank"],"first_row_in_archive_order":{"model":"X-CLIP (Cross-Lingual)","paper_title":"MSVD-Indonesian: A Benchmark for Multimodal Video-Text Tasks in Indonesian","paper_url":"/paper/msvd-indonesian-a-benchmark-for-multimodal","paper_date":"2023-06-20","arxiv_id":"2306.11341","code_links":[{"title":"willyfh/msvd-indonesian","url":"https://github.com/willyfh/msvd-indonesian"}],"syntology":null}}],"datasets":[{"url":"/dataset/kinetics","name":"Kinetics","full_name":"Kinetics Human Action Video Dataset","num_papers_in_archive":1341},{"url":"/dataset/msr-vtt","name":"MSR-VTT","full_name":"","num_papers_in_archive":640},{"url":"/dataset/marine-video-kit","name":"MVK","full_name":"Marine Video Kit","num_papers_in_archive":9},{"url":"/dataset/sakuga-42m","name":"Sakuga-42M","full_name":"","num_papers_in_archive":2},{"url":"/dataset/chinaopen","name":"ChinaOpen-1k","full_name":"","num_papers_in_archive":1},{"url":"/dataset/kinetic-gebc","name":"Kinetics-GEB+","full_name":"","num_papers_in_archive":1},{"url":"/dataset/msvd-indonesian","name":"MSVD-Indonesian","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/partially-relevant-video-retrieval","name":"Partially Relevant Video Retrieval"}],"parent_tasks":[{"url":"/task/10-shot-image-generation","name":"10-shot image generation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":51,"tagged_in_all":75,"items":[{"url":"/paper/vatt-transformers-for-multimodal-self","title":"VATT: Transformers for Multimodal Self-Supervised Learning from Raw Video, Audio and Text","date":"2021-04-22","arxiv_id":"2104.11178","repositories_listed":5,"syntology":{"n":8,"n_ran":5,"n_unverified":3,"n_pointer_only":8}},{"url":"/paper/clip4clip-an-empirical-study-of-clip-for-end","title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval","date":"2021-04-18","arxiv_id":"2104.08860","repositories_listed":5,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/frozen-in-time-a-joint-video-and-image","title":"Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval","date":"2021-04-01","arxiv_id":"2104.00650","repositories_listed":5,"syntology":{"n":11,"n_ran":3,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/end-to-end-learning-of-visual-representations","title":"End-to-End Learning of Visual Representations from Uncurated Instructional Videos","date":"2019-12-13","arxiv_id":"1912.06430","repositories_listed":4,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/howto100m-learning-a-text-video-embedding-by","title":"HowTo100M: Learning a Text-Video Embedding by Watching Hundred Million Narrated Video Clips","date":"2019-06-07","arxiv_id":"1906.03327","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/mdmmt-multidomain-multimodal-transformer-for","title":"MDMMT: Multidomain Multimodal Transformer for Video Retrieval","date":"2021-03-19","arxiv_id":"2103.10699","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":5}},{"url":"/paper/contextiq-a-multimodal-expert-based-video","title":"ContextIQ: A Multimodal Expert-Based Video Retrieval System for Contextual Advertising","date":"2024-10-29","arxiv_id":"2410.22233","repositories_listed":2,"syntology":null},{"url":"/paper/x-2-vlm-all-in-one-pre-trained-model-for","title":"X$^2$-VLM: All-In-One Pre-trained Model For Vision-Language Tasks","date":"2022-11-22","arxiv_id":"2211.12402","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":6}},{"url":"/paper/revealing-single-frame-bias-for-video-and","title":"Revealing Single Frame Bias for Video-and-Language Learning","date":"2022-06-07","arxiv_id":"2206.03428","repositories_listed":2,"syntology":{"n":12,"n_ran":4,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/fitclip-refining-large-scale-pretrained-image","title":"FitCLIP: Refining Large-Scale Pretrained Image-Text Models for Zero-Shot Video Understanding Tasks","date":"2022-03-24","arxiv_id":"2203.13371","repositories_listed":2,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/revitalize-region-feature-for-democratizing","title":"Revitalize Region Feature for Democratizing Video-Language Pre-training of Retrieval","date":"2022-03-15","arxiv_id":"2203.07720","repositories_listed":2,"syntology":null},{"url":"/paper/bridgeformer-bridging-video-text-retrieval","title":"Bridging Video-text Retrieval with Multiple Choice Questions","date":"2022-01-13","arxiv_id":"2201.04850","repositories_listed":2,"syntology":{"n":24,"n_ran":13,"n_unverified":11,"n_pointer_only":6}},{"url":"/paper/towards-efficient-partially-relevant-video","title":"Towards Efficient Partially Relevant Video Retrieval with Active Moment Discovering","date":"2025-04-15","arxiv_id":"2504.10920","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/tc-mgc-text-conditioned-multi-grained","title":"TC-MGC: Text-Conditioned Multi-Grained Contrastive Learning for Text-Video Retrieval","date":"2025-04-07","arxiv_id":"2504.04707","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/stablefusion-continual-video-retrieval-via","title":"StableFusion: Continual Video Retrieval via Frame Adaptation","date":"2025-03-13","arxiv_id":"2503.10111","repositories_listed":1,"syntology":null},{"url":"/paper/towards-efficient-and-effective-text-to-video","title":"Towards Efficient and Effective Text-to-Video Retrieval with Coarse-to-Fine Visual Representation Learning","date":"2024-01-01","arxiv_id":"2401.00701","repositories_listed":1,"syntology":null},{"url":"/paper/holistic-features-are-almost-sufficient-for","title":"Holistic Features are almost Sufficient for Text-to-Video Retrieval","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/videocon-robust-video-language-alignment-via","title":"VideoCon: Robust Video-Language Alignment via Contrast Captions","date":"2023-11-15","arxiv_id":"2311.10111","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/building-an-open-vocabulary-video-clip-model","title":"Building an Open-Vocabulary Video CLIP Model with Better Architectures, Optimization and Data","date":"2023-10-08","arxiv_id":"2310.05010","repositories_listed":1,"syntology":null},{"url":"/paper/prototype-based-aleatoric-uncertainty-1","title":"Prototype-based Aleatoric Uncertainty Quantification for Cross-modal Retrieval","date":"2023-09-29","arxiv_id":"2309.17093","repositories_listed":1,"syntology":{"n":19,"n_ran":12,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/unified-coarse-to-fine-alignment-for-video","title":"Unified Coarse-to-Fine Alignment for Video-Text Retrieval","date":"2023-09-18","arxiv_id":"2309.10091","repositories_listed":1,"syntology":{"n":14,"n_ran":8,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/msvd-indonesian-a-benchmark-for-multimodal","title":"MSVD-Indonesian: A Benchmark for Multimodal Video-Text Tasks in Indonesian","date":"2023-06-20","arxiv_id":"2306.11341","repositories_listed":1,"syntology":null},{"url":"/paper/meltr-meta-loss-transformer-for-learning-to","title":"MELTR: Meta Loss Transformer for Learning to Fine-tune Video Foundation Models","date":"2023-03-23","arxiv_id":"2303.13009","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/efficient-end-to-end-video-question-answering","title":"Efficient End-to-End Video Question Answering with Pyramidal Multimodal Transformer","date":"2023-02-04","arxiv_id":"2302.02136","repositories_listed":1,"syntology":null},{"url":"/paper/dual-learning-with-dynamic-knowledge","title":"Dual Learning with Dynamic Knowledge Distillation for Partially Relevant Video Retrieval","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/vindlu-a-recipe-for-effective-video-and","title":"VindLU: A Recipe for Effective Video-and-Language Pretraining","date":"2022-12-09","arxiv_id":"2212.05051","repositories_listed":1,"syntology":null},{"url":"/paper/are-all-combinations-equal-combining-textual","title":"Are All Combinations Equal? Combining Textual and Visual Features with Multiple Space Learning for Text-Based Video Retrieval","date":"2022-11-21","arxiv_id":"2211.11351","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-end-to-end-video","title":"An Empirical Study of End-to-End Video-Language Transformers with Masked Visual Modeling","date":"2022-09-04","arxiv_id":"2209.01540","repositories_listed":1,"syntology":null},{"url":"/paper/partially-relevant-video-retrieval","title":"Partially Relevant Video Retrieval","date":"2022-08-26","arxiv_id":"2208.12510","repositories_listed":1,"syntology":null},{"url":"/paper/clover-towards-a-unified-video-language","title":"Clover: Towards A Unified Video-Language Alignment and Fusion Model","date":"2022-07-16","arxiv_id":"2207.07885","repositories_listed":1,"syntology":null}],"syntology_records":16,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}