{"url":"/dataset/lsmdc","name":"LSMDC","full_name":"Large Scale Movie Description Challenge","description_markdown":"This dataset contains 118,081 short video clips extracted from 202 movies. Each video has a caption, either extracted from the movie script or from transcribed DVS (descriptive video services) for the visually impaired. The validation set contains 7408 clips and evaluation is performed on a test set of 1000 videos from movies disjoint from the training and val sets.\r\n\r\nSource: [Use What You Have: Video Retrieval Using Representations From Collaborative Experts](https://arxiv.org/abs/1907.13487)\r\nImage Source: [https://sites.google.com/site/describingmovies/](https://sites.google.com/site/describingmovies/)","description_withheld":null,"homepage":"https://sites.google.com/site/describingmovies/","introduced_date":"2015-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/a-dataset-for-movie-description","title":"A Dataset for Movie Description","first_author":"Anna Rohrbach","url":null},"license":{"name":"Custom","url":"https://sites.google.com/site/describingmovies/download?authuser=0"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Zero-Shot Learning","url":"/task/zero-shot-learning","datasets_with_task":"/datasets/task/zero-shot-learning"},{"name":"Video Retrieval","url":"/task/video-retrieval","datasets_with_task":"/datasets/task/video-retrieval"},{"name":"Zero-Shot Video Retrieval","url":"/task/zero-shot-video-retrieval","datasets_with_task":"/datasets/task/zero-shot-video-retrieval"},{"name":"Fill Mask","url":"/task/fill-mask","datasets_with_task":"/datasets/task/fill-mask"}],"languages":[],"variants":["LSMDC"],"data_loaders":[],"num_papers_in_archive":126,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-retrieval-on-lsmdc","task":"Video Retrieval","dataset_variant":"LSMDC","rows":38,"metrics":["text-to-video R@1","text-to-video R@5","text-to-video R@10","text-to-video Median Rank","text-to-video Mean Rank","video-to-text R@1","video-to-text R@10","video-to-text R@5","video-to-text Median Rank","video-to-text Mean Rank"],"first_row_in_archive_order":{"model":"InternVideo2-6B","paper":"/paper/internvideo2-scaling-video-foundation-models","metrics":{"text-to-video R@1":"46.4","video-to-text R@1":"46.7"},"code_links":[{"title":"opengvlab/internvideo","url":"https://github.com/opengvlab/internvideo"},{"title":"opengvlab/internvideo2","url":"https://github.com/opengvlab/internvideo2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-video-retrieval-on-lsmdc","task":"Zero-Shot Video Retrieval","dataset_variant":"LSMDC","rows":16,"metrics":["text-to-video R@1","text-to-video R@5","text-to-video R@10","text-to-video Median Rank","text-to-video Mean Rank","video-to-text R@1","video-to-text R@5","video-to-text R@10"],"first_row_in_archive_order":{"model":"InternVideo2-6B","paper":"/paper/internvideo2-scaling-video-foundation-models","metrics":{"text-to-video R@1":"33.8","text-to-video R@10":"62.2","text-to-video R@5":"55.9","video-to-text R@1":"30.1","video-to-text R@10":"54.8","video-to-text R@5":"47.7"},"code_links":[{"title":"opengvlab/internvideo","url":"https://github.com/opengvlab/internvideo"},{"title":"opengvlab/internvideo2","url":"https://github.com/opengvlab/internvideo2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-learning-on-lsmdc","task":"Zero-Shot Learning","dataset_variant":"LSMDC","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"FrozenBiLM","paper":"/paper/zero-shot-video-question-answering-via-frozen","metrics":{"Accuracy":"51.5"},"code_links":[{"title":"antoyang/FrozenBiLM","url":"https://github.com/antoyang/FrozenBiLM"},{"title":"klauscc/dam","url":"https://github.com/klauscc/dam"},{"title":"sts-vlcc/sts-vlcc","url":"https://github.com/sts-vlcc/sts-vlcc"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/internvideo2-scaling-video-foundation-models","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","date":"2024-03-22","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/vid-tldr-training-free-token-merging-for","title":"vid-TLDR: Training Free Token merging for Light-weight Video Transformer","date":"2024-03-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/howtocaption-prompting-llms-to-transform","title":"HowToCaption: Prompting LLMs to Transform Video Annotations at Scale","date":"2023-10-07","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/one-for-all-video-conversation-is-feasible","title":"BT-Adapter: Video Conversation is Feasible Without Video Instruction Tuning","date":"2023-09-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cosa-concatenated-sample-pretrained-vision","title":"COSA: Concatenated Sample Pretrained Vision-Language Foundation Model","date":"2023-06-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/valor-vision-audio-language-omni-perception","title":"VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and Dataset","date":"2023-04-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unmasked-teacher-towards-training-efficient","title":"Unmasked Teacher: Towards Training-Efficient Video Foundation Models","date":"2023-03-28","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/diffusionret-generative-text-video-retrieval","title":"DiffusionRet: Generative Text-Video Retrieval with Diffusion Model","date":"2023-03-17","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mplug-2-a-modularized-multi-modal-foundation","title":"mPLUG-2: A Modularized Multi-modal Foundation Model Across Text, Image and Video","date":"2023-02-01","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":9,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/revisiting-temporal-modeling-for-clip-based","title":"Revisiting Temporal Modeling for CLIP-based Image-to-Video Knowledge Transferring","date":"2023-01-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hitea-hierarchical-temporal-aware-video","title":"HiTeA: Hierarchical Temporal-Aware Video-Language Pre-training","date":"2022-12-30","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/seeing-what-you-miss-vision-language-pre","title":"Seeing What You Miss: Vision-Language Pre-training with Semantic Completion Learning","date":"2022-11-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/expectation-maximization-contrastive-learning","title":"Expectation-Maximization Contrastive Learning for Compact Video-and-Language Representations","date":"2022-11-21","rows_on_this_dataset":3,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/clip-vip-adapting-pre-trained-image-text","title":"CLIP-ViP: Adapting Pre-trained Image-Text Model to Video-Language Representation Alignment","date":"2022-09-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/an-empirical-study-of-end-to-end-video","title":"An Empirical Study of End-to-End Video-Language Transformers with Masked Visual Modeling","date":"2022-09-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/clover-towards-a-unified-video-language","title":"Clover: Towards A Unified Video-Language Alignment and Fusion Model","date":"2022-07-16","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/x-clip-end-to-end-multi-grained-contrastive","title":"X-CLIP: End-to-End Multi-grained Contrastive Learning for Video-Text Retrieval","date":"2022-07-15","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zero-shot-video-question-answering-via-frozen","title":"Zero-Shot Video Question Answering via Frozen Bidirectional Language Models","date":"2022-06-16","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":34,"samples_ran":14,"samples_unverified":20,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/centerclip-token-clustering-for-efficient","title":"CenterCLIP: Token Clustering for Efficient Text-Video Retrieval","date":"2022-05-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/miles-visual-bert-pre-training-with-injected","title":"MILES: Visual BERT Pre-training with Injected Language Semantics for Video-text Retrieval","date":"2022-04-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hunyuan-tvr-for-text-video-retrivial","title":"Tencent Text-Video Retrieval: Hierarchical Cross-Modal Interactions with Multi-Level Representations","date":"2022-04-07","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/x-pool-cross-modal-language-video-attention","title":"X-Pool: Cross-Modal Language-Video Attention for Text-Video Retrieval","date":"2022-03-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mdmmt-2-multidomain-multimodal-transformer","title":"MDMMT-2: Multidomain Multimodal Transformer for Video Retrieval, One More Step Towards Generalization","date":"2022-03-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/bridgeformer-bridging-video-text-retrieval","title":"Bridging Video-text Retrieval with Multiple Choice Questions","date":"2022-01-13","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":13,"samples_unverified":11,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cross-modal-retrieval-with-querybank","title":"Cross Modal Retrieval with Querybank Normalisation","date":"2021-12-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/advancing-high-resolution-video-language","title":"Advancing High-Resolution Video-Language Representation with Large-Scale Video Transcriptions","date":"2021-11-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-and-text-matching-with-conditioned","title":"Video and Text Matching with Conditioned Embeddings","date":"2021-10-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-video-text-retrieval-by-multi","title":"Improving Video-Text Retrieval by Multi-Stream Corpus Alignment and Dual Softmax Loss","date":"2021-09-09","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/clip4clip-an-empirical-study-of-clip-for-end","title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval","date":"2021-04-18","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/frozen-in-time-a-joint-video-and-image","title":"Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval","date":"2021-04-01","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":3,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mdmmt-multidomain-multimodal-transformer-for","title":"MDMMT: Multidomain Multimodal Transformer for Video Retrieval","date":"2021-03-19","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-straightforward-framework-for-video","title":"A Straightforward Framework For Video Retrieval Using CLIP","date":"2021-02-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multi-modal-transformer-for-video-retrieval","title":"Multi-modal Transformer for Video Retrieval","date":"2020-07-21","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":1,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/noise-estimation-using-density-estimation-for","title":"Noise Estimation Using Density Estimation for Self-Supervised Multimodal Learning","date":"2020-03-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/use-what-you-have-video-retrieval-using","title":"Use What You Have: Video Retrieval Using Representations From Collaborative Experts","date":"2019-07-31","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/howto100m-learning-a-text-video-embedding-by","title":"HowTo100M: Learning a Text-Video Embedding by Watching Hundred Million Narrated Video Clips","date":"2019-06-07","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-joint-sequence-fusion-model-for-video","title":"A Joint Sequence Fusion Model for Video Question Answering and Retrieval","date":"2018-08-07","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-a-text-video-embedding-from","title":"Learning a Text-Video Embedding from Incomplete and Heterogeneous Data","date":"2018-04-07","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/learning-from-video-and-text-via-large-scale","title":"Learning from Video and Text via Large-Scale Discriminative Clustering","date":"2017-07-27","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/end-to-end-concept-word-detection-for-video","title":"End-to-end Concept Word Detection for Video Captioning, Retrieval, and Question Answering","date":"2016-10-10","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":22,"samples_harvested":158,"samples_ran":79,"samples_unverified":79,"pointer_only_for_licence":30,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}