{"url":"/dataset/tgif","name":"TGIF","full_name":"Tumblr GIF","description_markdown":"The Tumblr GIF (TGIF) dataset contains 100K animated GIFs and 120K sentences describing visual content of the animated GIFs. The animated GIFs have been collected from Tumblr, from randomly selected posts published between May and June of 2015. The dataset provides the URLs of animated GIFs. The sentences are collected via crowdsourcing, with a carefully designed annotation interface that ensures high quality dataset. There is one sentence per animated GIF for the training and validation splits, and three sentences per GIF for the test split. The dataset can be used to evaluate animated GIF/video description techniques.\r\n\r\nSource: [GitHub](https://github.com/raingo/TGIF-Release)","description_withheld":null,"homepage":"https://github.com/raingo/TGIF-Release","introduced_date":"2016-04-10","introduced_date_note":null,"introduced_by":{"paper":"/paper/tgif-a-new-dataset-and-benchmark-on-animated","title":"TGIF: A New Dataset and Benchmark on Animated GIF Description","first_author":"Yuncheng Li","url":null},"license":{"name":"Custom (research-only, non-commercial, attribution)","url":"https://github.com/raingo/TGIF-Release#license"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Video Question Answering","url":"/task/video-question-answering","datasets_with_task":"/datasets/task/video-question-answering"},{"name":"Video Retrieval","url":"/task/video-retrieval","datasets_with_task":"/datasets/task/video-retrieval"},{"name":"Video Captioning","url":"/task/video-captioning","datasets_with_task":"/datasets/task/video-captioning"}],"languages":[],"variants":["TGIF"],"data_loaders":[{"repo":"https://github.com/raingo/TGIF-Release","url":"https://github.com/raingo/TGIF-Release","frameworks":[]}],"num_papers_in_archive":51,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-retrieval-on-tgif","task":"Video Retrieval","dataset_variant":"TGIF","rows":2,"metrics":["text-to-video R@1","text-to-video R@5","text-to-video R@10","text-to-video Mean Rank","text-to-video Median Rank"],"first_row_in_archive_order":{"model":"MDMMT-2","paper":"/paper/mdmmt-2-multidomain-multimodal-transformer","metrics":{"text-to-video Mean Rank":"94.1","text-to-video Median Rank":"7.0","text-to-video R@1":"25.5","text-to-video R@10":"55.7","text-to-video R@5":"46.1"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/mdmmt-2-multidomain-multimodal-transformer","title":"MDMMT-2: Multidomain Multimodal Transformer for Video Retrieval, One More Step Towards Generalization","date":"2022-03-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/lightweight-attentional-feature-fusion-for","title":"Lightweight Attentional Feature Fusion: A New Baseline for Text-to-Video Retrieval","date":"2021-12-03","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}