{"url":"/dataset/clotho","name":"Clotho","full_name":"Clotho","description_markdown":"**Clotho** is an audio captioning dataset, consisting of 4981 audio samples, and each audio sample has five captions (a total of 24 905 captions). Audio samples are of 15 to 30 s duration and captions are eight to 20 words long.\n\nSource: [https://zenodo.org/record/3490684](https://zenodo.org/record/3490684)\nImage Source: [https://arxiv.org/abs/1910.09387](https://arxiv.org/abs/1910.09387)","description_withheld":null,"homepage":"https://zenodo.org/record/3490684","introduced_date":"2019-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/clotho-an-audio-captioning-dataset","title":"Clotho: An Audio Captioning Dataset","first_author":"Konstantinos Drossos","url":null},"license":{"name":"Other (Attribution)","url":"https://zenodo.org/record/3490684"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"},{"name":"Multi-Task Learning","url":"/task/multi-task-learning","datasets_with_task":"/datasets/task/multi-task-learning"},{"name":"Audio captioning","url":"/task/audio-captioning","datasets_with_task":"/datasets/task/audio-captioning"},{"name":"Data Augmentation","url":"/task/data-augmentation","datasets_with_task":"/datasets/task/data-augmentation"},{"name":"Text to Audio Retrieval","url":"/task/text-to-audio-retrieval","datasets_with_task":"/datasets/task/text-to-audio-retrieval"},{"name":"Audio to Text Retrieval","url":"/task/audio-to-text-retrieval","datasets_with_task":"/datasets/task/audio-to-text-retrieval"},{"name":"Zero-shot Text to Audio Retrieval","url":"/task/zero-shot-text-to-audio-retrieval","datasets_with_task":"/datasets/task/zero-shot-text-to-audio-retrieval"},{"name":"Zero-shot Audio Captioning","url":"/task/zero-shot-audio-captioning","datasets_with_task":"/datasets/task/zero-shot-audio-captioning"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Clotho"],"data_loaders":[{"repo":"https://github.com/labbeti/aac-datasets","url":"https://aac-datasets.readthedocs.io/en/stable/","frameworks":["pytorch"]}],"num_papers_in_archive":202,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-to-audio-retrieval-on-clotho","task":"Text to Audio Retrieval","dataset_variant":"Clotho","rows":12,"metrics":["R@1","R@5","R@10","mAP@10"],"first_row_in_archive_order":{"model":"PaSST-RoBERTa & Estimated Audio–Caption Correspondences","paper":"/paper/estimated-audio-caption-correspondences","metrics":{"R@1":"27.69","R@10":"70.39","R@5":"57.03","mAP@10":"40.14"},"code_links":[{"title":"optimusprimus/salsa","url":"https://github.com/optimusprimus/salsa"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/audio-captioning-on-clotho","task":"Audio captioning","dataset_variant":"Clotho","rows":11,"metrics":["SPIDEr","CIDEr","SPICE","BLEU-4","METEOR","ROUGE-L","FENSE","SPIDEr-FL","Sentence-BERT"],"first_row_in_archive_order":{"model":"SLAM-AAC","paper":"/paper/slam-aac-enhancing-audio-captioning-with","metrics":{"CIDEr":"0.515","FENSE":"0.540","METEOR":"0.197","SPICE":"0.148","SPIDEr":"0.332","SPIDEr-FL":"0.330"},"code_links":[{"title":"X-LANCE/SLAM-LLM","url":"https://github.com/X-LANCE/SLAM-LLM"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-audio-captioning-on-clotho","task":"Zero-shot Audio Captioning","dataset_variant":"Clotho","rows":1,"metrics":["METEOR","BLEU-4","CIDEr","ROUGE-L","SPICE","SPIDEr"],"first_row_in_archive_order":{"model":"ZerAuCap","paper":"/paper/zero-shot-audio-captioning-with-audio","metrics":{"BLEU-4":"2.9","CIDEr":"14","METEOR":"9.4","ROUGE-L":"25.4","SPICE":"5.3","SPIDEr":"9.7"},"code_links":[{"title":"explainableml/zeraucap","url":"https://github.com/explainableml/zeraucap"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/audio-captioning-via-generative-pair-to-pair","title":"Enhancing Retrieval-Augmented Audio Captioning with Generation-Assisted Multimodal Querying and Progressive Learning","date":"2024-10-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/slam-aac-enhancing-audio-captioning-with","title":"SLAM-AAC: Enhancing Audio Captioning with Paraphrasing Augmentation and CLAP-Refine through LLMs","date":"2024-10-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/estimated-audio-caption-correspondences","title":"Estimated Audio-Caption Correspondences Improve Language-Based Audio Retrieval","date":"2024-08-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/enhancing-automated-audio-captioning-via","title":"Enhancing Automated Audio Captioning via Large Language Models with Optimized Audio Encoding","date":"2024-06-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/internvideo2-scaling-video-foundation-models","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","date":"2024-03-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/audio-flamingo-a-novel-audio-language-model","title":"Audio Flamingo: A Novel Audio Language Model with Few-Shot Learning and Dialogue Abilities","date":"2024-02-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zero-shot-audio-captioning-with-audio","title":"Zero-shot audio captioning with audio-language model guidance and audio context keywords","date":"2023-11-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":13,"samples_unverified":4,"pointer_only_for_licence":17,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/qwen-audio-advancing-universal-audio","title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","date":"2023-11-14","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":5,"samples_unverified":2,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/advancing-natural-language-based-audio","title":"Advancing Natural-Language Based Audio Retrieval with PaSST and Large Audio-Caption Data Sets","date":"2023-08-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":42,"samples_ran":15,"samples_unverified":27,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/one-peace-exploring-one-general","title":"ONE-PEACE: Exploring One General Representation Model Toward Unlimited Modalities","date":"2023-05-18","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/valor-vision-audio-language-omni-perception","title":"VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and Dataset","date":"2023-04-17","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/audio-retrieval-with-natural-language-queries-1","title":"Audio Retrieval with Natural Language Queries: A Benchmark Study","date":"2021-12-17","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/the-sjtu-system-for-dcase2021-challenge-task","title":"THE SJTU SYSTEM FOR DCASE2021 CHALLENGE TASK 6: AUDIO CAPTIONING BASED ON ENCODER PRE-TRAINING AND REINFORCEMENT LEARNING","date":"2021-07-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/the-dcase-2021-challenge-task-6-system","title":"THE DCASE 2021 CHALLENGE TASK 6 SYSTEM: AUTOMATED AUDIO CAPTIONING WITH WEAKLY SUPERVISED PRE-TRAING AND WORD SELECTION METHODS","date":"2021-07-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/audio-retrieval-with-natural-language-queries","title":"Audio Retrieval with Natural Language Queries","date":"2021-05-05","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/the-ntt-dcase2020-challenge-task-6-system","title":"The NTT DCASE2020 Challenge Task 6 system: Automated Audio Captioning with Keywords and Sentence Length Estimation","date":"2020-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/audio-captioning-using-gated-recurrent-units","title":"Audio Captioning using Gated Recurrent Units","date":"2020-06-05","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":88,"samples_ran":43,"samples_unverified":45,"pointer_only_for_licence":27,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}