{"url":"/task/zero-shot-audio-captioning","name":"Zero-shot Audio Captioning","slug":"zero-shot-audio-captioning","description_markdown":"Zero-shot audio captioning aims at automatically generating descriptive textual captions for audio content without any prior training for this task. Audio captioning is commonly concerned with ambient sounds, or sounds produced by a human performing an action.","categories":[{"name":"Audio","url":"/area/audio"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":6,"papers_with_code":4,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":2,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/zero-shot-audio-captioning-on-audiocaps","slug":"zero-shot-audio-captioning-on-audiocaps","dataset":"AudioCaps","dataset_url":"/dataset/audiocaps","rows_in_archive":4,"metrics":["METEOR","BLEU-4","CIDEr","ROUGE-L","SPICE","SPIDEr"],"first_row_in_archive_order":{"model":"Audio Flamingo","paper_title":"Audio Flamingo: A Novel Audio Language Model with Few-Shot Learning and Dialogue Abilities","paper_url":"/paper/audio-flamingo-a-novel-audio-language-model","paper_date":"2024-02-02","arxiv_id":"2402.01831","code_links":[{"title":"NVIDIA/audio-flamingo","url":"https://github.com/NVIDIA/audio-flamingo"}],"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}}},{"leaderboard":"/sota/zero-shot-audio-captioning-on-clotho","slug":"zero-shot-audio-captioning-on-clotho","dataset":"Clotho","dataset_url":"/dataset/clotho","rows_in_archive":1,"metrics":["METEOR","BLEU-4","CIDEr","ROUGE-L","SPICE","SPIDEr"],"first_row_in_archive_order":{"model":"ZerAuCap","paper_title":"Zero-shot audio captioning with audio-language model guidance and audio context keywords","paper_url":"/paper/zero-shot-audio-captioning-with-audio","paper_date":"2023-11-14","arxiv_id":"2311.08396","code_links":[{"title":"explainableml/zeraucap","url":"https://github.com/explainableml/zeraucap"}],"syntology":{"n":17,"n_ran":13,"n_unverified":4,"n_pointer_only":17}}}],"datasets":[{"url":"/dataset/audiocaps","name":"AudioCaps","full_name":"","num_papers_in_archive":279},{"url":"/dataset/clotho","name":"Clotho","full_name":"Clotho","num_papers_in_archive":202}],"subtasks":[],"parent_tasks":[{"url":"/task/audio-captioning","name":"Audio captioning"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":4,"of":4,"tagged_in_all":6,"items":[{"url":"/paper/drcap-decoding-clap-latents-with-retrieval","title":"DRCap: Decoding CLAP Latents with Retrieval-Augmented Generation for Zero-shot Audio Captioning","date":"2024-10-12","arxiv_id":"2410.09472","repositories_listed":1,"syntology":null},{"url":"/paper/an-eye-for-an-ear-zero-shot-audio-description","title":"An Eye for an Ear: Zero-shot Audio Description Leveraging an Image Captioner using Audiovisual Distribution Alignment","date":"2024-10-08","arxiv_id":"2410.05997","repositories_listed":1,"syntology":null},{"url":"/paper/audio-flamingo-a-novel-audio-language-model","title":"Audio Flamingo: A Novel Audio Language Model with Few-Shot Learning and Dialogue Abilities","date":"2024-02-02","arxiv_id":"2402.01831","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/zero-shot-audio-captioning-with-audio","title":"Zero-shot audio captioning with audio-language model guidance and audio context keywords","date":"2023-11-14","arxiv_id":"2311.08396","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_unverified":4,"n_pointer_only":17}}],"syntology_records":2,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-25T09:33:49+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}