{"url":"/dataset/avsd","name":"AVSD","full_name":"Audio-Visual Scene-Aware Dialog","description_markdown":"The Audio Visual Scene-Aware Dialog (**AVSD**) dataset, or DSTC7 Track 3, is a audio-visual dataset for dialogue understanding. The goal with the dataset and track was to design systems to generate responses in a dialog about a video, given the dialog history and audio-visual content of the video.\n\nSource: [The Eighth Dialog System Technology Challenge](https://arxiv.org/abs/1911.06394)\nImage Source: [http://workshop.colips.org/dstc7/papers/DSTC7_Task_3_overview_paper.pdf](http://workshop.colips.org/dstc7/papers/DSTC7_Task_3_overview_paper.pdf)","description_withheld":null,"homepage":"http://workshop.colips.org/dstc7/call.html","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/audio-visual-scene-aware-dialog-avsd","title":"Audio Visual Scene-Aware Dialog (AVSD) Challenge at DSTC7","first_author":"Huda Alamri","url":null},"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Scene-Aware Dialogue","url":"/task/scene-aware-dialogue","datasets_with_task":"/datasets/task/scene-aware-dialogue"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["AVSD"],"data_loaders":[],"num_papers_in_archive":14,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/scene-aware-dialogue-on-avsd","task":"Scene-Aware Dialogue","dataset_variant":"AVSD","rows":1,"metrics":["CIDEr"],"first_row_in_archive_order":{"model":"simple","paper":"/paper/a-simple-baseline-for-audio-visual-scene-1","metrics":{"CIDEr":"0.941"},"code_links":[{"title":"idansc/simple-avsd","url":"https://github.com/idansc/simple-avsd"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/a-simple-baseline-for-audio-visual-scene-1","title":"A Simple Baseline for Audio-Visual Scene-Aware Dialog","date":"2019-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}