{"url":"/dataset/image-chat","name":"Image-Chat","full_name":null,"description_markdown":"The IMAGE-CHAT dataset is a large collection of (image, style trait for speaker A, style trait for speaker B, dialogue between A & B) tuples that we collected using crowd-workers, Each dialogue consists of consecutive turns by speaker A and B. No particular constraints are placed on the kinds of utterance, only that we ask the speakers to both use the provided style trait, and to respond to the given image and dialogue history in an engaging way. The goal is not just to build a diagnostic dataset but a basis for training models that humans actually want to engage with.","description_withheld":null,"homepage":"http://parl.ai/projects/image_chat","introduced_date":"2020-07-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/image-chat-engaging-grounded-conversations","title":"Image-Chat: Engaging Grounded Conversations","first_author":"Kurt Shuster","url":null},"license":{"name":"http://parl.ai/projects/image_chat","url":"http://parl.ai/projects/image_chat"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Retrieval","url":"/task/text-retrieval","datasets_with_task":"/datasets/task/text-retrieval"},{"name":"Visual Dialog","url":"/task/visual-dialogue","datasets_with_task":"/datasets/task/visual-dialogue"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Image-Chat"],"data_loaders":[],"num_papers_in_archive":31,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-retrieval-on-image-chat","task":"Text Retrieval","dataset_variant":"Image-Chat","rows":3,"metrics":["R@1","R@5","Sum(R@1,5)"],"first_row_in_archive_order":{"model":"PaCE","paper":"/paper/pace-unified-multi-modal-dialogue-pre","metrics":{"R@1":"51.9","R@5":"76.8","Sum(R@1,5)":"128.7"},"code_links":[{"title":"AlibabaResearch/DAMO-ConvAI","url":"https://github.com/AlibabaResearch/DAMO-ConvAI/tree/main/pace"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-dialog-on-image-chat","task":"Visual Dialog","dataset_variant":"Image-Chat","rows":1,"metrics":["BLEU-4","F1","ROUGE-L"],"first_row_in_archive_order":{"model":"Multi-Modal BlenderBot","paper":"/paper/multi-modal-open-domain-dialogue","metrics":{"BLEU-4":"40","F1":"13.1","ROUGE-L":"18"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/pace-unified-multi-modal-dialogue-pre","title":"PaCE: Unified Multi-modal Dialogue Pre-training with Progressive and Compositional Experts","date":"2023-05-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vlmo-unified-vision-language-pre-training","title":"VLMo: Unified Vision-Language Pre-Training with Mixture-of-Modality-Experts","date":"2021-11-03","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/multi-modal-open-domain-dialogue","title":"Multi-Modal Open-Domain Dialogue","date":"2020-10-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/engaging-image-chat-modeling-personality-in","title":"Image Chat: Engaging Grounded Conversations","date":"2018-11-02","rows_on_this_dataset":1,"code_links":3,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":1,"samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}