{"url":"/dataset/photochat","name":"PhotoChat","full_name":null,"description_markdown":"PhotoChat, the first dataset that casts light on the photo sharing behavior in online messaging. PhotoChat contains 12k dialogues,\r\neach of which is paired with a user photo that is shared during the conversation. Based on this dataset, we propose two tasks to fa\u0002cilitate research on image-text modeling: a\r\nphoto-sharing intent prediction task that pre\u0002dicts whether one intends to share a photo in the next conversation turn, and a photo retrieval task that retrieves the most relevant photo according to the dialogue context.","description_withheld":null,"homepage":"","introduced_date":"2021-07-06","introduced_date_note":null,"introduced_by":{"paper":"/paper/photochat-a-human-human-dialogue-dataset-with","title":"PhotoChat: A Human-Human Dialogue Dataset with Photo Sharing Behavior for Joint Image-Text Modeling","first_author":"Xiaoxue Zang","url":null},"license":{"name":"https://github.com/google-research/google-research/tree/master/multimodalchat","url":"https://github.com/google-research/google-research/tree/master/multimodalchat"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Image Retrieval","url":"/task/image-retrieval","datasets_with_task":"/datasets/task/image-retrieval"},{"name":"Multimodal Intent Recognition","url":"/task/multimodal-intent-recognition","datasets_with_task":"/datasets/task/multimodal-intent-recognition"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["PhotoChat"],"data_loaders":[],"num_papers_in_archive":20,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/multimodal-intent-recognition-on-photochat","task":"Multimodal Intent Recognition","dataset_variant":"PhotoChat","rows":6,"metrics":["F1","Precision","Recall"],"first_row_in_archive_order":{"model":"PaCE","paper":"/paper/pace-unified-multi-modal-dialogue-pre","metrics":{"F1":"63.8","Precision":"63.3","Recall":"68"},"code_links":[{"title":"AlibabaResearch/DAMO-ConvAI","url":"https://github.com/AlibabaResearch/DAMO-ConvAI/tree/main/pace"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-retrieval-on-photochat","task":"Image Retrieval","dataset_variant":"PhotoChat","rows":5,"metrics":["R1","R@10","R@5","Sum(R@1,5,10)"],"first_row_in_archive_order":{"model":"PaCE","paper":"/paper/pace-unified-multi-modal-dialogue-pre","metrics":{"R1":"15.2","R@10":"49.6","R@5":"36.7","Sum(R@1,5,10)":"101.5"},"code_links":[{"title":"AlibabaResearch/DAMO-ConvAI","url":"https://github.com/AlibabaResearch/DAMO-ConvAI/tree/main/pace"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/pace-unified-multi-modal-dialogue-pre","title":"PaCE: Unified Multi-modal Dialogue Pre-training with Progressive and Compositional Experts","date":"2023-05-24","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vlmo-unified-vision-language-pre-training","title":"VLMo: Unified Vision-Language Pre-Training with Mixture-of-Modality-Experts","date":"2021-11-03","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/photochat-a-human-human-dialogue-dataset-with","title":"PhotoChat: A Human-Human Dialogue Dataset with Photo Sharing Behavior for Joint Image-Text Modeling","date":"2021-07-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/vilt-vision-and-language-transformer-without","title":"ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision","date":"2021-02-05","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/exploring-the-limits-of-transfer-learning","title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","date":"2019-10-23","rows_on_this_dataset":2,"code_links":57,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":31,"samples_ran":21,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/albert-a-lite-bert-for-self-supervised","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","date":"2019-09-26","rows_on_this_dataset":1,"code_links":48,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":126,"samples_ran":68,"samples_unverified":58,"pointer_only_for_licence":22,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","rows_on_this_dataset":1,"code_links":534,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":659,"samples_ran":208,"samples_unverified":451,"pointer_only_for_licence":149,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/stacked-cross-attention-for-image-text","title":"Stacked Cross Attention for Image-Text Matching","date":"2018-03-21","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":16,"samples_ran":7,"samples_unverified":9,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":6,"samples_harvested":837,"samples_ran":305,"samples_unverified":532,"pointer_only_for_licence":173,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}