{"url":"/task/multimodal-intent-recognition","name":"Multimodal Intent Recognition","slug":"multimodal-intent-recognition","description_markdown":"Intent recognition on multimodal content.\r\n\r\nImage source: [MIntRec: A New Dataset for Multimodal Intent Recognition](https://paperswithcode.com/dataset/mintrec)","categories":[{"name":"Miscellaneous","url":"/area/miscellaneous"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":16,"papers_with_code":10,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/multimodal-intent-recognition-on-mintrec","slug":"multimodal-intent-recognition-on-mintrec","dataset":"MIntRec","dataset_url":"/dataset/mintrec","rows_in_archive":6,"metrics":["Accuracy (20 classes)","Accuracy (Binary)"],"first_row_in_archive_order":{"model":"Human","paper_title":"MIntRec: A New Dataset for Multimodal Intent Recognition","paper_url":"/paper/mintrec-a-new-dataset-for-multimodal-intent","paper_date":"2022-09-09","arxiv_id":"2209.04355","code_links":[{"title":"thuiar/mintrec","url":"https://github.com/thuiar/mintrec"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}},{"leaderboard":"/sota/multimodal-intent-recognition-on-photochat","slug":"multimodal-intent-recognition-on-photochat","dataset":"PhotoChat","dataset_url":"/dataset/photochat","rows_in_archive":6,"metrics":["F1","Precision","Recall"],"first_row_in_archive_order":{"model":"PaCE","paper_title":"PaCE: Unified Multi-modal Dialogue Pre-training with Progressive and Compositional Experts","paper_url":"/paper/pace-unified-multi-modal-dialogue-pre","paper_date":"2023-05-24","arxiv_id":"2305.14839","code_links":[{"title":"AlibabaResearch/DAMO-ConvAI","url":"https://github.com/AlibabaResearch/DAMO-ConvAI/tree/main/pace"}],"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/multimodal-intent-recognition-on-mmdialog","slug":"multimodal-intent-recognition-on-mmdialog","dataset":"MMDialog","dataset_url":"/dataset/mmdialog","rows_in_archive":4,"metrics":["F1"],"first_row_in_archive_order":{"model":"PaCE","paper_title":"PaCE: Unified Multi-modal Dialogue Pre-training with Progressive and Compositional Experts","paper_url":"/paper/pace-unified-multi-modal-dialogue-pre","paper_date":"2023-05-24","arxiv_id":"2305.14839","code_links":[{"title":"AlibabaResearch/DAMO-ConvAI","url":"https://github.com/AlibabaResearch/DAMO-ConvAI/tree/main/pace"}],"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/photochat","name":"PhotoChat","full_name":"","num_papers_in_archive":20},{"url":"/dataset/mmdialog","name":"MMDialog","full_name":"","num_papers_in_archive":17},{"url":"/dataset/mintrec","name":"MIntRec","full_name":"","num_papers_in_archive":14}],"subtasks":[],"parent_tasks":[{"url":"/task/intent-recognition","name":"Intent Recognition"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":10,"of":10,"tagged_in_all":16,"items":[{"url":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","arxiv_id":"1810.04805","repositories_listed":534,"syntology":{"n":659,"n_ran":204,"n_unverified":455,"n_pointer_only":149}},{"url":"/paper/exploring-the-limits-of-transfer-learning","title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","date":"2019-10-23","arxiv_id":"1910.10683","repositories_listed":57,"syntology":{"n":31,"n_ran":2,"n_unverified":29,"n_pointer_only":0}},{"url":"/paper/albert-a-lite-bert-for-self-supervised","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","date":"2019-09-26","arxiv_id":"1909.11942","repositories_listed":48,"syntology":{"n":126,"n_ran":46,"n_unverified":80,"n_pointer_only":22}},{"url":"/paper/vilt-vision-and-language-transformer-without","title":"ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision","date":"2021-02-05","arxiv_id":"2102.03334","repositories_listed":6,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/mintrec2-0-a-large-scale-benchmark-dataset","title":"MIntRec2.0: A Large-scale Benchmark Dataset for Multimodal Intent Recognition and Out-of-scope Detection in Conversations","date":"2024-03-16","arxiv_id":"2403.10943","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":6}},{"url":"/paper/token-level-contrastive-learning-with","title":"Token-Level Contrastive Learning with Modality-Aware Prompting for Multimodal Intent Recognition","date":"2023-12-22","arxiv_id":"2312.14667","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/pace-unified-multi-modal-dialogue-pre","title":"PaCE: Unified Multi-modal Dialogue Pre-training with Progressive and Compositional Experts","date":"2023-05-24","arxiv_id":"2305.14839","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/speech-text-dialog-pre-training-for-spoken","title":"Speech-Text Dialog Pre-training for Spoken Dialog Understanding with Explicit Cross-Modal Alignment","date":"2023-05-19","arxiv_id":"2305.11579","repositories_listed":1,"syntology":null},{"url":"/paper/mmdialog-a-large-scale-multi-turn-dialogue","title":"MMDialog: A Large-scale Multi-turn Dialogue Dataset Towards Multi-modal Open-domain Conversation","date":"2022-11-10","arxiv_id":"2211.05719","repositories_listed":1,"syntology":null},{"url":"/paper/mintrec-a-new-dataset-for-multimodal-intent","title":"MIntRec: A New Dataset for Multimodal Intent Recognition","date":"2022-09-09","arxiv_id":"2209.04355","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":1,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}