{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/mdmmt-multidomain-multimodal-transformer-for","title":"MDMMT: Multidomain Multimodal Transformer for Video Retrieval","arxiv_id":"2103.10699","date":"2021-03-19","proceeding":null,"authors":["Maksim Dzabraev","Maksim Kalashnikov","Stepan Komkov","Aleksandr Petiushko"],"abstract":"We present a new state-of-the-art on the text to video retrieval task on MSRVTT and LSMDC benchmarks where our model outperforms all previous solutions by a large margin. Moreover, state-of-the-art results are achieved with a single model on two datasets without finetuning. This multidomain generalisation is achieved by a proper combination of different video caption datasets. We show that training on different datasets can improve test results of each other. Additionally we check intersection between many popular datasets and found that MSRVTT has a significant overlap between the test and the train parts, and the same situation is observed for ActivityNet.","url_abs":"https://arxiv.org/abs/2103.10699v1","url_pdf":"https://arxiv.org/pdf/2103.10699v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"mdmmt-multidomain-multimodal-transformer-for","repo_url":"https://github.com/papermsucode/mdmmt","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"mdmmt-multidomain-multimodal-transformer-for","repo_url":"https://github.com/willard-yuan/video-text-retrieval-papers","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"mdmmt-multidomain-multimodal-transformer-for","repo_url":"https://github.com/towhee-io/towhee","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"retrieval","task_name":"Retrieval"},{"task_slug":"text-to-video-retrieval","task_name":"Text to Video Retrieval"},{"task_slug":"video-retrieval","task_name":"Video Retrieval"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/video-retrieval-on-lsmdc","task":"Video Retrieval","dataset":"LSMDC","model":"MDMMT","rank_in_archive_order":25,"of":38,"metrics":{"text-to-video Mean Rank":"58.0","text-to-video Median Rank":"12.3","text-to-video R@1":"18.8","text-to-video R@10":"47.9","text-to-video R@5":"38.5"},"uses_additional_data":true},{"leaderboard":"/sota/video-retrieval-on-msr-vtt","task":"Video Retrieval","dataset":"MSR-VTT","model":"MDMMT","rank_in_archive_order":31,"of":40,"metrics":{"text-to-video Mean Rank":"52.8","text-to-video Median Rank":"6","text-to-video R@1":"23.1","text-to-video R@10":"61.8","text-to-video R@5":"49.8"},"uses_additional_data":true},{"leaderboard":"/sota/video-retrieval-on-msr-vtt-1ka","task":"Video Retrieval","dataset":"MSR-VTT-1kA","model":"MDMMT","rank_in_archive_order":40,"of":63,"metrics":{"text-to-video Mean Rank":"16.5","text-to-video Median Rank":"2","text-to-video R@1":"38.9","text-to-video R@10":"79.7","text-to-video R@5":"69.0"},"uses_additional_data":true}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2103.10699","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.10699"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/papermsucode/mdmmt","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/willard-yuan/video-text-retrieval-papers","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/towhee-io/towhee","reach":null}],"summary":{"ran_draft_wrong":2,"ran_fixture":2,"unverified":1},"by_repo_kind":{"official":{"samples":5,"ran":4,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":5,"samples":[{"code_sha256_prefix":"0f79e4325229f4bd","entry":"arg_modality","repo":"papermsucode/mdmmt","repo_kind":"official","path":"create_capts.py","file_url":"https://github.com/papermsucode/mdmmt/blob/HEAD/create_capts.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"0f79e4325229f4bd"}},{"code_sha256_prefix":"a64329e322b2eb4c","entry":"find_segment_t0","repo":"papermsucode/mdmmt","repo_kind":"official","path":"create_capts.py","file_url":"https://github.com/papermsucode/mdmmt/blob/HEAD/create_capts.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"a64329e322b2eb4c"}},{"code_sha256_prefix":"0ad66cf45cf5d9a2","entry":"find_segment_t1","repo":"papermsucode/mdmmt","repo_kind":"official","path":"create_capts.py","file_url":"https://github.com/papermsucode/mdmmt/blob/HEAD/create_capts.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"0ad66cf45cf5d9a2"}},{"code_sha256_prefix":"15fe3fd41e21930e","entry":"read_frames_center_crop","repo":"papermsucode/mdmmt","repo_kind":"official","path":"dumper.py","file_url":"https://github.com/papermsucode/mdmmt/blob/HEAD/dumper.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"15fe3fd41e21930e"}},{"code_sha256_prefix":"186530570952d72e","entry":"ffmpeg_audio_reader","repo":"papermsucode/mdmmt","repo_kind":"official","path":"dumper.py","file_url":"https://github.com/papermsucode/mdmmt/blob/HEAD/dumper.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"186530570952d72e"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}