{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/both-text-and-images-leaked-a-systematic","title":"Both Text and Images Leaked! A Systematic Analysis of Multimodal LLM Data Contamination","arxiv_id":"2411.03823","date":"2024-11-06","proceeding":null,"authors":["Dingjie Song","Sicheng Lai","Shunian Chen","Lichao Sun","Benyou Wang"],"abstract":"The rapid progression of multimodal large language models (MLLMs) has demonstrated superior performance on various multimodal benchmarks. However, the issue of data contamination during training creates challenges in performance evaluation and comparison. While numerous methods exist for detecting models' contamination in large language models (LLMs), they are less effective for MLLMs due to their various modalities and multiple training phases. In this study, we introduce a multimodal data contamination detection framework, MM-Detect, designed for MLLMs. Our experimental results indicate that MM-Detect is quite effective and sensitive in identifying varying degrees of contamination, and can highlight significant performance improvements due to the leakage of multimodal benchmark training sets. Furthermore, we explore whether the contamination originates from the base LLMs used by MLLMs or the multimodal training phase, providing new insights into the stages at which contamination may be introduced.","url_abs":"https://arxiv.org/abs/2411.03823v2","url_pdf":"https://arxiv.org/pdf/2411.03823v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"both-text-and-images-leaked-a-systematic","repo_url":"https://github.com/MLLM-Data-Contamination/MM-Detect","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}}],"tasks":[],"methods":[{"method_slug":"base","method_name":"BASE"},{"method_slug":"set","method_name":"SET"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/2411.03823","atlas_url":"https://app.syntology.ai/?focus=2411.03823","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03823"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/MLLM-Data-Contamination/MM-Detect","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"ran":1,"ran_fixture":1,"unverified":3},"by_repo_kind":{"official":{"samples":5,"ran":2,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"57e95166ade5f25c","entry":"build_transform","repo":"MLLM-Data-Contamination/MM-Detect","repo_kind":"official","path":"mm_detect/mllms/internvl2.py","file_url":"https://github.com/MLLM-Data-Contamination/MM-Detect/blob/HEAD/mm_detect/mllms/internvl2.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"57e95166ade5f25c"}},{"code_sha256_prefix":"bd77f5f8067f18e9","entry":"find_closest_aspect_ratio","repo":"MLLM-Data-Contamination/MM-Detect","repo_kind":"official","path":"mm_detect/mllms/internvl2.py","file_url":"https://github.com/MLLM-Data-Contamination/MM-Detect/blob/HEAD/mm_detect/mllms/internvl2.py","link_basis":"harvester_set","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"bd77f5f8067f18e9"}},{"code_sha256_prefix":"e7b40194e0800766","entry":"encode_image","repo":"MLLM-Data-Contamination/MM-Detect","repo_kind":"official","path":"mm_detect/mllms/gpt.py","file_url":"https://github.com/MLLM-Data-Contamination/MM-Detect/blob/HEAD/mm_detect/mllms/gpt.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"e7b40194e0800766"}},{"code_sha256_prefix":"4f4a4d8ac2936d96","entry":"image_to_base64","repo":"MLLM-Data-Contamination/MM-Detect","repo_kind":"official","path":"mm_detect/base_contamination_checker.py","file_url":"https://github.com/MLLM-Data-Contamination/MM-Detect/blob/HEAD/mm_detect/base_contamination_checker.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"4f4a4d8ac2936d96"}},{"code_sha256_prefix":"de385d1e1eb04efa","entry":"split_model","repo":"MLLM-Data-Contamination/MM-Detect","repo_kind":"official","path":"mm_detect/mllms/internvl2.py","file_url":"https://github.com/MLLM-Data-Contamination/MM-Detect/blob/HEAD/mm_detect/mllms/internvl2.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"de385d1e1eb04efa"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}