{"url":"/sota/image-manipulation-localization-on-cocoglide","task":{"name":"Image Manipulation Localization","url":"/task/image-manipulation-localization","note":null},"dataset":{"name":"CocoGlide","url":null},"category":"Computer Vision","categories":["Computer Vision"],"category_note":null,"description":"The task of segmenting parts of images or image parts that have been tampered with or manipulated (sometimes also referred to as doctored). This typically encompasses image splicing, copy-move, or image inpainting.","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["Average Pixel F1(Fixed threshold)"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"Average Pixel F1(Fixed threshold)":"higher"}},"counts":{"rows":11,"rows_with_code":9,"rows_with_paper_page":11,"rows_dated":10,"rows_using_additional_data":0},"rows":[{"rank_in_archive_order":1,"model":"CMX (RGB+SRM)","metrics":{"Average Pixel F1(Fixed threshold)":".585"},"uses_additional_data":false,"paper_date":"2022-03-09","paper":"/paper/cmx-cross-modal-fusion-for-rgb-x-semantic","paper_url":"https://arxiv.org/abs/2203.04838v5","paper_title":"CMX: Cross-Modal Fusion for RGB-X Semantic Segmentation with Transformers","code":"https://github.com/huaaaliu/rgbx_semantic_segmentation","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":0}},{"rank_in_archive_order":2,"model":"Late Fusion","metrics":{"Average Pixel F1(Fixed threshold)":".574"},"uses_additional_data":false,"paper_date":"2023-12-04","paper":"/paper/exploring-multi-modal-fusion-for-image","paper_url":"https://arxiv.org/abs/2312.01790v2","paper_title":"MMFusion: Combining Image Forensic Filters for Visual Manipulation Detection and Localization","code":"https://github.com/idt-iti/mmfusion-iml","n_code_links":1,"syntology":null},{"rank_in_archive_order":3,"model":"CMX (RGB+Bayar)","metrics":{"Average Pixel F1(Fixed threshold)":".566"},"uses_additional_data":false,"paper_date":"2022-03-09","paper":"/paper/cmx-cross-modal-fusion-for-rgb-x-semantic","paper_url":"https://arxiv.org/abs/2203.04838v5","paper_title":"CMX: Cross-Modal Fusion for RGB-X Semantic Segmentation with Transformers","code":"https://github.com/huaaaliu/rgbx_semantic_segmentation","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":0}},{"rank_in_archive_order":4,"model":"Early Fusion","metrics":{"Average Pixel F1(Fixed threshold)":".553"},"uses_additional_data":false,"paper_date":"2023-12-04","paper":"/paper/exploring-multi-modal-fusion-for-image","paper_url":"https://arxiv.org/abs/2312.01790v2","paper_title":"MMFusion: Combining Image Forensic Filters for Visual Manipulation Detection and Localization","code":"https://github.com/idt-iti/mmfusion-iml","n_code_links":1,"syntology":null},{"rank_in_archive_order":5,"model":"TruFor","metrics":{"Average Pixel F1(Fixed threshold)":".523"},"uses_additional_data":false,"paper_date":"2022-12-21","paper":"/paper/trufor-leveraging-all-round-clues-for","paper_url":"https://arxiv.org/abs/2212.10957v3","paper_title":"TruFor: Leveraging all-round clues for trustworthy image forgery detection and localization","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":6,"model":"ManTraNet","metrics":{"Average Pixel F1(Fixed threshold)":".516"},"uses_additional_data":false,"paper_date":"2019-06-01","paper":"/paper/mantra-net-manipulation-tracing-network-for","paper_url":"http://openaccess.thecvf.com/content_CVPR_2019/html/Wu_ManTra-Net_Manipulation_Tracing_Network_for_Detection_and_Localization_of_Image_CVPR_2019_paper.html","paper_title":"ManTra-Net: Manipulation Tracing Network for Detection and Localization of Image Forgeries With Anomalous Features","code":"https://github.com/ISICV/ManTraNet","n_code_links":3,"syntology":null},{"rank_in_archive_order":7,"model":"CMX (RGB+NP++)","metrics":{"Average Pixel F1(Fixed threshold)":".516"},"uses_additional_data":false,"paper_date":"2022-03-09","paper":"/paper/cmx-cross-modal-fusion-for-rgb-x-semantic","paper_url":"https://arxiv.org/abs/2203.04838v5","paper_title":"CMX: Cross-Modal Fusion for RGB-X Semantic Segmentation with Transformers","code":"https://github.com/huaaaliu/rgbx_semantic_segmentation","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":0}},{"rank_in_archive_order":8,"model":"MVSS-Net","metrics":{"Average Pixel F1(Fixed threshold)":".486"},"uses_additional_data":false,"paper_date":"2021-04-14","paper":"/paper/image-manipulation-detection-by-multi-view","paper_url":"https://arxiv.org/abs/2104.06832v2","paper_title":"Image Manipulation Detection by Multi-View Multi-Scale Supervision","code":"https://github.com/dong03/MVSS-Net","n_code_links":2,"syntology":{"n_ran":9,"n_unverified":4,"n_samples":13,"n_pointer_only_licence":13}},{"rank_in_archive_order":9,"model":"CR-CNN","metrics":{"Average Pixel F1(Fixed threshold)":".447"},"uses_additional_data":false,"paper_date":"2019-11-19","paper":"/paper/constrained-r-cnn-a-general-image","paper_url":"https://arxiv.org/abs/1911.08217v3","paper_title":"Constrained R-CNN: A general image manipulation detection model","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":10,"model":"CAT-Net v2","metrics":{"Average Pixel F1(Fixed threshold)":".434"},"uses_additional_data":false,"paper_date":"2021-08-30","paper":"/paper/learning-jpeg-compression-artifacts-for-image","paper_url":"https://arxiv.org/abs/2108.12947v2","paper_title":"Learning JPEG Compression Artifacts for Image Manipulation Detection and Localization","code":"https://github.com/mjkwon2021/cat-net","n_code_links":1,"syntology":null},{"rank_in_archive_order":11,"model":"SPAN","metrics":{"Average Pixel F1(Fixed threshold)":".298"},"uses_additional_data":false,"paper_date":null,"paper":"/paper/span-spatial-pyramid-attention-network-for","paper_url":"https://www.ecva.net/papers/eccv_2020/papers_ECCV/html/3720_ECCV_2020_paper.php","paper_title":"SPAN: Spatial Pyramid Attention Network for Image Manipulation Localization","code":"https://github.com/ZhiHanZ/IRIS0-SPAN","n_code_links":1,"syntology":null}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":4,"rows_with_any_sample_ran":4,"distinct_papers_with_graph_line":2,"distinct_papers_with_any_sample_ran":2,"samples_over_distinct_papers":{"n_ran":11,"n_unverified":4,"n_samples":15,"n_pointer_only_licence":13,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":15,"n_unverified":4,"n_samples":19,"n_pointer_only_licence":13,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}