{"url":"/sota/text-based-image-editing-on-pie-bench","task":{"name":"Text-based Image Editing","url":"/task/text-based-image-editing","note":null},"dataset":{"name":"PIE-Bench","url":"/dataset/pie-bench"},"category":"Computer Vision","categories":["Computer Vision","Natural Language Processing"],"category_note":null,"description":"Nose should be sharped and lips should be less fat","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["Background PSNR","Background LPIPS","Structure Distance","CLIPSIM"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"Background PSNR":"higher","Background LPIPS":null,"Structure Distance":"lower","CLIPSIM":null}},"counts":{"rows":18,"rows_with_code":17,"rows_with_paper_page":18,"rows_dated":18,"rows_using_additional_data":0},"rows":[{"rank_in_archive_order":1,"model":"KV-Edit","metrics":{"Background LPIPS":"9.92","Background PSNR":"35.87","CLIPSIM":"25.63","Structure Distance":"1.98"},"uses_additional_data":false,"paper_date":"2025-02-24","paper":"/paper/kv-edit-training-free-image-editing-for","paper_url":"https://arxiv.org/abs/2502.17363v1","paper_title":"KV-Edit: Training-Free Image Editing for Precise Background Preservation","code":"https://github.com/Xilluill/KV-Edit","n_code_links":1,"syntology":{"n_ran":15,"n_unverified":16,"n_samples":31,"n_pointer_only_licence":0}},{"rank_in_archive_order":2,"model":"Virtual Inversion+Unified Attention Control+LCM","metrics":{"Background LPIPS":"47.58","Background PSNR":"28.51","CLIPSIM":"25.03","Structure Distance":"13.78"},"uses_additional_data":false,"paper_date":"2023-12-07","paper":"/paper/inversion-free-image-editing-with-natural","paper_url":"https://arxiv.org/abs/2312.04965v1","paper_title":"Inversion-Free Image Editing with Natural Language","code":"https://github.com/sled-group/InfEdit","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":0,"n_samples":1,"n_pointer_only_licence":1}},{"rank_in_archive_order":3,"model":"Virtual Inversion+ViMAEdit","metrics":{"Background LPIPS":"45.67","Background PSNR":"28.27","CLIPSIM":"25.91","Structure Distance":"12.65"},"uses_additional_data":false,"paper_date":"2024-10-14","paper":"/paper/vision-guided-and-mask-enhanced-adaptive","paper_url":"https://arxiv.org/abs/2410.10496v2","paper_title":"Vision-guided and Mask-enhanced Adaptive Denoising for Prompt-based Image Editing","code":"https://github.com/Null-0000/ViMAEdit","n_code_links":1,"syntology":null},{"rank_in_archive_order":4,"model":"Virtual Inversion+Prompt-to-Prompt","metrics":{"Background LPIPS":"47.98","Background PSNR":"27.52","CLIPSIM":"24.89","Structure Distance":"14.22"},"uses_additional_data":false,"paper_date":"2023-12-07","paper":"/paper/inversion-free-image-editing-with-natural","paper_url":"https://arxiv.org/abs/2312.04965v1","paper_title":"Inversion-Free Image Editing with Natural Language","code":"https://github.com/sled-group/InfEdit","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":0,"n_samples":1,"n_pointer_only_licence":1}},{"rank_in_archive_order":5,"model":"Direct Inversion+Prompt-to-Prompt","metrics":{"Background LPIPS":"54.55","Background PSNR":"27.22","CLIPSIM":"25.02","Structure Distance":"11.65"},"uses_additional_data":false,"paper_date":"2023-10-02","paper":"/paper/direct-inversion-boosting-diffusion-based","paper_url":"https://arxiv.org/abs/2310.01506v2","paper_title":"Direct Inversion: Boosting Diffusion-based Editing with 3 Lines of Code","code":"https://github.com/cure-lab/pnpinversion","n_code_links":3,"syntology":{"n_ran":4,"n_unverified":2,"n_samples":6,"n_pointer_only_licence":6}},{"rank_in_archive_order":6,"model":"Null-Text Inversion+Prompt-to-Prompt","metrics":{"Background LPIPS":"60.67","Background PSNR":"27.03","CLIPSIM":"24.75","Structure Distance":"13.44"},"uses_additional_data":false,"paper_date":"2022-11-17","paper":"/paper/null-text-inversion-for-editing-real-images","paper_url":"https://arxiv.org/abs/2211.09794v1","paper_title":"Null-text Inversion for Editing Real Images using Guided Diffusion Models","code":"https://github.com/google/prompt-to-prompt","n_code_links":4,"syntology":{"n_ran":1,"n_unverified":0,"n_samples":1,"n_pointer_only_licence":0}},{"rank_in_archive_order":7,"model":"Virtual Inversion+Prompt-to-Prompt+LCM","metrics":{"Background LPIPS":"55.85","Background PSNR":"26.64","CLIPSIM":"24.57","Structure Distance":"15.61"},"uses_additional_data":false,"paper_date":"2023-12-07","paper":"/paper/inversion-free-image-editing-with-natural","paper_url":"https://arxiv.org/abs/2312.04965v1","paper_title":"Inversion-Free Image Editing with Natural Language","code":"https://github.com/sled-group/InfEdit","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":0,"n_samples":1,"n_pointer_only_licence":1}},{"rank_in_archive_order":8,"model":"Negative-Prompt Inversion+Prompt-to-Prompt","metrics":{"Background LPIPS":"69.01","Background PSNR":"26.21","CLIPSIM":"24.61","Structure Distance":"16.17"},"uses_additional_data":false,"paper_date":"2023-05-26","paper":"/paper/negative-prompt-inversion-fast-image","paper_url":"https://arxiv.org/abs/2305.16807v2","paper_title":"Negative-prompt Inversion: Fast Image Inversion for Editing with Text-guided Diffusion Models","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":9,"model":"StyleDiffusion+Prompt-to-Prompt","metrics":{"Background LPIPS":"66.10","Background PSNR":"26.05","CLIPSIM":"24.78","Structure Distance":"11.65"},"uses_additional_data":false,"paper_date":"2023-03-28","paper":"/paper/stylediffusion-prompt-embedding-inversion-for","paper_url":"https://arxiv.org/abs/2303.15649v3","paper_title":"StyleDiffusion: Prompt-Embedding Inversion for Text-Based Editing","code":"https://github.com/sen-mao/StyleDiffusion","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":0,"n_samples":1,"n_pointer_only_licence":1}},{"rank_in_archive_order":10,"model":"FireFlow","metrics":{"Background LPIPS":"123.6","Background PSNR":"23.03","CLIPSIM":"26.02","Structure Distance":"27.1"},"uses_additional_data":false,"paper_date":"2024-12-10","paper":"/paper/fireflow-fast-inversion-of-rectified-flow-for","paper_url":"https://arxiv.org/abs/2412.07517v1","paper_title":"FireFlow: Fast Inversion of Rectified Flow for Image Semantic Editing","code":"https://github.com/holmesshuan/fireflow","n_code_links":1,"syntology":{"n_ran":4,"n_unverified":7,"n_samples":11,"n_pointer_only_licence":0}},{"rank_in_archive_order":11,"model":"Direct Inversion+MasaCtrl","metrics":{"Background LPIPS":"87.94","Background PSNR":"22.64","CLIPSIM":"24.38","Structure Distance":"24.70"},"uses_additional_data":false,"paper_date":"2023-10-02","paper":"/paper/direct-inversion-boosting-diffusion-based","paper_url":"https://arxiv.org/abs/2310.01506v2","paper_title":"Direct Inversion: Boosting Diffusion-based Editing with 3 Lines of Code","code":"https://github.com/cure-lab/pnpinversion","n_code_links":3,"syntology":{"n_ran":4,"n_unverified":2,"n_samples":6,"n_pointer_only_licence":6}},{"rank_in_archive_order":12,"model":"Direct Inversion+Plug-and-Play","metrics":{"Background LPIPS":"106.06","Background PSNR":"22.46","CLIPSIM":"25.41","Structure Distance":"24.29"},"uses_additional_data":false,"paper_date":"2023-10-02","paper":"/paper/direct-inversion-boosting-diffusion-based","paper_url":"https://arxiv.org/abs/2310.01506v2","paper_title":"Direct Inversion: Boosting Diffusion-based Editing with 3 Lines of Code","code":"https://github.com/cure-lab/pnpinversion","n_code_links":3,"syntology":{"n_ran":4,"n_unverified":2,"n_samples":6,"n_pointer_only_licence":6}},{"rank_in_archive_order":13,"model":"DDIM Inversion+Plug-and-Play","metrics":{"Background LPIPS":"113.46","Background PSNR":"22.28","CLIPSIM":"25.41","Structure Distance":"28.22"},"uses_additional_data":false,"paper_date":"2022-11-22","paper":"/paper/plug-and-play-diffusion-features-for-text","paper_url":"https://arxiv.org/abs/2211.12572v1","paper_title":"Plug-and-Play Diffusion Features for Text-Driven Image-to-Image Translation","code":"https://github.com/MichalGeyer/plug-and-play","n_code_links":4,"syntology":{"n_ran":1,"n_unverified":0,"n_samples":1,"n_pointer_only_licence":0}},{"rank_in_archive_order":14,"model":"DDIM Inversion+MasaCtrl","metrics":{"Background LPIPS":"106.62","Background PSNR":"22.17","CLIPSIM":"23.96","Structure Distance":"28.38"},"uses_additional_data":false,"paper_date":"2023-04-17","paper":"/paper/masactrl-tuning-free-mutual-self-attention","paper_url":"https://arxiv.org/abs/2304.08465v1","paper_title":"MasaCtrl: Tuning-Free Mutual Self-Attention Control for Consistent Image Synthesis and Editing","code":"https://github.com/tencentarc/masactrl","n_code_links":4,"syntology":{"n_ran":4,"n_unverified":3,"n_samples":7,"n_pointer_only_licence":1}},{"rank_in_archive_order":15,"model":"Direct Inversion+Pix2Pix-Zero","metrics":{"Background LPIPS":"138.98","Background PSNR":"21.53","CLIPSIM":"23.31","Structure Distance":"49.22"},"uses_additional_data":false,"paper_date":"2023-10-02","paper":"/paper/direct-inversion-boosting-diffusion-based","paper_url":"https://arxiv.org/abs/2310.01506v2","paper_title":"Direct Inversion: Boosting Diffusion-based Editing with 3 Lines of Code","code":"https://github.com/cure-lab/pnpinversion","n_code_links":3,"syntology":{"n_ran":4,"n_unverified":2,"n_samples":6,"n_pointer_only_licence":6}},{"rank_in_archive_order":16,"model":"DDIM Inversion+Pix2Pix-Zero","metrics":{"Background LPIPS":"172.22","Background PSNR":"20.44","CLIPSIM":"22.80","Structure Distance":"61.68"},"uses_additional_data":false,"paper_date":"2023-02-06","paper":"/paper/zero-shot-image-to-image-translation","paper_url":"https://arxiv.org/abs/2302.03027v1","paper_title":"Zero-shot Image-to-Image Translation","code":"https://github.com/pix2pixzero/pix2pix-zero","n_code_links":2,"syntology":{"n_ran":0,"n_unverified":3,"n_samples":3,"n_pointer_only_licence":0}},{"rank_in_archive_order":17,"model":"DDIM Inversion+Prompt-to-Prompt","metrics":{"Background LPIPS":"208.80","Background PSNR":"17.87","CLIPSIM":"25.01","Structure Distance":"69.43"},"uses_additional_data":false,"paper_date":"2022-08-02","paper":"/paper/prompt-to-prompt-image-editing-with-cross","paper_url":"https://arxiv.org/abs/2208.01626v1","paper_title":"Prompt-to-Prompt Image Editing with Cross Attention Control","code":"https://github.com/google/prompt-to-prompt","n_code_links":7,"syntology":{"n_ran":6,"n_unverified":9,"n_samples":15,"n_pointer_only_licence":3}},{"rank_in_archive_order":18,"model":"FireFlow (Add Q)","metrics":{"Background LPIPS":"239.4","Background PSNR":"16.49","CLIPSIM":"27.33","Structure Distance":"70.9"},"uses_additional_data":false,"paper_date":"2024-12-10","paper":"/paper/fireflow-fast-inversion-of-rectified-flow-for","paper_url":"https://arxiv.org/abs/2412.07517v1","paper_title":"FireFlow: Fast Inversion of Rectified Flow for Image Semantic Editing","code":"https://github.com/holmesshuan/fireflow","n_code_links":1,"syntology":{"n_ran":4,"n_unverified":7,"n_samples":11,"n_pointer_only_licence":0}}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":16,"rows_with_any_sample_ran":15,"distinct_papers_with_graph_line":10,"distinct_papers_with_any_sample_ran":9,"samples_over_distinct_papers":{"n_ran":37,"n_unverified":40,"n_samples":77,"n_pointer_only_licence":12,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":55,"n_unverified":53,"n_samples":108,"n_pointer_only_licence":32,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}