{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/better-aligning-text-to-image-models-with","title":"Human Preference Score: Better Aligning Text-to-Image Models with Human Preference","arxiv_id":"2303.14420","date":"2023-03-25","proceeding":"ICCV 2023 1","authors":["Xiaoshi Wu","Keqiang Sun","Feng Zhu","Rui Zhao","Hongsheng Li"],"abstract":"Recent years have witnessed a rapid growth of deep generative models, with text-to-image models gaining significant attention from the public. However, existing models often generate images that do not align well with human preferences, such as awkward combinations of limbs and facial expressions. To address this issue, we collect a dataset of human choices on generated images from the Stable Foundation Discord channel. Our experiments demonstrate that current evaluation metrics for generative models do not correlate well with human choices. Thus, we train a human preference classifier with the collected dataset and derive a Human Preference Score (HPS) based on the classifier. Using HPS, we propose a simple yet effective method to adapt Stable Diffusion to better align with human preferences. Our experiments show that HPS outperforms CLIP in predicting human choices and has good generalization capability toward images generated from other models. By tuning Stable Diffusion with the guidance of HPS, the adapted model is able to generate images that are more preferred by human users. The project page is available here: https://tgxs002.github.io/align_sd_web/ .","url_abs":"https://arxiv.org/abs/2303.14420v2","url_pdf":"https://arxiv.org/pdf/2303.14420v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"better-aligning-text-to-image-models-with","repo_url":"https://github.com/tgxs002/align_sd","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}}],"tasks":[],"methods":[{"method_slug":"align","method_name":"ALIGN"},{"method_slug":"clip","method_name":"CLIP"},{"method_slug":"diffusion","method_name":"Diffusion"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/2303.14420","atlas_url":"https://app.syntology.ai/?focus=2303.14420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.14420"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/tgxs002/align_sd","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"ran":3,"unverified":1},"by_repo_kind":{"official":{"samples":4,"ran":3,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"fa69ab1c069908fe","entry":"is_blurry","repo":"tgxs002/align_sd","repo_kind":"official","path":"process_diffusiondb.py","file_url":"https://github.com/tgxs002/align_sd/blob/HEAD/process_diffusiondb.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"fa69ab1c069908fe"}},{"code_sha256_prefix":"4cb345d10a4a752b","entry":"softmax","repo":"tgxs002/align_sd","repo_kind":"official","path":"select_training_images.py","file_url":"https://github.com/tgxs002/align_sd/blob/HEAD/select_training_images.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"4cb345d10a4a752b"}},{"code_sha256_prefix":"25a13777998d6d16","entry":"to_device","repo":"tgxs002/align_sd","repo_kind":"official","path":"process_diffusiondb.py","file_url":"https://github.com/tgxs002/align_sd/blob/HEAD/process_diffusiondb.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"25a13777998d6d16"}},{"code_sha256_prefix":"6431708d8f3ff177","entry":"get_full_repo_name","repo":"tgxs002/align_sd","repo_kind":"official","path":"train_text_to_image_lora.py","file_url":"https://github.com/tgxs002/align_sd/blob/HEAD/train_text_to_image_lora.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"6431708d8f3ff177"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}