{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/usp-unified-self-supervised-pretraining-for","title":"USP: Unified Self-Supervised Pretraining for Image Generation and Understanding","arxiv_id":"2503.06132","date":"2025-03-08","proceeding":null,"authors":["Xiangxiang Chu","Renda Li","Yong Wang"],"abstract":"Recent studies have highlighted the interplay between diffusion models and representation learning. Intermediate representations from diffusion models can be leveraged for downstream visual tasks, while self-supervised vision models can enhance the convergence and generation quality of diffusion models. However, transferring pretrained weights from vision models to diffusion models is challenging due to input mismatches and the use of latent spaces. To address these challenges, we propose Unified Self-supervised Pretraining (USP), a framework that initializes diffusion models via masked latent modeling in a Variational Autoencoder (VAE) latent space. USP achieves comparable performance in understanding tasks while significantly improving the convergence speed and generation quality of diffusion models. Our code will be publicly available at https://github.com/cxxgtxy/USP.","url_abs":"https://arxiv.org/abs/2503.06132v1","url_pdf":"https://arxiv.org/pdf/2503.06132v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"usp-unified-self-supervised-pretraining-for","repo_url":"https://github.com/cxxgtxy/usp","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"image-generation","task_name":"Image Generation"},{"task_slug":"representation-learning","task_name":"Representation Learning"}],"methods":[{"method_slug":"diffusion","method_name":"Diffusion"},{"method_slug":"speed","method_name":"SPEED"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2503.06132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06132"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/cxxgtxy/usp","reach":null},{"provenance":"deterministic:regex_extraction","url":"https://github.com/AMAP-ML/USP","reach":null}],"summary":{"ran":2,"unverified":2},"by_repo_kind":{"official":{"samples":2,"ran":1,"repositories":1},"found_in_text":{"samples":2,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":4,"samples":[{"code_sha256_prefix":"08a60fcbc2703677","entry":"DiTBlock","repo":"cxxgtxy/usp","repo_kind":"official","path":"pretrain/models_dit.py","file_url":"https://github.com/cxxgtxy/usp/blob/HEAD/pretrain/models_dit.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"08a60fcbc2703677"}},{"code_sha256_prefix":"f722666b1c98510c","entry":"FinalLayer","repo":"AMAP-ML/USP","repo_kind":"found_in_text","path":"generation/SiT/models.py","file_url":"https://github.com/AMAP-ML/USP/blob/HEAD/generation/SiT/models.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"f722666b1c98510c"}},{"code_sha256_prefix":"55497dad79b64c76","entry":"DiT","repo":"cxxgtxy/usp","repo_kind":"official","path":"pretrain/models_dit.py","file_url":"https://github.com/cxxgtxy/usp/blob/HEAD/pretrain/models_dit.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"55497dad79b64c76"}},{"code_sha256_prefix":"24d4fd1474533f21","entry":"SiT","repo":"AMAP-ML/USP","repo_kind":"found_in_text","path":"generation/SiT/models.py","file_url":"https://github.com/AMAP-ML/USP/blob/HEAD/generation/SiT/models.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"24d4fd1474533f21"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}