{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/beyond-appearance-a-semantic-controllable","title":"Beyond Appearance: a Semantic Controllable Self-Supervised Learning Framework for Human-Centric Visual Tasks","arxiv_id":"2303.17602","date":"2023-03-30","proceeding":"CVPR 2023 1","authors":["Weihua Chen","Xianzhe Xu","Jian Jia","Hao Luo","Yaohua Wang","Fan Wang","Rong Jin","Xiuyu Sun"],"abstract":"Human-centric visual tasks have attracted increasing research attention due to their widespread applications. In this paper, we aim to learn a general human representation from massive unlabeled human images which can benefit downstream human-centric tasks to the maximum extent. We call this method SOLIDER, a Semantic cOntrollable seLf-supervIseD lEaRning framework. Unlike the existing self-supervised learning methods, prior knowledge from human images is utilized in SOLIDER to build pseudo semantic labels and import more semantic information into the learned representation. Meanwhile, we note that different downstream tasks always require different ratios of semantic information and appearance information. For example, human parsing requires more semantic information, while person re-identification needs more appearance information for identification purpose. So a single learned representation cannot fit for all requirements. To solve this problem, SOLIDER introduces a conditional network with a semantic controller. After the model is trained, users can send values to the controller to produce representations with different ratios of semantic information, which can fit different needs of downstream tasks. Finally, SOLIDER is verified on six downstream human-centric visual tasks. It outperforms state of the arts and builds new baselines for these tasks. The code is released in https://github.com/tinyvision/SOLIDER.","url_abs":"https://arxiv.org/abs/2303.17602v1","url_pdf":"https://arxiv.org/pdf/2303.17602v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"beyond-appearance-a-semantic-controllable","repo_url":"https://github.com/tinyvision/SOLIDER","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"beyond-appearance-a-semantic-controllable","repo_url":"https://github.com/DengpanFu/LUPerson","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"beyond-appearance-a-semantic-controllable","repo_url":"https://github.com/hasanirtiza/Pedestron","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"beyond-appearance-a-semantic-controllable","repo_url":"https://github.com/modelscope/modelscope","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}}],"tasks":[{"task_slug":"human-parsing","task_name":"Human Parsing"},{"task_slug":"pedestrian-attribute-recognition","task_name":"Pedestrian Attribute Recognition"},{"task_slug":"pedestrian-detection","task_name":"Pedestrian Detection"},{"task_slug":"person-re-identification","task_name":"Person Re-Identification"},{"task_slug":"person-search","task_name":"Person Search"},{"task_slug":"pose-estimation","task_name":"Pose Estimation"},{"task_slug":"self-supervised-learning","task_name":"Self-Supervised Learning"},{"task_slug":"semantic-segmentation","task_name":"Semantic Segmentation"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/pedestrian-attribute-recognition-on-pa-100k","task":"Pedestrian Attribute Recognition","dataset":"PA-100K","model":"SOLIDER","rank_in_archive_order":5,"of":13,"metrics":{"Accuracy":"86.38"},"uses_additional_data":true},{"leaderboard":"/sota/pedestrian-detection-on-citypersons","task":"Pedestrian Detection","dataset":"CityPersons","model":"SOLIDER","rank_in_archive_order":9,"of":22,"metrics":{"Heavy MR^-2":"39.4","Reasonable MR^-2":"9.7"},"uses_additional_data":true},{"leaderboard":"/sota/person-re-identification-on-msmt17","task":"Person Re-Identification","dataset":"MSMT17","model":"SOLIDER (with re-ranking)","rank_in_archive_order":2,"of":43,"metrics":{"Rank-1":"91.7","mAP":"86.5"},"uses_additional_data":true},{"leaderboard":"/sota/person-re-identification-on-msmt17","task":"Person Re-Identification","dataset":"MSMT17","model":"SOLIDER (without re-ranking)","rank_in_archive_order":6,"of":43,"metrics":{"Rank-1":"90.7","mAP":"77.1"},"uses_additional_data":true},{"leaderboard":"/sota/person-re-identification-on-market-1501","task":"Person Re-Identification","dataset":"Market-1501","model":"SOLIDER","rank_in_archive_order":7,"of":135,"metrics":{"Rank-1":"96.9","mAP":"93.9"},"uses_additional_data":false},{"leaderboard":"/sota/person-re-identification-on-market-1501","task":"Person Re-Identification","dataset":"Market-1501","model":"SOLIDER (RK)","rank_in_archive_order":10,"of":135,"metrics":{"Rank-1":"96.7","mAP":"95.6"},"uses_additional_data":false},{"leaderboard":"/sota/person-re-identification-on-occluded-dukemtmc","task":"Person Re-Identification","dataset":"Occluded-DukeMTMC","model":"SOLIDER","rank_in_archive_order":7,"of":32,"metrics":{" Rank-1":"71.2","mAP":"61.9"},"uses_additional_data":false},{"leaderboard":"/sota/person-search-on-cuhk-sysu","task":"Person Search","dataset":"CUHK-SYSU","model":"SOLIDER","rank_in_archive_order":6,"of":16,"metrics":{"MAP":"95.5","Top-1":"95.8"},"uses_additional_data":false},{"leaderboard":"/sota/person-search-on-prw","task":"Person Search","dataset":"PRW","model":"SOLIDER","rank_in_archive_order":2,"of":15,"metrics":{"Top-1":"86.7","mAP":"59.8"},"uses_additional_data":false},{"leaderboard":"/sota/pose-estimation-on-coco","task":"Pose Estimation","dataset":"COCO (Common Objects in Context)","model":"SOLIDER (swin-B)","rank_in_archive_order":8,"of":10,"metrics":{"AP":"76.6","AR":"81.5"},"uses_additional_data":false},{"leaderboard":"/sota/semantic-segmentation-on-lip-val","task":"Semantic Segmentation","dataset":"LIP val","model":"SOLIDER","rank_in_archive_order":4,"of":13,"metrics":{"mIoU":"60.50%"},"uses_additional_data":false}],"syntology":{"syntology_url":"https://syntology.ai/paper/2303.17602","atlas_url":"https://app.syntology.ai/?focus=2303.17602","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17602"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/modelscope/modelscope","reach":{"status":"ok","spdx":"Apache-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/DengpanFu/LUPerson","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/hasanirtiza/Pedestron","reach":{"status":"ok","spdx":"Apache-2.0"}},{"provenance":"deterministic:regex_extraction","url":"https://github.com/tinyvision/SOLIDER","reach":null}],"summary":{"ran":1,"unverified":1},"by_repo_kind":{"official":{"samples":1,"ran":0,"repositories":1},"listed":{"samples":1,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":1,"samples":[{"code_sha256_prefix":"d94422fb61f76b24","entry":"LUPNet","repo":"DengpanFu/LUPerson","repo_kind":"listed","path":"lup_moco/libs/encoder.py","file_url":"https://github.com/DengpanFu/LUPerson/blob/HEAD/lup_moco/libs/encoder.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"d94422fb61f76b24"}},{"code_sha256_prefix":"e3eca5e066e1e795","entry":"DINOLoss","repo":"tinyvision/SOLIDER","repo_kind":"official","path":"main_solider.py","file_url":"https://github.com/tinyvision/SOLIDER/blob/HEAD/main_solider.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"e3eca5e066e1e795"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}