{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/self-supervised-learning-of-a-facial","title":"Self-supervised learning of a facial attribute embedding from video","arxiv_id":"1808.06882","date":"2018-08-21","proceeding":null,"authors":["Olivia Wiles","A. Sophia Koepke","Andrew Zisserman"],"abstract":"We propose a self-supervised framework for learning facial attributes by\nsimply watching videos of a human face speaking, laughing, and moving over\ntime. To perform this task, we introduce a network, Facial Attributes-Net\n(FAb-Net), that is trained to embed multiple frames from the same video\nface-track into a common low-dimensional space. With this approach, we make\nthree contributions: first, we show that the network can leverage information\nfrom multiple source frames by predicting confidence/attention masks for each\nframe; second, we demonstrate that using a curriculum learning regime improves\nthe learned embedding; finally, we demonstrate that the network learns a\nmeaningful face embedding that encodes information about head pose, facial\nlandmarks and facial expression, i.e. facial attributes, without having been\nsupervised with any labelled data. We are comparable or superior to\nstate-of-the-art self-supervised methods on these tasks and approach the\nperformance of supervised methods.","url_abs":"http://arxiv.org/abs/1808.06882v1","url_pdf":"http://arxiv.org/pdf/1808.06882v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"self-supervised-learning-of-a-facial","repo_url":"https://github.com/jiarenchang/facecycle","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"self-supervised-learning-of-a-facial","repo_url":"https://github.com/oawiles/FAb-Net","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"unanswered"}}],"tasks":[{"task_slug":"attribute","task_name":"Attribute"},{"task_slug":"self-supervised-learning","task_name":"Self-Supervised Learning"},{"task_slug":"unsupervised-facial-landmark-detection","task_name":"Unsupervised Facial Landmark Detection"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/unsupervised-facial-landmark-detection-on","task":"Unsupervised Facial Landmark Detection","dataset":"300W","model":"FAb-Net","rank_in_archive_order":2,"of":4,"metrics":{"NME":"5.71"},"uses_additional_data":false},{"leaderboard":"/sota/unsupervised-facial-landmark-detection-on-1","task":"Unsupervised Facial Landmark Detection","dataset":"MAFL","model":"FAB-Net","rank_in_archive_order":6,"of":13,"metrics":{"NME":"3.44"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1808.06882","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}