{"url":"/method/zoomnet","slug":"zoomnet","name":"ZoomNet","full_name":"ZoomNet","full_name_withheld":false,"description_markdown":"**ZoomNet** is a 2D human whole-body pose estimation technique. It aims to localize dense landmarks on the entire human body including face, hands, body, and feet. ZoomNet follows the top-down paradigm. Given a human bounding box of each person, ZoomNet first localizes the easy-to-detect body keypoints and estimates the rough position of hands and face. Then it zooms in to focus on the hand/face areas and predicts keypoints using features with higher resolution for accurate localization. Unlike previous approaches which usually assemble multiple networks, ZoomNet has a single network that is end-to-end trainable. It unifies five network heads including the human body pose estimator, hand and face detectors, and hand and face pose estimators into a single network with shared low-level features.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Whole-Body Human Pose Estimation in the Wild","paper":"/paper/whole-body-human-pose-estimation-in-the-wild","first_author":"Sheng Jin","n_authors":8,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/whole-body-human-pose-estimation-in-the-wild"},"source":{"url":"https://arxiv.org/abs/2007.11858v1","title":"Whole-Body Human Pose Estimation in the Wild","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Pose Estimation Models","url":"/methods/category/pose-estimation-models","pwc_aliases":[]}],"n_papers_tagged":3,"archive_num_papers":3,"papers_newest_first":[{"paper":"/paper/zoomnas-searching-for-whole-body-human-pose","title":"ZoomNAS: Searching for Whole-body Human Pose Estimation in the Wild","date":"2022-08-23","arxiv_id":"2208.11547","n_code_links":1,"syntology":null},{"paper":"/paper/zoom-in-and-out-a-mixed-scale-triplet-network","title":"Zoom In and Out: A Mixed-scale Triplet Network for Camouflaged Object Detection","date":"2022-03-05","arxiv_id":"2203.02688","n_code_links":1,"syntology":{"ran":12,"of":13,"unverified":1,"pointer_only":0}},{"paper":"/paper/whole-body-human-pose-estimation-in-the-wild","title":"Whole-Body Human Pose Estimation in the Wild","date":"2020-07-23","arxiv_id":"2007.11858","n_code_links":2,"syntology":null}],"papers_shown":3,"tasks":[{"task":"/task/2d-human-pose-estimation","name":"2D Human Pose Estimation","papers":2},{"task":"/task/pose-estimation","name":"Pose Estimation","papers":2},{"task":"/task/camouflaged-object-segmentation","name":"Camouflaged Object Segmentation","papers":1},{"task":"/task/facial-landmark-detection","name":"Facial Landmark Detection","papers":1},{"task":"/task/hand-pose-estimation","name":"Hand Pose Estimation","papers":1},{"task":"/task/image-segmentation","name":"Image Segmentation","papers":1},{"task":"/task/keypoint-estimation","name":"Keypoint Estimation","papers":1},{"task":"/task/architecture-search","name":"Neural Architecture Search","papers":1},{"task":"/task/object-detection","name":"Object Detection","papers":1},{"task":null,"name":"Triplet","papers":1},{"task":"/task/object-detection-1","name":"object-detection","papers":1}],"tasks_shown":11,"n_tasks":11,"usage_by_year":[{"year":"2020","papers":1},{"year":"2022","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/zoomnet"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}