{"url":"/method/stacked-hourglass-network","slug":"stacked-hourglass-network","name":"Stacked Hourglass Network","full_name":"Stacked Hourglass Network","full_name_withheld":false,"description_markdown":"**Stacked Hourglass Networks** are a type of convolutional neural network for pose estimation. They are based on the successive steps of pooling and upsampling that are done to produce a final set of predictions.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Stacked Hourglass Networks for Human Pose Estimation","paper":"/paper/stacked-hourglass-networks-for-human-pose","first_author":"Alejandro Newell","n_authors":3,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/stacked-hourglass-networks-for-human-pose"},"source":{"url":"http://arxiv.org/abs/1603.06937v2","title":"Stacked Hourglass Networks for Human Pose Estimation","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Pose Estimation Models","url":"/methods/category/pose-estimation-models","pwc_aliases":[]}],"n_papers_tagged":28,"archive_num_papers":28,"papers_newest_first":[{"paper":"/paper/to-perceive-or-not-to-perceive-lightweight","title":"To Perceive or Not to Perceive: Lightweight Stacked Hourglass Network","date":"2023-02-09","arxiv_id":"2302.04815","n_code_links":1,"syntology":null},{"paper":"/paper/robust-table-detection-and-structure","title":"Robust Table Detection and Structure Recognition from Heterogeneous Document Images","date":"2022-03-17","arxiv_id":"2203.09056","n_code_links":0,"syntology":null},{"paper":null,"title":"Deep Point Cloud Reconstruction","date":"2021-11-23","arxiv_id":"2111.11704","n_code_links":0,"syntology":null},{"paper":"/paper/stacked-hourglass-network-with-a-multi-level","title":"Stacked Hourglass Network with a Multi-level Attention Mechanism: Where to Look for Intervertebral Disc Labeling","date":"2021-08-14","arxiv_id":"2108.06554","n_code_links":1,"syntology":null},{"paper":"/paper/volnet-estimating-human-body-part-volumes","title":"VolNet: Estimating Human Body Part Volumes from a Single RGB Image","date":"2021-07-05","arxiv_id":"2107.02259","n_code_links":0,"syntology":null},{"paper":null,"title":"Automatic segmentation of vertebral features on ultrasound spine images using Stacked Hourglass Network","date":"2021-05-09","arxiv_id":"2105.03847","n_code_links":0,"syntology":null},{"paper":null,"title":"TetraPackNet: Four-Corner-Based Object Detection in Logistics Use-Cases","date":"2021-04-19","arxiv_id":"2104.09123","n_code_links":0,"syntology":null},{"paper":"/paper/traffic-camera-calibration-via-vehicle","title":"Traffic Camera Calibration via Vehicle Vanishing Point Detection","date":"2021-03-21","arxiv_id":"2103.11438","n_code_links":2,"syntology":null},{"paper":"/paper/relationnet-bridging-visual-representations","title":"RelationNet++: Bridging Visual Representations for Object Detection via Transformer Decoder","date":"2020-10-29","arxiv_id":"2010.15831","n_code_links":4,"syntology":{"ran":0,"of":5,"unverified":5,"pointer_only":0}},{"paper":"/paper/houghnet-integrating-near-and-long-range","title":"HoughNet: Integrating near and long-range evidence for bottom-up object detection","date":"2020-07-05","arxiv_id":"2007.02355","n_code_links":2,"syntology":null},{"paper":null,"title":"3D Pose Detection in Videos: Focusing on Occlusion","date":"2020-06-24","arxiv_id":"2006.13517","n_code_links":0,"syntology":null},{"paper":"/paper/rgbd-dog-predicting-canine-pose-from-rgbd","title":"RGBD-Dog: Predicting Canine Pose from RGBD Sensors","date":"2020-04-16","arxiv_id":"2004.07788","n_code_links":1,"syntology":null},{"paper":null,"title":"Single upper limb pose estimation method based on improved stacked hourglass network","date":"2020-04-16","arxiv_id":"2004.07456","n_code_links":0,"syntology":null},{"paper":"/paper/spotnet-self-attention-multi-task-network-for","title":"SpotNet: Self-Attention Multi-Task Network for Object Detection","date":"2020-02-13","arxiv_id":"2002.05540","n_code_links":1,"syntology":null},{"paper":null,"title":"Multistage Model for Robust Face Alignment Using Deep Neural Networks","date":"2020-02-04","arxiv_id":"2002.01075","n_code_links":0,"syntology":null},{"paper":"/paper/matrixnets-a-new-scale-and-aspect-ratio-aware","title":"MatrixNets: A New Scale and Aspect Ratio Aware Architecture for Object Detection","date":"2020-01-09","arxiv_id":"2001.03194","n_code_links":1,"syntology":null},{"paper":"/paper/simple-pose-rethinking-and-improving-a-bottom","title":"Simple Pose: Rethinking and Improving a Bottom-up Approach for Multi-Person Pose Estimation","date":"2019-11-24","arxiv_id":"1911.10529","n_code_links":8,"syntology":null},{"paper":null,"title":"Single-shot 3D multi-person pose estimation in complex images","date":"2019-11-08","arxiv_id":"1911.03391","n_code_links":0,"syntology":null},{"paper":null,"title":"Analyzing Large Receptive Field Convolutional Networks for Distant Speech Recognition","date":"2019-10-15","arxiv_id":"1910.07047","n_code_links":0,"syntology":null},{"paper":null,"title":"Multi-task Localization and Segmentation for X-ray Guided Planning in Knee Surgery","date":"2019-07-24","arxiv_id":"1907.10465","n_code_links":0,"syntology":null},{"paper":"/paper/190408900","title":"CornerNet-Lite: Efficient Keypoint Based Object Detection","date":"2019-04-18","arxiv_id":"1904.08900","n_code_links":6,"syntology":{"ran":1,"of":27,"unverified":26,"pointer_only":0}},{"paper":"/paper/centernet-object-detection-with-keypoint","title":"CenterNet: Keypoint Triplets for Object Detection","date":"2019-04-17","arxiv_id":"1904.08189","n_code_links":20,"syntology":{"ran":2,"of":11,"unverified":9,"pointer_only":2}},{"paper":"/paper/group-wise-correlation-stereo-network","title":"Group-wise Correlation Stereo Network","date":"2019-03-10","arxiv_id":"1903.04025","n_code_links":2,"syntology":{"ran":0,"of":2,"unverified":2,"pointer_only":0}},{"paper":null,"title":"Exploring Stereovision-Based 3-D Scene Reconstruction for Augmented Reality","date":"2019-02-17","arxiv_id":"1902.06255","n_code_links":0,"syntology":null},{"paper":"/paper/bottom-up-object-detection-by-grouping","title":"Bottom-up Object Detection by Grouping Extreme and Center Points","date":"2019-01-23","arxiv_id":"1901.08043","n_code_links":2,"syntology":{"ran":1,"of":5,"unverified":4,"pointer_only":0}},{"paper":"/paper/cornernet-detecting-objects-as-paired","title":"CornerNet: Detecting Objects as Paired Keypoints","date":"2018-08-03","arxiv_id":"1808.01244","n_code_links":5,"syntology":{"ran":3,"of":11,"unverified":8,"pointer_only":0}},{"paper":null,"title":"Semi-Automatic RECIST Labeling on CT Scans with Cascaded Convolutional Neural Networks","date":"2018-06-25","arxiv_id":"1806.09507","n_code_links":0,"syntology":null},{"paper":"/paper/stacked-hourglass-networks-for-human-pose","title":"Stacked Hourglass Networks for Human Pose Estimation","date":"2016-03-22","arxiv_id":"1603.06937","n_code_links":46,"syntology":{"ran":2,"of":25,"unverified":23,"pointer_only":1}}],"papers_shown":28,"tasks":[{"task":"/task/object","name":"Object","papers":9},{"task":"/task/object-detection","name":"Object Detection","papers":9},{"task":"/task/object-detection-1","name":"object-detection","papers":9},{"task":"/task/pose-estimation","name":"Pose Estimation","papers":8},{"task":"/task/decoder","name":"Decoder","papers":2},{"task":"/task/multi-person-pose-estimation","name":"Multi-Person Pose Estimation","papers":2},{"task":"/task/multi-task-learning","name":"Multi-Task Learning","papers":2},{"task":"/task/segmentation","name":"Segmentation","papers":2},{"task":"/task/semantic-segmentation","name":"Semantic Segmentation","papers":2},{"task":"/task/stereo-matching-1","name":"Stereo Matching","papers":2},{"task":"/task/stereo-matching","name":"Stereo Matching Hand","papers":2},{"task":"/task/2d-human-pose-estimation","name":"2D Human Pose Estimation","papers":1},{"task":"/task/3d-human-pose-estimation","name":"3D Human Pose Estimation","papers":1},{"task":"/task/3d-multi-person-human-pose-estimation","name":"3D Multi-Person Human Pose Estimation","papers":1},{"task":"/task/3d-multi-person-pose-estimation","name":"3D Multi-Person Pose Estimation","papers":1},{"task":"/task/3d-pose-estimation","name":"3D Pose Estimation","papers":1},{"task":"/task/3d-volumetric-reconstruction","name":"3D Volumetric Reconstruction","papers":1},{"task":"/task/automatic-speech-recognition-2","name":"Automatic Speech Recognition","papers":1},{"task":"/task/automatic-speech-recognition","name":"Automatic Speech Recognition (ASR)","papers":1},{"task":"/task/autonomous-driving","name":"Autonomous Driving","papers":1}],"tasks_shown":20,"n_tasks":45,"usage_by_year":[{"year":"2016","papers":1},{"year":"2018","papers":2},{"year":"2019","papers":9},{"year":"2020","papers":8},{"year":"2021","papers":6},{"year":"2022","papers":1},{"year":"2023","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/stacked-hourglass-network"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}