{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/hyperseg-patch-wise-hypernetwork-for-real","title":"HyperSeg: Patch-wise Hypernetwork for Real-time Semantic Segmentation","arxiv_id":"2012.11582","date":"2020-12-21","proceeding":"CVPR 2021 1","authors":["Yuval Nirkin","Lior Wolf","Tal Hassner"],"abstract":"We present a novel, real-time, semantic segmentation network in which the encoder both encodes and generates the parameters (weights) of the decoder. Furthermore, to allow maximal adaptivity, the weights at each decoder block vary spatially. For this purpose, we design a new type of hypernetwork, composed of a nested U-Net for drawing higher level context features, a multi-headed weight generating module which generates the weights of each block in the decoder immediately before they are consumed, for efficient memory utilization, and a primary network that is composed of novel dynamic patch-wise convolutions. Despite the usage of less-conventional blocks, our architecture obtains real-time performance. In terms of the runtime vs. accuracy trade-off, we surpass state of the art (SotA) results on popular semantic segmentation benchmarks: PASCAL VOC 2012 (val. set) and real-time semantic segmentation on Cityscapes, and CamVid. The code is available: https://nirkin.com/hyperseg.","url_abs":"https://arxiv.org/abs/2012.11582v2","url_pdf":"https://arxiv.org/pdf/2012.11582v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"hyperseg-patch-wise-hypernetwork-for-real","repo_url":"https://github.com/YuvalNirkin/hyperseg","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}}],"tasks":[{"task_slug":"decoder","task_name":"Decoder"},{"task_slug":"dichotomous-image-segmentation","task_name":"Dichotomous Image Segmentation"},{"task_slug":"real-time-semantic-segmentation","task_name":"Real-Time Semantic Segmentation"},{"task_slug":"segmentation","task_name":"Segmentation"},{"task_slug":"semantic-segmentation","task_name":"Semantic Segmentation"}],"methods":[{"method_slug":"concatenated-skip-connection","method_name":"Concatenated Skip Connection"},{"method_slug":"convolution","method_name":"Convolution"},{"method_slug":"hypernetwork","method_name":"HyperNetwork"},{"method_slug":"max-pooling","method_name":"Max Pooling"},{"method_slug":"relu","method_name":"ReLU"},{"method_slug":"u-net","method_name":"U-Net"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/dichotomous-image-segmentation-on-dis-te1","task":"Dichotomous Image Segmentation","dataset":"DIS-TE1","model":"HySM","rank_in_archive_order":7,"of":22,"metrics":{"E-measure":"0.803","HCE":"205","MAE":"0.082","S-Measure":"0.761","max F-Measure":"0.695","weighted F-measure":"0.597"},"uses_additional_data":false},{"leaderboard":"/sota/dichotomous-image-segmentation-on-dis-te2","task":"Dichotomous Image Segmentation","dataset":"DIS-TE2","model":"HySM","rank_in_archive_order":7,"of":22,"metrics":{"E-measure":"0.832","HCE":"451","MAE":"0.085","S-Measure":"0.794","max F-Measure":"0.759","weighted F-measure":"0.667"},"uses_additional_data":false},{"leaderboard":"/sota/dichotomous-image-segmentation-on-dis-te3","task":"Dichotomous Image Segmentation","dataset":"DIS-TE3","model":"HySM","rank_in_archive_order":8,"of":22,"metrics":{"E-measure":"0.857","HCE":"887","MAE":"0.079","S-Measure":"0.811","max F-Measure":"0.792","weighted F-measure":"0.701"},"uses_additional_data":false},{"leaderboard":"/sota/dichotomous-image-segmentation-on-dis-te4","task":"Dichotomous Image Segmentation","dataset":"DIS-TE4","model":"HySM","rank_in_archive_order":8,"of":22,"metrics":{"E-measure":"0.842","HCE":"3331","MAE":"0.091","S-Measure":"0.802","max F-Measure":"0.782","weighted F-measure":"0.693"},"uses_additional_data":false},{"leaderboard":"/sota/dichotomous-image-segmentation-on-dis-vd","task":"Dichotomous Image Segmentation","dataset":"DIS-VD","model":"HySM","rank_in_archive_order":10,"of":24,"metrics":{"E-measure":"0.814","HCE":"1324","MAE":"0.096","S-Measure":"0.773","max F-Measure":"0.734","weighted F-measure":"0.640"},"uses_additional_data":false},{"leaderboard":"/sota/real-time-semantic-segmentation-on-camvid","task":"Real-Time Semantic Segmentation","dataset":"CamVid","model":"HyperSeg-L","rank_in_archive_order":7,"of":29,"metrics":{"Frame (fps)":"16.6","Time (ms)":"60.2","mIoU":"79.1"},"uses_additional_data":true},{"leaderboard":"/sota/real-time-semantic-segmentation-on-camvid","task":"Real-Time Semantic Segmentation","dataset":"CamVid","model":"HyperSeg-S","rank_in_archive_order":9,"of":29,"metrics":{"Frame (fps)":"38.0","Time (ms)":"26.3","mIoU":"78.4"},"uses_additional_data":false},{"leaderboard":"/sota/real-time-semantic-segmentation-on-cityscapes","task":"Real-Time Semantic Segmentation","dataset":"Cityscapes test","model":"HyperSeg-M","rank_in_archive_order":11,"of":39,"metrics":{"Frame (fps)":"36.9","Time (ms)":"27.1","mIoU":"75.8%"},"uses_additional_data":false},{"leaderboard":"/sota/semantic-segmentation-on-pascal-voc-2012-val","task":"Semantic Segmentation","dataset":"PASCAL VOC 2012 val","model":"HyperSeg-L","rank_in_archive_order":9,"of":29,"metrics":{"mIoU":"80.61%"},"uses_additional_data":false}],"syntology":{"syntology_url":null,"atlas_url":"https://app.syntology.ai/?focus=2012.11582","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}