{"url":"/method/vip-deeplab","slug":"vip-deeplab","name":"ViP-DeepLab","full_name":"ViP-DeepLab","full_name_withheld":false,"description_markdown":"**ViP-DeepLab** is a model for depth-aware video panoptic segmentation. It extends Panoptic-[DeepLab](https://paperswithcode.com/method/deeplab) by adding a depth prediction head to perform monocular depth estimation and a next-frame instance branch which regresses to the object centers in frame $t$ for frame $t + 1$.  This allows the model to jointly perform video panoptic segmentation and monocular depth estimation.","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":"https://arxiv.org/abs/2012.05258v1","title":"ViP-DeepLab: Learning Visual Perception with Depth-aware Video Panoptic Segmentation","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Video Panoptic Segmentation Models","url":"/methods/category/video-panoptic-segmentation-models","pwc_aliases":[]}],"n_papers_tagged":1,"archive_num_papers":null,"papers_newest_first":[{"paper":"/paper/vip-deeplab-learning-visual-perception-with","title":"ViP-DeepLab: Learning Visual Perception with Depth-aware Video Panoptic Segmentation","date":"2020-12-09","arxiv_id":"2012.05258","n_code_links":2,"syntology":{"ran":3,"of":3,"unverified":0,"pointer_only":3}}],"papers_shown":1,"tasks":[{"task":"/task/depth-estimation","name":"Depth Estimation","papers":1},{"task":"/task/depth-aware-video-panoptic-segmentation","name":"Depth-aware Video Panoptic Segmentation","papers":1},{"task":"/task/monocular-depth-estimation","name":"Monocular Depth Estimation","papers":1},{"task":"/task/panoptic-segmentation","name":"Panoptic Segmentation","papers":1},{"task":"/task/segmentation","name":"Segmentation","papers":1},{"task":"/task/video-panoptic-segmentation","name":"Video Panoptic Segmentation","papers":1}],"tasks_shown":6,"n_tasks":6,"usage_by_year":[{"year":"2020","papers":1}],"row_source":"embedded","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/vip-deeplab"},"syntology_read_at":"2026-09-25T09:33:49+00:00"}