{"url":"/method/vistr","slug":"vistr","name":"VisTR","full_name":"VisTR","full_name_withheld":false,"description_markdown":"**VisTR** is a [Transformer](https://paperswithcode.com/method/transformer) based video instance segmentation model. It views video instance segmentation as a direct end-to-end parallel sequence decoding/prediction problem. Given a video clip consisting of multiple image frames as input, VisTR outputs the sequence of masks for each instance in the video in order directly. At the core is a new, effective instance sequence matching and segmentation strategy, which supervises and segments instances at the sequence level as a whole. VisTR frames the instance segmentation and tracking in the same perspective of similarity learning, thus considerably simplifying the overall pipeline and is significantly different from existing approaches.","description_state":"present","introduced_year":null,"introduced_by":{"title":"End-to-End Video Instance Segmentation with Transformers","paper":"/paper/end-to-end-video-instance-segmentation-with","first_author":"Yuqing Wang","n_authors":7,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/end-to-end-video-instance-segmentation-with"},"source":{"url":"https://arxiv.org/abs/2011.14503v5","title":"End-to-End Video Instance Segmentation with Transformers","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Video Instance Segmentation Models","url":"/methods/category/video-instance-segmentation-models","pwc_aliases":[]},{"area":"Computer Vision","area_id":"computer-vision","collection":"Instance Segmentation Models","url":"/methods/category/instance-segmentation-models","pwc_aliases":[]}],"n_papers_tagged":3,"archive_num_papers":3,"papers_newest_first":[{"paper":"/paper/deformable-vistr-spatio-temporal-deformable","title":"Deformable VisTR: Spatio temporal deformable attention for video instance segmentation","date":"2022-03-12","arxiv_id":"2203.06318","n_code_links":1,"syntology":null},{"paper":null,"title":"Efficient Video Instance Segmentation via Tracklet Query and Proposal","date":"2022-03-03","arxiv_id":"2203.01853","n_code_links":0,"syntology":null},{"paper":"/paper/end-to-end-video-instance-segmentation-with","title":"End-to-End Video Instance Segmentation with Transformers","date":"2020-11-30","arxiv_id":"2011.14503","n_code_links":2,"syntology":null}],"papers_shown":3,"tasks":[{"task":"/task/instance-segmentation","name":"Instance Segmentation","papers":3},{"task":"/task/semantic-segmentation","name":"Semantic Segmentation","papers":3},{"task":"/task/video-instance-segmentation","name":"Video Instance Segmentation","papers":3},{"task":"/task/segmentation","name":"Segmentation","papers":2},{"task":null,"name":"GPU","papers":1},{"task":"/task/video-understanding","name":"Video Understanding","papers":1}],"tasks_shown":6,"n_tasks":6,"usage_by_year":[{"year":"2020","papers":1},{"year":"2022","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/vistr"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}