{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/context-aware-video-instance-segmentation","title":"Context-Aware Video Instance Segmentation","arxiv_id":"2407.03010","date":"2024-07-03","proceeding":null,"authors":["Seunghun Lee","Jiwan Seo","Kiljoon Han","Minwoo Choi","Sunghoon Im"],"abstract":"In this paper, we introduce the Context-Aware Video Instance Segmentation (CAVIS), a novel framework designed to enhance instance association by integrating contextual information adjacent to each object. To efficiently extract and leverage this information, we propose the Context-Aware Instance Tracker (CAIT), which merges contextual data surrounding the instances with the core instance features to improve tracking accuracy. Additionally, we introduce the Prototypical Cross-frame Contrastive (PCC) loss, which ensures consistency in object-level features across frames, thereby significantly enhancing instance matching accuracy. CAVIS demonstrates superior performance over state-of-the-art methods on all benchmark datasets in video instance segmentation (VIS) and video panoptic segmentation (VPS). Notably, our method excels on the OVIS dataset, which is known for its particularly challenging videos.","url_abs":"https://arxiv.org/abs/2407.03010v1","url_pdf":"https://arxiv.org/pdf/2407.03010v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"context-aware-video-instance-segmentation","repo_url":"https://github.com/Seung-Hun-Lee/CAVIS","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"instance-segmentation","task_name":"Instance Segmentation"},{"task_slug":"panoptic-segmentation","task_name":"Panoptic Segmentation"},{"task_slug":"segmentation","task_name":"Segmentation"},{"task_slug":"semantic-segmentation","task_name":"Semantic Segmentation"},{"task_slug":"video-instance-segmentation","task_name":"Video Instance Segmentation"},{"task_slug":"video-panoptic-segmentation","task_name":"Video Panoptic Segmentation"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/video-instance-segmentation-on-ovis-1","task":"Video Instance Segmentation","dataset":"OVIS validation","model":"CAVIS(VIT-L, Offline)","rank_in_archive_order":2,"of":44,"metrics":{"AP50":"82.6","AP75":"63.5","AR1":"21.2","AR10":"61.8","mask AP":"57.1"},"uses_additional_data":true},{"leaderboard":"/sota/video-instance-segmentation-on-youtube-vis-2","task":"Video Instance Segmentation","dataset":"YouTube-VIS 2021","model":"CAVIS(VIT-L, Offline)","rank_in_archive_order":1,"of":26,"metrics":{"AP50":"87.3","AP75":"73.2","AR1":"49.7","AR10":"70.3","mask AP":"65.3"},"uses_additional_data":true},{"leaderboard":"/sota/video-instance-segmentation-on-youtube-vis-1","task":"Video Instance Segmentation","dataset":"YouTube-VIS validation","model":"CAVIS(ViT-L, Online)","rank_in_archive_order":1,"of":44,"metrics":{"AP50":"89.3","AP75":"76.2","AR1":"58.3","AR10":"73.6","mask AP":"68.9"},"uses_additional_data":true},{"leaderboard":"/sota/video-instance-segmentation-on-youtube-vis-3","task":"Video Instance Segmentation","dataset":"Youtube-VIS 2022 Validation","model":"CAVIS (VIT-L)","rank_in_archive_order":2,"of":7,"metrics":{"mAP_L":"48.6"},"uses_additional_data":true},{"leaderboard":"/sota/video-panoptic-segmentation-on-vipseg","task":"Video Panoptic Segmentation","dataset":"VIPSeg","model":"CAVIS(VIT-L)","rank_in_archive_order":1,"of":12,"metrics":{"STQ":"56.1","VPQ":"58.5"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2407.03010","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}