{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/video-object-segmentation-with-adaptive","title":"Video Object Segmentation with Adaptive Feature Bank and Uncertain-Region Refinement","arxiv_id":"2010.07958","date":"2020-10-15","proceeding":"NeurIPS 2020 12","authors":["Yongqing Liang","Xin Li","Navid Jafari","Qin Chen"],"abstract":"We propose a new matching-based framework for semi-supervised video object segmentation (VOS). Recently, state-of-the-art VOS performance has been achieved by matching-based algorithms, in which feature banks are created to store features for region matching and classification. However, how to effectively organize information in the continuously growing feature bank remains under-explored, and this leads to inefficient design of the bank. We introduce an adaptive feature bank update scheme to dynamically absorb new features and discard obsolete features. We also design a new confidence loss and a fine-grained segmentation module to enhance the segmentation accuracy in uncertain regions. On public benchmarks, our algorithm outperforms existing state-of-the-arts.","url_abs":"https://arxiv.org/abs/2010.07958v1","url_pdf":"https://arxiv.org/pdf/2010.07958v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"video-object-segmentation-with-adaptive","repo_url":"https://github.com/xmlyqing00/AFB-URR","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}}],"tasks":[{"task_slug":"segmentation","task_name":"Segmentation"},{"task_slug":"semantic-segmentation","task_name":"Semantic Segmentation"},{"task_slug":"semi-supervised-video-object-segmentation","task_name":"Semi-Supervised Video Object Segmentation"},{"task_slug":"video-object-segmentation","task_name":"Video Object Segmentation"},{"task_slug":"video-semantic-segmentation","task_name":"Video Semantic Segmentation"}],"methods":[{"method_slug":"vos","method_name":"VOS"}],"datasets_introduced":[{"slug":"long-video-dataset","name":"Long Video Dataset","full_name":""},{"slug":"long-video-dataset-3x","name":"Long Video Dataset (3X)","full_name":""}],"methods_introduced":[],"results":[{"leaderboard":"/sota/semi-supervised-video-object-segmentation-on-20","task":"Semi-Supervised Video Object Segmentation","dataset":"DAVIS (no YouTube-VOS training)","model":"AFB-URR","rank_in_archive_order":12,"of":26,"metrics":{"D17 val (F)":"76.1","D17 val (G)":"74.6","D17 val (J)":"73.0","FPS":"4.00"},"uses_additional_data":false},{"leaderboard":"/sota/visual-object-tracking-on-davis-2017","task":"Semi-Supervised Video Object Segmentation","dataset":"DAVIS 2017 (val)","model":"AFB-URR","rank_in_archive_order":56,"of":81,"metrics":{"F-measure (Decay)":"15.5","F-measure (Mean)":"76.1","F-measure (Recall)":"87.0","J&F":"74.6","Jaccard (Decay)":"13.8","Jaccard (Mean)":"73.0","Jaccard (Recall)":"85.3"},"uses_additional_data":false},{"leaderboard":"/sota/semi-supervised-video-object-segmentation-on-13","task":"Semi-Supervised Video Object Segmentation","dataset":"Long Video Dataset","model":"AFB-URR","rank_in_archive_order":6,"of":9,"metrics":{"F":"84.5","J":"82.9","J&F":"83.7"},"uses_additional_data":true},{"leaderboard":"/sota/semi-supervised-video-object-segmentation-on-14","task":"Semi-Supervised Video Object Segmentation","dataset":"Long Video Dataset (3X)","model":"AFB-URR","rank_in_archive_order":2,"of":2,"metrics":{"F":"84.6","J":"82.9","J&F":"83.8"},"uses_additional_data":true},{"leaderboard":"/sota/video-object-segmentation-on-youtube-vos","task":"Semi-Supervised Video Object Segmentation","dataset":"YouTube-VOS 2018","model":"AFB-URR","rank_in_archive_order":41,"of":53,"metrics":{"F-Measure (Seen)":"83.1","F-Measure (Unseen)":"82.6","Jaccard (Seen)":"78.8","Jaccard (Unseen)":"74.1","Overall":"79.6"},"uses_additional_data":true}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2010.07958","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}