{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/efficient-video-object-segmentation-via","title":"Efficient Video Object Segmentation via Network Modulation","arxiv_id":"1802.01218","date":"2018-02-04","proceeding":"CVPR 2018 6","authors":["Linjie Yang","Yanran Wang","Xuehan Xiong","Jianchao Yang","Aggelos K. Katsaggelos"],"abstract":"Video object segmentation targets at segmenting a specific object throughout\na video sequence, given only an annotated first frame. Recent deep learning\nbased approaches find it effective by fine-tuning a general-purpose\nsegmentation model on the annotated frame using hundreds of iterations of\ngradient descent. Despite the high accuracy these methods achieve, the\nfine-tuning process is inefficient and fail to meet the requirements of real\nworld applications. We propose a novel approach that uses a single forward pass\nto adapt the segmentation model to the appearance of a specific object.\nSpecifically, a second meta neural network named modulator is learned to\nmanipulate the intermediate layers of the segmentation network given limited\nvisual and spatial information of the target object. The experiments show that\nour approach is 70times faster than fine-tuning approaches while achieving\nsimilar accuracy.","url_abs":"http://arxiv.org/abs/1802.01218v1","url_pdf":"http://arxiv.org/pdf/1802.01218v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"efficient-video-object-segmentation-via","repo_url":"https://github.com/linjieyangsc/video_seg","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"tf","reach":{"status":"unanswered"}}],"tasks":[{"task_slug":"object","task_name":"Object"},{"task_slug":"segmentation","task_name":"Segmentation"},{"task_slug":"semantic-segmentation","task_name":"Semantic Segmentation"},{"task_slug":"semi-supervised-video-object-segmentation","task_name":"Semi-Supervised Video Object Segmentation"},{"task_slug":"video-instance-segmentation","task_name":"Video Instance Segmentation"},{"task_slug":"video-object-segmentation","task_name":"Video Object Segmentation"},{"task_slug":"video-semantic-segmentation","task_name":"Video Semantic Segmentation"},{"task_slug":"visual-object-tracking","task_name":"Visual Object Tracking"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/one-shot-visual-object-segmentation-on","task":"One-shot visual object segmentation","dataset":"YouTube-VOS 2018","model":"OSMN","rank_in_archive_order":2,"of":2,"metrics":{"Jaccard (Seen)":"60.0"},"uses_additional_data":false},{"leaderboard":"/sota/visual-object-tracking-on-davis-2016","task":"Semi-Supervised Video Object Segmentation","dataset":"DAVIS 2016","model":"OSMN","rank_in_archive_order":68,"of":78,"metrics":{"F-measure (Decay)":"10.6","F-measure (Mean)":"72.9","F-measure (Recall)":"84.0","J&F":"73.45","Jaccard (Decay)":"9.0","Jaccard (Mean)":"74.0","Jaccard (Recall)":"87.6"},"uses_additional_data":false},{"leaderboard":"/sota/semi-supervised-video-object-segmentation-on-1","task":"Semi-Supervised Video Object Segmentation","dataset":"DAVIS 2017 (test-dev)","model":"OSMN","rank_in_archive_order":59,"of":59,"metrics":{"F-measure (Decay)":"17.4","F-measure (Recall)":"47.4","J&F":"41.3","Jaccard (Decay)":"19.0","Jaccard (Mean)":"37.7","Jaccard (Recall)":"38.9"},"uses_additional_data":false},{"leaderboard":"/sota/visual-object-tracking-on-davis-2017","task":"Semi-Supervised Video Object Segmentation","dataset":"DAVIS 2017 (val)","model":"OSMN","rank_in_archive_order":78,"of":81,"metrics":{"F-measure (Decay)":"24.3","F-measure (Mean)":"57.1","F-measure (Recall)":"66.1","J&F":"54.8","Jaccard (Decay)":"21.5","Jaccard (Mean)":"52.5","Jaccard (Recall)":"60.9"},"uses_additional_data":false},{"leaderboard":"/sota/video-object-segmentation-on-youtube-vos","task":"Semi-Supervised Video Object Segmentation","dataset":"YouTube-VOS 2018","model":"OSMN","rank_in_archive_order":52,"of":53,"metrics":{"F-Measure (Seen)":"60.1","F-Measure (Unseen)":"44.0","Jaccard (Seen)":"60.0","Jaccard (Unseen)":"40.6","Overall":"51.2","Speed  (FPS)":"7.14"},"uses_additional_data":false},{"leaderboard":"/sota/video-instance-segmentation-on-youtube-vis-1","task":"Video Instance Segmentation","dataset":"YouTube-VIS validation","model":"OSMN","rank_in_archive_order":43,"of":44,"metrics":{"AP50":"28.6","AP75":"33.1","mask AP":"29.1"},"uses_additional_data":false},{"leaderboard":"/sota/visual-object-tracking-on-youtube-vos","task":"Visual Object Tracking","dataset":"YouTube-VOS 2018","model":"OSMN","rank_in_archive_order":4,"of":9,"metrics":{"F-Measure (Seen)":"60.1","F-Measure (Unseen)":"44.0","Jaccard (Seen)":"60.0","O (Average of Measures)":"51.2"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1802.01218","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}