{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/learning-video-object-segmentation-from-2","title":"Learning Video Object Segmentation from Unlabeled Videos","arxiv_id":"2003.05020","date":"2020-03-10","proceeding":"CVPR 2020 6","authors":["Xiankai Lu","Wenguan Wang","Jianbing Shen","Yu-Wing Tai","David Crandall","Steven C. H. Hoi"],"abstract":"We propose a new method for video object segmentation (VOS) that addresses object pattern learning from unlabeled videos, unlike most existing methods which rely heavily on extensive annotated data. We introduce a unified unsupervised/weakly supervised learning framework, called MuG, that comprehensively captures intrinsic properties of VOS at multiple granularities. Our approach can help advance understanding of visual patterns in VOS and significantly reduce annotation burden. With a carefully-designed architecture and strong representation learning ability, our learned model can be applied to diverse VOS settings, including object-level zero-shot VOS, instance-level zero-shot VOS, and one-shot VOS. Experiments demonstrate promising performance in these settings, as well as the potential of MuG in leveraging unlabeled data to further improve the segmentation accuracy.","url_abs":"https://arxiv.org/abs/2003.05020v1","url_pdf":"https://arxiv.org/pdf/2003.05020v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"learning-video-object-segmentation-from-2","repo_url":"https://github.com/carrierlxk/MuG","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"object","task_name":"Object"},{"task_slug":"representation-learning","task_name":"Representation Learning"},{"task_slug":"segmentation","task_name":"Segmentation"},{"task_slug":"semantic-segmentation","task_name":"Semantic Segmentation"},{"task_slug":"semi-supervised-video-object-segmentation","task_name":"Semi-Supervised Video Object Segmentation"},{"task_slug":"unsupervised-video-object-segmentation","task_name":"Unsupervised Video Object Segmentation"},{"task_slug":"video-object-segmentation","task_name":"Video Object Segmentation"},{"task_slug":"video-semantic-segmentation","task_name":"Video Semantic Segmentation"},{"task_slug":"weakly-supervised-learning","task_name":"Weakly-supervised Learning"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/visual-object-tracking-on-davis-2016","task":"Semi-Supervised Video Object Segmentation","dataset":"DAVIS 2016","model":"MuG-W","rank_in_archive_order":75,"of":78,"metrics":{"F-measure (Decay)":"27.2","F-measure (Mean)":"63.6","F-measure (Recall)":"67.7","J&F":"64.65","Jaccard (Decay)":"26.4","Jaccard (Mean)":"65.7","Jaccard (Recall)":"77.7"},"uses_additional_data":false},{"leaderboard":"/sota/visual-object-tracking-on-davis-2017","task":"Semi-Supervised Video Object Segmentation","dataset":"DAVIS 2017 (val)","model":"MuG-W","rank_in_archive_order":77,"of":81,"metrics":{"F-measure (Decay)":"37.4","F-measure (Mean)":"58.0","F-measure (Recall)":"62.2","J&F":"56.05","Jaccard (Decay)":"32.5","Jaccard (Mean)":"54.1","Jaccard (Recall)":"60.5"},"uses_additional_data":false},{"leaderboard":"/sota/unsupervised-video-object-segmentation-on-5","task":"Unsupervised Video Object Segmentation","dataset":"DAVIS 2017 (test-dev)","model":"MuG-W","rank_in_archive_order":4,"of":6,"metrics":{"F-measure (Decay)":"-1.7","F-measure (Mean)":"44.5","F-measure (Recall)":"46.6","J&F":"41.7","Jaccard (Decay)":"-2.7","Jaccard (Mean)":"38.9","Jaccard (Recall)":"44.3"},"uses_additional_data":false}],"syntology":{"syntology_url":"https://syntology.ai/paper/2003.05020","atlas_url":"https://app.syntology.ai/?focus=2003.05020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.05020"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/carrierlxk/MuG","reach":null}],"summary":{"ran_honours":1},"by_repo_kind":{"official":{"samples":1,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":1,"samples":[{"code_sha256_prefix":"86753726be289c26","entry":"adjust_learning_rate","repo":"carrierlxk/MuG","repo_kind":"official","path":"MuG_GOT_global_new_residual.py","file_url":"https://github.com/carrierlxk/MuG/blob/HEAD/MuG_GOT_global_new_residual.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"86753726be289c26"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}