{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/tackling-background-distraction-in-video","title":"Tackling Background Distraction in Video Object Segmentation","arxiv_id":"2207.06953","date":"2022-07-14","proceeding":null,"authors":["Suhwan Cho","Heansung Lee","Minhyeok Lee","Chaewon Park","Sungjun Jang","Minjung Kim","Sangyoun Lee"],"abstract":"Semi-supervised video object segmentation (VOS) aims to densely track certain designated objects in videos. One of the main challenges in this task is the existence of background distractors that appear similar to the target objects. We propose three novel strategies to suppress such distractors: 1) a spatio-temporally diversified template construction scheme to obtain generalized properties of the target objects; 2) a learnable distance-scoring function to exclude spatially-distant distractors by exploiting the temporal consistency between two consecutive frames; 3) swap-and-attach augmentation to force each object to have unique features by providing training samples containing entangled objects. On all public benchmark datasets, our model achieves a comparable performance to contemporary state-of-the-art approaches, even with real-time performance. Qualitative results also demonstrate the superiority of our approach over existing methods. We believe our approach will be widely used for future VOS research.","url_abs":"https://arxiv.org/abs/2207.06953v3","url_pdf":"https://arxiv.org/pdf/2207.06953v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"tackling-background-distraction-in-video","repo_url":"https://github.com/suhwan-cho/tbd","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"object","task_name":"Object"},{"task_slug":"semantic-segmentation","task_name":"Semantic Segmentation"},{"task_slug":"semi-supervised-video-object-segmentation","task_name":"Semi-Supervised Video Object Segmentation"},{"task_slug":"video-object-segmentation","task_name":"Video Object Segmentation"},{"task_slug":"video-semantic-segmentation","task_name":"Video Semantic Segmentation"}],"methods":[{"method_slug":"vos","method_name":"VOS"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/semi-supervised-video-object-segmentation-on-20","task":"Semi-Supervised Video Object Segmentation","dataset":"DAVIS (no YouTube-VOS training)","model":"TBD","rank_in_archive_order":2,"of":26,"metrics":{"D16 val (F)":"86.2","D16 val (G)":"86.8","D16 val (J)":"87.5","D17 test (F)":"72.2","D17 test (G)":"69.4","D17 test (J)":"66.6","D17 val (F)":"82.3","D17 val (G)":"80.0","D17 val (J)":"77.6","FPS":"50.1"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2207.06953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.06953"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/suhwan-cho/tbd","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"deterministic:regex_extraction","url":"https://github.com/suhwan-cho/TBD","reach":{"status":"ok","spdx":"MIT"}}],"summary":{"ran_honours":1,"unverified":2},"by_repo_kind":{"official":{"samples":3,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"35390a54a05ee368","entry":"get_padding","repo":"suhwan-cho/tbd","repo_kind":"official","path":"tbd.py","file_url":"https://github.com/suhwan-cho/tbd/blob/HEAD/tbd.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"35390a54a05ee368"}},{"code_sha256_prefix":"5e122be6d47af828","entry":"aggregate_objects","repo":"suhwan-cho/tbd","repo_kind":"official","path":"tbd.py","file_url":"https://github.com/suhwan-cho/tbd/blob/HEAD/tbd.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"5e122be6d47af828"}},{"code_sha256_prefix":"d982bd5d3bffc1e5","entry":"attach_padding","repo":"suhwan-cho/tbd","repo_kind":"official","path":"tbd.py","file_url":"https://github.com/suhwan-cho/tbd/blob/HEAD/tbd.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"d982bd5d3bffc1e5"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}