{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/video-object-segmentation-using-space-time","title":"Video Object Segmentation using Space-Time Memory Networks","arxiv_id":"1904.00607","date":"2019-04-01","proceeding":"ICCV 2019 10","authors":["Seoung Wug Oh","Joon-Young Lee","Ning Xu","Seon Joo Kim"],"abstract":"We propose a novel solution for semi-supervised video object segmentation. By the nature of the problem, available cues (e.g. video frame(s) with object masks) become richer with the intermediate predictions. However, the existing methods are unable to fully exploit this rich source of information. We resolve the issue by leveraging memory networks and learn to read relevant information from all available sources. In our framework, the past frames with object masks form an external memory, and the current frame as the query is segmented using the mask information in the memory. Specifically, the query and the memory are densely matched in the feature space, covering all the space-time pixel locations in a feed-forward fashion. Contrast to the previous approaches, the abundant use of the guidance information allows us to better handle the challenges such as appearance changes and occlussions. We validate our method on the latest benchmark sets and achieved the state-of-the-art performance (overall score of 79.4 on Youtube-VOS val set, J of 88.7 and 79.2 on DAVIS 2016/2017 val set respectively) while having a fast runtime (0.16 second/frame on DAVIS 2016 val set).","url_abs":"https://arxiv.org/abs/1904.00607v2","url_pdf":"https://arxiv.org/pdf/1904.00607v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"video-object-segmentation-using-space-time","repo_url":"https://github.com/hkchengrex/Mask-Propagation","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"video-object-segmentation-using-space-time","repo_url":"https://github.com/seoungwugoh/STM","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"video-object-segmentation-using-space-time","repo_url":"https://github.com/ywcmaike/TianchiVideoCharacterSegmentationPreliminary","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}}],"tasks":[{"task_slug":"interactive-video-object-segmentation","task_name":"Interactive Video Object Segmentation"},{"task_slug":"object","task_name":"Object"},{"task_slug":"one-shot-visual-object-segmentation","task_name":"One-shot visual object segmentation"},{"task_slug":"semantic-segmentation","task_name":"Semantic Segmentation"},{"task_slug":"semi-supervised-video-object-segmentation","task_name":"Semi-Supervised Video Object Segmentation"},{"task_slug":"video-object-segmentation","task_name":"Video Object Segmentation"},{"task_slug":"video-semantic-segmentation","task_name":"Video Semantic Segmentation"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/interactive-video-object-segmentation-on","task":"Interactive Video Object Segmentation","dataset":"DAVIS 2017","model":"STM","rank_in_archive_order":4,"of":7,"metrics":{"AUC-J&F":"0.803","J&F@60s":"0.848"},"uses_additional_data":true},{"leaderboard":"/sota/semi-supervised-video-object-segmentation-on-20","task":"Semi-Supervised Video Object Segmentation","dataset":"DAVIS (no YouTube-VOS training)","model":"STM","rank_in_archive_order":16,"of":26,"metrics":{"D16 val (F)":"88.1","D16 val (G)":"86.5","D16 val (J)":"84.8","D17 val (F)":"74.0","D17 val (G)":"71.6","D17 val (J)":"69.2","FPS":"6.25"},"uses_additional_data":false},{"leaderboard":"/sota/visual-object-tracking-on-davis-2016","task":"Semi-Supervised Video Object Segmentation","dataset":"DAVIS 2016","model":"STM","rank_in_archive_order":33,"of":78,"metrics":{"F-measure (Decay)":"4.2","F-measure (Mean)":"90.1","F-measure (Recall)":"95.2","J&F":"89.4","Jaccard (Decay)":"5.0","Jaccard (Mean)":"88.7","Jaccard (Recall)":"97.4"},"uses_additional_data":true},{"leaderboard":"/sota/semi-supervised-video-object-segmentation-on-1","task":"Semi-Supervised Video Object Segmentation","dataset":"DAVIS 2017 (test-dev)","model":"STM","rank_in_archive_order":38,"of":59,"metrics":{"F-measure (Decay)":"17.5","F-measure (Mean)":"75.2","F-measure (Recall)":"83.0","J&F":"72.2","Jaccard (Decay)":"16.9","Jaccard (Mean)":"69.3","Jaccard (Recall)":"78.0"},"uses_additional_data":true},{"leaderboard":"/sota/visual-object-tracking-on-davis-2017","task":"Semi-Supervised Video Object Segmentation","dataset":"DAVIS 2017 (val)","model":"STM","rank_in_archive_order":41,"of":81,"metrics":{"F-measure (Decay)":"10.5","F-measure (Mean)":"84.3","F-measure (Recall)":"91.8","J&F":"81.75","Jaccard (Decay)":"8.0","Jaccard (Mean)":"79.2","Jaccard (Recall)":"88.7"},"uses_additional_data":true},{"leaderboard":"/sota/video-object-segmentation-on-youtube-vos","task":"Semi-Supervised Video Object Segmentation","dataset":"YouTube-VOS 2018","model":"STM","rank_in_archive_order":46,"of":53,"metrics":{"Overall":"68.2"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1904.00607","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.00607"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/ywcmaike/TianchiVideoCharacterSegmentationPreliminary","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/seoungwugoh/STM","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/hkchengrex/Mask-Propagation","reach":null}],"summary":{"ran_fixture":1},"by_repo_kind":{"listed":{"samples":1,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"3663ac48a0c49097","entry":"comp_binary","repo":"hkchengrex/Mask-Propagation","repo_kind":"listed","path":"try_correspondence.py","file_url":"https://github.com/hkchengrex/Mask-Propagation/blob/HEAD/try_correspondence.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"3663ac48a0c49097"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}