{"url":"/sota/unsupervised-video-object-segmentation-on-12","task":{"name":"Unsupervised Video Object Segmentation","url":"/task/unsupervised-video-object-segmentation","note":null},"dataset":{"name":"YouTube-Objects","url":null},"category":"Computer Vision","categories":["Computer Vision"],"category_note":null,"description":"The unsupervised scenario assumes that the user does not interact with the algorithm to obtain the segmentation masks. Methods should provide a set of object candidates with no overlapping pixels that span through the whole video sequence. This set of objects should contain at least the objects that capture human attention when watching the whole video sequence i.e objects that are more likely to be followed by human gaze.","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["J"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"J":null}},"counts":{"rows":16,"rows_with_code":13,"rows_with_paper_page":16,"rows_dated":15,"rows_using_additional_data":0},"rows":[{"rank_in_archive_order":1,"model":"FakeFlow","metrics":{"J":"75.1"},"uses_additional_data":false,"paper_date":"2024-07-16","paper":"/paper/improving-unsupervised-video-object-1","paper_url":"https://arxiv.org/abs/2407.11714v1","paper_title":"Improving Unsupervised Video Object Segmentation via Fake Flow Generation","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":2,"model":"AMP","metrics":{"J":"75.0"},"uses_additional_data":false,"paper_date":"2023-03-18","paper":"/paper/adaptive-multi-source-predictor-for-zero-shot","paper_url":"https://arxiv.org/abs/2303.10383v2","paper_title":"Adaptive Multi-source Predictor for Zero-shot Video Object Segmentation","code":"https://github.com/xiaoqi-zhao-dlut/multi-source-aps-zvos","n_code_links":1,"syntology":null},{"rank_in_archive_order":3,"model":"DPA","metrics":{"J":"73.7"},"uses_additional_data":false,"paper_date":"2022-11-22","paper":"/paper/domain-alignment-and-temporal-aggregation-for","paper_url":"https://arxiv.org/abs/2211.12036v3","paper_title":"Dual Prototype Attention for Unsupervised Video Object Segmentation","code":"https://github.com/hydragon516/dpa","n_code_links":1,"syntology":null},{"rank_in_archive_order":4,"model":"TMO++ (MiT-b1, MS)","metrics":{"J":"73.5"},"uses_additional_data":false,"paper_date":"2023-09-26","paper":"/paper/treating-motion-as-option-with-output","paper_url":"https://arxiv.org/abs/2309.14786v1","paper_title":"Treating Motion as Option with Output Selection for Unsupervised Video Object Segmentation","code":"https://github.com/suhwan-cho/tmo","n_code_links":1,"syntology":null},{"rank_in_archive_order":5,"model":"TMO++ (RN-101)","metrics":{"J":"73.1"},"uses_additional_data":false,"paper_date":"2023-09-26","paper":"/paper/treating-motion-as-option-with-output","paper_url":"https://arxiv.org/abs/2309.14786v1","paper_title":"Treating Motion as Option with Output Selection for Unsupervised Video Object Segmentation","code":"https://github.com/suhwan-cho/tmo","n_code_links":1,"syntology":null},{"rank_in_archive_order":6,"model":"TMO++ (MiT-b1)","metrics":{"J":"73.0"},"uses_additional_data":false,"paper_date":"2023-09-26","paper":"/paper/treating-motion-as-option-with-output","paper_url":"https://arxiv.org/abs/2309.14786v1","paper_title":"Treating Motion as Option with Output Selection for Unsupervised Video Object Segmentation","code":"https://github.com/suhwan-cho/tmo","n_code_links":1,"syntology":null},{"rank_in_archive_order":7,"model":"TMO (RN-101)","metrics":{"J":"71.5"},"uses_additional_data":false,"paper_date":"2022-09-04","paper":"/paper/treating-motion-as-option-to-reduce-motion","paper_url":"https://arxiv.org/abs/2209.03138v5","paper_title":"Treating Motion as Option to Reduce Motion Dependency in Unsupervised Video Object Segmentation","code":"https://github.com/suhwan-cho/tmo","n_code_links":2,"syntology":{"n_ran":0,"n_unverified":4,"n_samples":4,"n_pointer_only_licence":0}},{"rank_in_archive_order":8,"model":"AMC-Net","metrics":{"J":"71.1"},"uses_additional_data":false,"paper_date":"2021-01-01","paper":"/paper/learning-motion-appearance-co-attention-for","paper_url":"http://openaccess.thecvf.com//content/ICCV2021/html/Yang_Learning_Motion-Appearance_Co-Attention_for_Zero-Shot_Video_Object_Segmentation_ICCV_2021_paper.html","paper_title":"Learning Motion-Appearance Co-Attention for Zero-Shot Video Object Segmentation","code":"https://github.com/isyangshu/amc-net","n_code_links":1,"syntology":null},{"rank_in_archive_order":9,"model":"TMO (MiT-b1)","metrics":{"J":"71.1"},"uses_additional_data":false,"paper_date":"2022-09-04","paper":"/paper/treating-motion-as-option-to-reduce-motion","paper_url":"https://arxiv.org/abs/2209.03138v5","paper_title":"Treating Motion as Option to Reduce Motion Dependency in Unsupervised Video Object Segmentation","code":"https://github.com/suhwan-cho/tmo","n_code_links":2,"syntology":{"n_ran":0,"n_unverified":4,"n_samples":4,"n_pointer_only_licence":0}},{"rank_in_archive_order":10,"model":"AGNN","metrics":{"J":"70.8"},"uses_additional_data":false,"paper_date":"2020-01-19","paper":"/paper/zero-shot-video-object-segmentation-via-1","paper_url":"https://arxiv.org/abs/2001.06807v1","paper_title":"Zero-Shot Video Object Segmentation via Attentive Graph Neural Networks","code":"https://github.com/carrierlxk/AGNN","n_code_links":1,"syntology":{"n_ran":4,"n_unverified":1,"n_samples":5,"n_pointer_only_licence":5}},{"rank_in_archive_order":11,"model":"COSNet","metrics":{"J":"70.5"},"uses_additional_data":false,"paper_date":"2020-01-19","paper":"/paper/see-more-know-more-unsupervised-video-object-1","paper_url":"https://arxiv.org/abs/2001.06810v1","paper_title":"See More, Know More: Unsupervised Video Object Segmentation with Co-Attention Siamese Networks","code":"https://github.com/carrierlxk/COSNet","n_code_links":1,"syntology":null},{"rank_in_archive_order":12,"model":"WCS-Net","metrics":{"J":"70.5"},"uses_additional_data":false,"paper_date":null,"paper":"/paper/unsupervised-video-object-segmentation-with-2","paper_url":"https://www.ecva.net/papers/eccv_2020/papers_ECCV/html/2189_ECCV_2020_paper.php","paper_title":"Unsupervised Video Object Segmentation with Joint Hotspot Tracking","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":13,"model":"RTNet","metrics":{"J":"70.1"},"uses_additional_data":false,"paper_date":"2021-06-19","paper":"/paper/reciprocal-transformations-for-unsupervised","paper_url":"http://openaccess.thecvf.com//content/CVPR2021/html/Ren_Reciprocal_Transformations_for_Unsupervised_Video_Object_Segmentation_CVPR_2021_paper.html","paper_title":"Reciprocal Transformations for Unsupervised Video Object Segmentation","code":"https://github.com/OliverRensu/RTNet","n_code_links":1,"syntology":null},{"rank_in_archive_order":14,"model":"AGS","metrics":{"J":"69.7"},"uses_additional_data":false,"paper_date":"2019-06-01","paper":"/paper/learning-unsupervised-video-object","paper_url":"http://openaccess.thecvf.com/content_CVPR_2019/html/Wang_Learning_Unsupervised_Video_Object_Segmentation_Through_Visual_Attention_CVPR_2019_paper.html","paper_title":"Learning Unsupervised Video Object Segmentation Through Visual Attention","code":"https://github.com/wenguanwang/AGS","n_code_links":1,"syntology":null},{"rank_in_archive_order":15,"model":"MATNet","metrics":{"J":"69.0"},"uses_additional_data":false,"paper_date":"2020-03-09","paper":"/paper/motion-attentive-transition-for-zero-shot","paper_url":"https://arxiv.org/abs/2003.04253v3","paper_title":"Motion-Attentive Transition for Zero-Shot Video Object Segmentation","code":"https://github.com/tfzhou/MATNet","n_code_links":1,"syntology":null},{"rank_in_archive_order":16,"model":"PDB","metrics":{"J":"65.5"},"uses_additional_data":false,"paper_date":"2018-09-01","paper":"/paper/pyramid-dilated-deeper-convlstm-for-video","paper_url":"http://openaccess.thecvf.com/content_ECCV_2018/html/Hongmei_Song_Pseudo_Pyramid_Deeper_ECCV_2018_paper.html","paper_title":"Pyramid Dilated Deeper ConvLSTM for Video Salient Object Detection","code":null,"n_code_links":0,"syntology":null}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":3,"rows_with_any_sample_ran":1,"distinct_papers_with_graph_line":2,"distinct_papers_with_any_sample_ran":1,"samples_over_distinct_papers":{"n_ran":4,"n_unverified":5,"n_samples":9,"n_pointer_only_licence":5,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":4,"n_unverified":9,"n_samples":13,"n_pointer_only_licence":5,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}