{"url":"/task/unsupervised-video-object-segmentation","name":"Unsupervised Video Object Segmentation","slug":"unsupervised-video-object-segmentation","description_markdown":"The unsupervised scenario assumes that the user does not interact with the algorithm to obtain the segmentation masks. Methods should provide a set of object candidates with no overlapping pixels that span through the whole video sequence. This set of objects should contain at least the objects that capture human attention when watching the whole video sequence i.e objects that are more likely to be followed by human gaze.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":89,"papers_with_code":52,"benchmarks":6,"benchmark_tables_in_archive":6,"benchmark_tables_shown":6,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":8,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/unsupervised-video-object-segmentation-on-10","slug":"unsupervised-video-object-segmentation-on-10","dataset":"DAVIS 2016 val","dataset_url":"/dataset/davis-2016","rows_in_archive":25,"metrics":["G","J","F"],"first_row_in_archive_order":{"model":"GSANet","paper_title":"Guided Slot Attention for Unsupervised Video Object Segmentation","paper_url":"/paper/guided-slot-attention-for-unsupervised-video","paper_date":"2023-03-15","arxiv_id":"2303.08314","code_links":[{"title":"hydragon516/gsanet","url":"https://github.com/hydragon516/gsanet"}],"syntology":null}},{"leaderboard":"/sota/unsupervised-video-object-segmentation-on-12","slug":"unsupervised-video-object-segmentation-on-12","dataset":"YouTube-Objects","dataset_url":null,"rows_in_archive":16,"metrics":["J"],"first_row_in_archive_order":{"model":"FakeFlow","paper_title":"Improving Unsupervised Video Object Segmentation via Fake Flow Generation","paper_url":"/paper/improving-unsupervised-video-object-1","paper_date":"2024-07-16","arxiv_id":"2407.11714","code_links":[],"syntology":null}},{"leaderboard":"/sota/unsupervised-video-object-segmentation-on-11","slug":"unsupervised-video-object-segmentation-on-11","dataset":"FBMS test","dataset_url":null,"rows_in_archive":15,"metrics":["J"],"first_row_in_archive_order":{"model":"FakeFlow","paper_title":"Improving Unsupervised Video Object Segmentation via Fake Flow Generation","paper_url":"/paper/improving-unsupervised-video-object-1","paper_date":"2024-07-16","arxiv_id":"2407.11714","code_links":[],"syntology":null}},{"leaderboard":"/sota/unsupervised-video-object-segmentation-on-4","slug":"unsupervised-video-object-segmentation-on-4","dataset":"DAVIS 2017 (val)","dataset_url":"/dataset/davis-2017","rows_in_archive":10,"metrics":["J&F","Jaccard (Mean)","Jaccard (Recall)","F-measure (Mean)","F-measure (Recall)"],"first_row_in_archive_order":{"model":"DEVA (EntitySeg)","paper_title":"Tracking Anything with Decoupled Video Segmentation","paper_url":"/paper/tracking-anything-with-decoupled-video","paper_date":"2023-09-07","arxiv_id":"2309.03903","code_links":[{"title":"hkchengrex/Tracking-Anything-with-DEVA","url":"https://github.com/hkchengrex/Tracking-Anything-with-DEVA"}],"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":10}}},{"leaderboard":"/sota/unsupervised-video-object-segmentation-on-5","slug":"unsupervised-video-object-segmentation-on-5","dataset":"DAVIS 2017 (test-dev)","dataset_url":"/dataset/davis-2017","rows_in_archive":6,"metrics":["J&F","Jaccard (Mean)","Jaccard (Recall)","Jaccard (Decay)","F-measure (Mean)","F-measure (Recall)","F-measure (Decay)"],"first_row_in_archive_order":{"model":"DEVA (EntitySeg)","paper_title":"Tracking Anything with Decoupled Video Segmentation","paper_url":"/paper/tracking-anything-with-decoupled-video","paper_date":"2023-09-07","arxiv_id":"2309.03903","code_links":[{"title":"hkchengrex/Tracking-Anything-with-DEVA","url":"https://github.com/hkchengrex/Tracking-Anything-with-DEVA"}],"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":10}}},{"leaderboard":"/sota/unsupervised-video-object-segmentation-on-3","slug":"unsupervised-video-object-segmentation-on-3","dataset":"SegTrack v2","dataset_url":"/dataset/segtrack-v2-1","rows_in_archive":4,"metrics":["Mean IoU","Jaccard (Mean)"],"first_row_in_archive_order":{"model":"FrameSelect","paper_title":"Mask Selection and Propagation for Unsupervised Video Object Segmentation","paper_url":"/paper/mask-selection-and-propagation-for","paper_date":"2021-01-05","arxiv_id":null,"code_links":[{"title":"vidit98/FrameSelect","url":"https://github.com/vidit98/FrameSelect"}],"syntology":null}}],"datasets":[{"url":"/dataset/davis-2017","name":"DAVIS 2017","full_name":"DAVIS 2017","num_papers_in_archive":308},{"url":"/dataset/davis-2016","name":"DAVIS 2016","full_name":"DAVIS 2016","num_papers_in_archive":231},{"url":"/dataset/fbms","name":"FBMS","full_name":"Freiburg-Berkeley Motion Segmentation","num_papers_in_archive":126},{"url":"/dataset/segtrack-v2-1","name":"SegTrack-v2","full_name":"","num_papers_in_archive":107},{"url":"/dataset/referring-expressions-for-davis-2016-2017","name":"Referring Expressions for DAVIS 2016 & 2017","full_name":"","num_papers_in_archive":82},{"url":"/dataset/mose","name":"MOSE","full_name":"Complex Video Object Segmentation","num_papers_in_archive":49},{"url":"/dataset/fbms-59","name":"FBMS-59","full_name":"Freiburg-Berkeley Motion Segmentation","num_papers_in_archive":19},{"url":"/dataset/bl30k","name":"BL30K","full_name":"","num_papers_in_archive":11}],"subtasks":[],"parent_tasks":[{"url":"/task/video-object-segmentation","name":"Video Object Segmentation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":52,"tagged_in_all":89,"items":[{"url":"/paper/exploiting-geometric-constraints-on-dense","title":"EpO-Net: Exploiting Geometric Constraints on Dense Trajectories for Motion Saliency","date":"2019-09-29","arxiv_id":"1909.13258","repositories_listed":3,"syntology":null},{"url":"/paper/online-unsupervised-video-object-segmentation","title":"Online Unsupervised Video Object Segmentation via Contrastive Motion Clustering","date":"2023-06-21","arxiv_id":"2306.12048","repositories_listed":2,"syntology":null},{"url":"/paper/treating-motion-as-option-to-reduce-motion","title":"Treating Motion as Option to Reduce Motion Dependency in Unsupervised Video Object Segmentation","date":"2022-09-04","arxiv_id":"2209.03138","repositories_listed":2,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/d2conv3d-dynamic-dilated-convolutions-for","title":"D2Conv3D: Dynamic Dilated Convolutions for Object Segmentation in Videos","date":"2021-11-15","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/mast-a-memory-augmented-self-supervised","title":"MAST: A Memory-Augmented Self-supervised Tracker","date":"2020-02-18","arxiv_id":"2002.07793","repositories_listed":2,"syntology":null},{"url":"/paper/joint-task-self-supervised-learning-for","title":"Joint-task Self-supervised Learning for Temporal Correspondence","date":"2019-09-26","arxiv_id":"1909.11895","repositories_listed":2,"syntology":{"n":23,"n_ran":4,"n_unverified":19,"n_pointer_only":0}},{"url":"/paper/tukey-inspired-video-object-segmentation","title":"Tukey-Inspired Video Object Segmentation","date":"2018-11-19","arxiv_id":"1811.07958","repositories_listed":2,"syntology":null},{"url":"/paper/learning-motion-and-temporal-cues-for","title":"Learning Motion and Temporal Cues for Unsupervised Video Object Segmentation","date":"2025-01-14","arxiv_id":"2501.07806","repositories_listed":1,"syntology":null},{"url":"/paper/treating-motion-as-option-with-output","title":"Treating Motion as Option with Output Selection for Unsupervised Video Object Segmentation","date":"2023-09-26","arxiv_id":"2309.14786","repositories_listed":1,"syntology":null},{"url":"/paper/tracking-anything-with-decoupled-video","title":"Tracking Anything with Decoupled Video Segmentation","date":"2023-09-07","arxiv_id":"2309.03903","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":10}},{"url":"/paper/uvosam-a-mask-free-paradigm-for-unsupervised","title":"UVOSAM: A Mask-free Paradigm for Unsupervised Video Object Segmentation via Segment Anything Model","date":"2023-05-22","arxiv_id":"2305.12659","repositories_listed":1,"syntology":null},{"url":"/paper/bootstrapping-objectness-from-videos-by","title":"Bootstrapping Objectness from Videos by Relaxed Common Fate and Visual Grouping","date":"2023-04-17","arxiv_id":"2304.08025","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-multi-source-predictor-for-zero-shot","title":"Adaptive Multi-source Predictor for Zero-shot Video Object Segmentation","date":"2023-03-18","arxiv_id":"2303.10383","repositories_listed":1,"syntology":null},{"url":"/paper/guided-slot-attention-for-unsupervised-video","title":"Guided Slot Attention for Unsupervised Video Object Segmentation","date":"2023-03-15","arxiv_id":"2303.08314","repositories_listed":1,"syntology":null},{"url":"/paper/domain-alignment-and-temporal-aggregation-for","title":"Dual Prototype Attention for Unsupervised Video Object Segmentation","date":"2022-11-22","arxiv_id":"2211.12036","repositories_listed":1,"syntology":null},{"url":"/paper/a-simple-and-powerful-global-optimization-for","title":"A Simple and Powerful Global Optimization for Unsupervised Video Object Segmentation","date":"2022-09-19","arxiv_id":"2209.09341","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/unsupervised-video-object-segmentation-via","title":"Unsupervised Video Object Segmentation via Prototype Memory Network","date":"2022-09-08","arxiv_id":"2209.03712","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-feature-alignment-network-for","title":"Hierarchical Feature Alignment Network for Unsupervised Video Object Segmentation","date":"2022-07-18","arxiv_id":"2207.08485","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-motion-compensated-network-for","title":"Implicit Motion-Compensated Network for Unsupervised Video Object Segmentation","date":"2022-04-06","arxiv_id":"2204.02791","repositories_listed":1,"syntology":null},{"url":"/paper/in-n-out-generative-learning-for-dense","title":"In-N-Out Generative Learning for Dense Unsupervised Video Segmentation","date":"2022-03-29","arxiv_id":"2203.15312","repositories_listed":1,"syntology":null},{"url":"/paper/autoencoder-based-background-reconstruction","title":"Autoencoder-based background reconstruction and foreground segmentation with background noise estimation","date":"2021-12-15","arxiv_id":"2112.08001","repositories_listed":1,"syntology":null},{"url":"/paper/d-2conv3d-dynamic-dilated-convolutions-for","title":"D^2Conv3D: Dynamic Dilated Convolutions for Object Segmentation in Videos","date":"2021-11-15","arxiv_id":"2111.07774","repositories_listed":1,"syntology":null},{"url":"/paper/dense-unsupervised-learning-for-video","title":"Dense Unsupervised Learning for Video Segmentation","date":"2021-11-11","arxiv_id":"2111.06265","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/multi-source-fusion-and-automatic-predictor","title":"Multi-Source Fusion and Automatic Predictor Selection for Zero-Shot Video Object Segmentation","date":"2021-08-11","arxiv_id":"2108.05076","repositories_listed":1,"syntology":null},{"url":"/paper/full-duplex-strategy-for-video-object","title":"Full-Duplex Strategy for Video Object Segmentation","date":"2021-08-06","arxiv_id":"2108.03151","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/reciprocal-transformations-for-unsupervised","title":"Reciprocal Transformations for Unsupervised Video Object Segmentation","date":"2021-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/video-instance-segmentation-with-a-propose","title":"Video Instance Segmentation with a Propose-Reduce Paradigm","date":"2021-03-25","arxiv_id":"2103.13746","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":5}},{"url":"/paper/mask-selection-and-propagation-for","title":"Mask Selection and Propagation for Unsupervised Video Object Segmentation","date":"2021-01-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-motion-appearance-co-attention-for","title":"Learning Motion-Appearance Co-Attention for Zero-Shot Video Object Segmentation","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/making-a-case-for-3d-convolutions-for-object","title":"Making a Case for 3D Convolutions for Object Segmentation in Videos","date":"2020-08-26","arxiv_id":"2008.11516","repositories_listed":1,"syntology":{"n":9,"n_ran":3,"n_unverified":6,"n_pointer_only":0}}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}