{"url":"/task/video-object-segmentation","name":"Video Object Segmentation","slug":"video-object-segmentation","description_markdown":"Video object segmentation is a binary labeling problem aiming to separate foreground object(s) from the background region of a video.\r\n\r\nFor leaderboards please refer to the different subtasks.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":551,"papers_with_code":294,"benchmarks":13,"benchmark_tables_in_archive":13,"benchmark_tables_shown":13,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":19,"subtasks":7,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/video-object-segmentation-on-davis-2016","slug":"video-object-segmentation-on-davis-2016","dataset":"DAVIS 2016","dataset_url":"/dataset/davis-2016","rows_in_archive":24,"metrics":["J&F","F-Score","Jaccard (Mean)","mIoU","Contour Accuracy"],"first_row_in_archive_order":{"model":"ISVOS (BL30K, MS)","paper_title":"Look Before You Match: Instance Understanding Matters in Video Object Segmentation","paper_url":"/paper/look-before-you-match-instance-understanding","paper_date":"2022-12-13","arxiv_id":"2212.06826","code_links":[],"syntology":null}},{"leaderboard":"/sota/video-object-segmentation-on-davis-2017-val","slug":"video-object-segmentation-on-davis-2017-val","dataset":"DAVIS 2017 (val)","dataset_url":"/dataset/davis-2017","rows_in_archive":17,"metrics":["Mean Jaccard & F-Measure","F-measure","Jaccard"],"first_row_in_archive_order":{"model":"XMem (BLK30K, MS)","paper_title":"XMem: Long-Term Video Object Segmentation with an Atkinson-Shiffrin Memory Model","paper_url":"/paper/xmem-long-term-video-object-segmentation-with","paper_date":"2022-07-14","arxiv_id":"2207.07115","code_links":[{"title":"hkchengrex/XMem","url":"https://github.com/hkchengrex/XMem"},{"title":"tianyuan168326/videosemanticcompression-pytorch","url":"https://github.com/tianyuan168326/videosemanticcompression-pytorch"}],"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":2}}},{"leaderboard":"/sota/video-object-segmentation-on-youtube-vos-1","slug":"video-object-segmentation-on-youtube-vos-1","dataset":"YouTube-VOS 2018","dataset_url":"/dataset/youtube-vos","rows_in_archive":17,"metrics":["Mean Jaccard & F-Measure","Jaccard (Seen)","Jaccard (Unseen)","F-Measure (Seen)","F-Measure (Unseen)"],"first_row_in_archive_order":{"model":"XMem (BL30K, MS)","paper_title":"XMem: Long-Term Video Object Segmentation with an Atkinson-Shiffrin Memory Model","paper_url":"/paper/xmem-long-term-video-object-segmentation-with","paper_date":"2022-07-14","arxiv_id":"2207.07115","code_links":[{"title":"hkchengrex/XMem","url":"https://github.com/hkchengrex/XMem"},{"title":"tianyuan168326/videosemanticcompression-pytorch","url":"https://github.com/tianyuan168326/videosemanticcompression-pytorch"}],"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":2}}},{"leaderboard":"/sota/video-object-segmentation-on-davis-2017-test-1","slug":"video-object-segmentation-on-davis-2017-test-1","dataset":"DAVIS 2017 (test-dev)","dataset_url":"/dataset/davis-2017","rows_in_archive":10,"metrics":["Jaccard","F-measure","Mean Jaccard & F-Measure"],"first_row_in_archive_order":{"model":"BATMAN","paper_title":"BATMAN: Bilateral Attention Transformer in Motion-Appearance Neighboring Space for Video Object Segmentation","paper_url":"/paper/batman-bilateral-attention-transformer-in","paper_date":"2022-08-01","arxiv_id":"2208.01159","code_links":[],"syntology":null}},{"leaderboard":"/sota/video-object-segmentation-on-youtube-vos-2019-2","slug":"video-object-segmentation-on-youtube-vos-2019-2","dataset":"YouTube-VOS 2019","dataset_url":"/dataset/youtube-vos","rows_in_archive":10,"metrics":["Mean Jaccard & F-Measure","Jaccard (Seen)","Jaccard (Unseen)","F-Measure (Seen)","F-Measure (Unseen)"],"first_row_in_archive_order":{"model":"XMem (BL30K,MS)","paper_title":"XMem: Long-Term Video Object Segmentation with an Atkinson-Shiffrin Memory Model","paper_url":"/paper/xmem-long-term-video-object-segmentation-with","paper_date":"2022-07-14","arxiv_id":"2207.07115","code_links":[{"title":"hkchengrex/XMem","url":"https://github.com/hkchengrex/XMem"},{"title":"tianyuan168326/videosemanticcompression-pytorch","url":"https://github.com/tianyuan168326/videosemanticcompression-pytorch"}],"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":2}}},{"leaderboard":"/sota/video-object-segmentation-on-davis-2017","slug":"video-object-segmentation-on-davis-2017","dataset":"DAVIS 2017","dataset_url":"/dataset/davis-2017","rows_in_archive":5,"metrics":["Jaccard (Mean)","mIoU","J&F","F-Score"],"first_row_in_archive_order":{"model":"AOC-MF (val)","paper_title":"Towards Robust Video Object Segmentation with Adaptive Object Calibration","paper_url":"/paper/towards-robust-video-object-segmentation-with","paper_date":"2022-07-02","arxiv_id":"2207.00887","code_links":[{"title":"jerryx1110/robust-video-object-segmentation","url":"https://github.com/jerryx1110/robust-video-object-segmentation"}],"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":4}}},{"leaderboard":"/sota/video-object-segmentation-on-m-3-vos","slug":"video-object-segmentation-on-m-3-vos","dataset":"M$^3$-VOS","dataset_url":"/dataset/m-3-vos","rows_in_archive":4,"metrics":["Average IOU"],"first_row_in_archive_order":{"model":"ReVOS","paper_title":"M^3-VOS: Multi-Phase, Multi-Transition, and Multi-Scenery Video Object Segmentation","paper_url":"/paper/m-3-vos-multi-phase-multi-transition-and-1","paper_date":"2025-06-15","arxiv_id":null,"code_links":[{"title":"zixuan-chen/M3VOS_Experiment","url":"https://github.com/zixuan-chen/M3VOS_Experiment"}],"syntology":null}},{"leaderboard":"/sota/video-object-segmentation-on-davis-2017-test","slug":"video-object-segmentation-on-davis-2017-test","dataset":"DAVIS-2017 (test-dev)","dataset_url":null,"rows_in_archive":2,"metrics":["Mean Jaccard & F-Measure","Jaccard","F-measure"],"first_row_in_archive_order":{"model":"XMem (BL30K, MS)","paper_title":"XMem: Long-Term Video Object Segmentation with an Atkinson-Shiffrin Memory Model","paper_url":"/paper/xmem-long-term-video-object-segmentation-with","paper_date":"2022-07-14","arxiv_id":"2207.07115","code_links":[{"title":"hkchengrex/XMem","url":"https://github.com/hkchengrex/XMem"},{"title":"tianyuan168326/videosemanticcompression-pytorch","url":"https://github.com/tianyuan168326/videosemanticcompression-pytorch"}],"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":2}}},{"leaderboard":"/sota/video-object-segmentation-on-fbms","slug":"video-object-segmentation-on-fbms","dataset":"FBMS","dataset_url":"/dataset/fbms","rows_in_archive":2,"metrics":["F-Score","Jaccard (Mean)"],"first_row_in_archive_order":{"model":"DFNet","paper_title":"Learning Discriminative Feature with CRF for Unsupervised Video Object Segmentation","paper_url":"/paper/learning-discriminative-feature-with-crf-for","paper_date":"2020-08-04","arxiv_id":"2008.01270","code_links":[],"syntology":null}},{"leaderboard":"/sota/video-object-segmentation-on-youtube-1","slug":"video-object-segmentation-on-youtube-1","dataset":"YouTube","dataset_url":null,"rows_in_archive":2,"metrics":["Average","mIoU"],"first_row_in_archive_order":{"model":"Ours","paper_title":"Multi-Source Fusion and Automatic Predictor Selection for Zero-Shot Video Object Segmentation","paper_url":"/paper/multi-source-fusion-and-automatic-predictor","paper_date":"2021-08-11","arxiv_id":"2108.05076","code_links":[{"title":"xiaoqi-zhao-dlut/multi-source-aps-zvos","url":"https://github.com/xiaoqi-zhao-dlut/multi-source-aps-zvos"}],"syntology":null}},{"leaderboard":"/sota/video-object-segmentation-on-fbms-59","slug":"video-object-segmentation-on-fbms-59","dataset":"FBMS-59","dataset_url":"/dataset/fbms-59","rows_in_archive":1,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"LOCATE","paper_title":"LOCATE: Self-supervised Object Discovery via Flow-guided Graph-cut and Bootstrapped Self-training","paper_url":"/paper/locate-self-supervised-object-discovery-via","paper_date":"2023-08-22","arxiv_id":"2308.11239","code_links":[{"title":"silky1708/locate","url":"https://github.com/silky1708/locate"}],"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":6}}},{"leaderboard":"/sota/video-object-segmentation-on-mose","slug":"video-object-segmentation-on-mose","dataset":"MOSE","dataset_url":"/dataset/mose","rows_in_archive":1,"metrics":["J&F"],"first_row_in_archive_order":{"model":"Cutie","paper_title":"Putting the Object Back into Video Object Segmentation","paper_url":"/paper/putting-the-object-back-into-video-object","paper_date":"2023-10-19","arxiv_id":"2310.12982","code_links":[{"title":"hkchengrex/Cutie","url":"https://github.com/hkchengrex/Cutie"}],"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/video-object-segmentation-on-segtrack-v2-1","slug":"video-object-segmentation-on-segtrack-v2-1","dataset":"SegTrack-v2","dataset_url":"/dataset/segtrack-v2-1","rows_in_archive":1,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"LOCATE","paper_title":"LOCATE: Self-supervised Object Discovery via Flow-guided Graph-cut and Bootstrapped Self-training","paper_url":"/paper/locate-self-supervised-object-discovery-via","paper_date":"2023-08-22","arxiv_id":"2308.11239","code_links":[{"title":"silky1708/locate","url":"https://github.com/silky1708/locate"}],"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":6}}}],"datasets":[{"url":"/dataset/davis","name":"DAVIS","full_name":"Densely Annotated VIdeo Segmentation","num_papers_in_archive":734},{"url":"/dataset/davis-2017","name":"DAVIS 2017","full_name":"DAVIS 2017","num_papers_in_archive":308},{"url":"/dataset/davis-2016","name":"DAVIS 2016","full_name":"DAVIS 2016","num_papers_in_archive":231},{"url":"/dataset/youtube-vos","name":"YouTube-VOS 2018","full_name":"Youtube Video Object Segmentation","num_papers_in_archive":203},{"url":"/dataset/fbms","name":"FBMS","full_name":"Freiburg-Berkeley Motion Segmentation","num_papers_in_archive":126},{"url":"/dataset/segtrack-v2-1","name":"SegTrack-v2","full_name":"","num_papers_in_archive":107},{"url":"/dataset/referring-expressions-for-davis-2016-2017","name":"Referring Expressions for DAVIS 2016 & 2017","full_name":"","num_papers_in_archive":82},{"url":"/dataset/mose","name":"MOSE","full_name":"Complex Video Object Segmentation","num_papers_in_archive":49},{"url":"/dataset/avsbench","name":"AVSBench","full_name":"Audio −Visual Segmentation","num_papers_in_archive":20},{"url":"/dataset/fbms-59","name":"FBMS-59","full_name":"Freiburg-Berkeley Motion Segmentation","num_papers_in_archive":19},{"url":"/dataset/lvos","name":"LVOS","full_name":"","num_papers_in_archive":19},{"url":"/dataset/bl30k","name":"BL30K","full_name":"","num_papers_in_archive":11},{"url":"/dataset/vost","name":"VOST","full_name":"","num_papers_in_archive":11},{"url":"/dataset/armbench","name":"ARMBench","full_name":"","num_papers_in_archive":4},{"url":"/dataset/m-3-vos","name":"M$^3$-VOS","full_name":"M$^3$-VOS: Multi-Phase, Multi-Transition, and Multi-Scenery Video Object Segmentation","num_papers_in_archive":4},{"url":"/dataset/odms","name":"ODMS","full_name":"Object Depth via Motion and Segmentation","num_papers_in_archive":2},{"url":"/dataset/pumavos","name":"PUMaVOS","full_name":"Partial and Unusual Masks for Video Object Segmentation","num_papers_in_archive":2},{"url":"/dataset/visor","name":"VISOR - Semi supervised video object segmentation","full_name":"val","num_papers_in_archive":1},{"url":"/dataset/infinity-spills-basic-dataset","name":"Infinity Spills Basic Dataset","full_name":"Infinity Spills Basic Dataset","num_papers_in_archive":0}],"subtasks":[{"url":"/task/interactive-video-object-segmentation","name":"Interactive Video Object Segmentation"},{"url":"/task/long-tail-video-object-segmentation","name":"Long-tail Video Object Segmentation"},{"url":"/task/referring-video-object-segmentation","name":"Referring Video Object Segmentation"},{"url":"/task/semi-supervised-video-object-segmentation","name":"Semi-Supervised Video Object Segmentation"},{"url":"/task/unsupervised-video-object-segmentation","name":"Unsupervised Video Object Segmentation"},{"url":"/task/video-salient-object-detection","name":"Video Salient Object Detection"},{"url":"/task/video-shadow-detection","name":"Video Shadow Detection"}],"parent_tasks":[{"url":"/task/video","name":"Video"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":294,"tagged_in_all":551,"items":[{"url":"/paper/emerging-properties-in-self-supervised-vision","title":"Emerging Properties in Self-Supervised Vision Transformers","date":"2021-04-29","arxiv_id":"2104.14294","repositories_listed":32,"syntology":{"n":20,"n_ran":5,"n_unverified":15,"n_pointer_only":2}},{"url":"/paper/2408-00714","title":"SAM 2: Segment Anything in Images and Videos","date":"2024-08-01","arxiv_id":"2408.00714","repositories_listed":11,"syntology":{"n":49,"n_ran":28,"n_unverified":21,"n_pointer_only":0}},{"url":"/paper/one-shot-video-object-segmentation","title":"One-Shot Video Object Segmentation","date":"2016-11-16","arxiv_id":"1611.05198","repositories_listed":8,"syntology":null},{"url":"/paper/rethinking-self-supervised-correspondence","title":"Rethinking Self-supervised Correspondence Learning: A Video Frame-level Similarity Perspective","date":"2021-03-31","arxiv_id":"2103.17263","repositories_listed":5,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/modular-interactive-video-object-segmentation","title":"Modular Interactive Video Object Segmentation: Interaction-to-Mask, Propagation and Difference-Aware Fusion","date":"2021-03-14","arxiv_id":"2103.07941","repositories_listed":5,"syntology":{"n":18,"n_ran":9,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/premvos-proposal-generation-refinement-and","title":"PReMVOS: Proposal-generation, Refinement and Merging for Video Object Segmentation","date":"2018-07-24","arxiv_id":"1807.09190","repositories_listed":5,"syntology":{"n":15,"n_ran":3,"n_unverified":12,"n_pointer_only":3}},{"url":"/paper/video-polyp-segmentation-a-deep-learning","title":"Video Polyp Segmentation: A Deep Learning Perspective","date":"2022-03-27","arxiv_id":"2203.14291","repositories_listed":4,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/make-one-shot-video-object-segmentation-1","title":"Make One-Shot Video Object Segmentation Efficient Again","date":"2020-12-03","arxiv_id":"2012.01866","repositories_listed":4,"syntology":null},{"url":"/paper/interactive-video-object-segmentation-using","title":"Interactive Video Object Segmentation Using Global and Local Transfer Modules","date":"2020-07-16","arxiv_id":"2007.08139","repositories_listed":4,"syntology":{"n":15,"n_ran":0,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/youtube-vos-sequence-to-sequence-video-object","title":"YouTube-VOS: Sequence-to-Sequence Video Object Segmentation","date":"2018-09-03","arxiv_id":"1809.00461","repositories_listed":4,"syntology":null},{"url":"/paper/lucid-data-dreaming-for-video-object","title":"Lucid Data Dreaming for Video Object Segmentation","date":"2017-03-28","arxiv_id":"1703.09554","repositories_listed":4,"syntology":null},{"url":"/paper/segment-everything-everywhere-all-at-once","title":"Segment Everything Everywhere All at Once","date":"2023-04-13","arxiv_id":"2304.06718","repositories_listed":3,"syntology":null},{"url":"/paper/seggpt-segmenting-everything-in-context","title":"SegGPT: Segmenting Everything In Context","date":"2023-04-06","arxiv_id":"2304.03284","repositories_listed":3,"syntology":{"n":10,"n_ran":6,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/epic-kitchens-visor-benchmark-video","title":"EPIC-KITCHENS VISOR Benchmark: VIdeo Segmentations and Object Relations","date":"2022-09-26","arxiv_id":"2209.13064","repositories_listed":3,"syntology":{"n":7,"n_ran":7,"n_unverified":0,"n_pointer_only":7}},{"url":"/paper/rethinking-space-time-networks-with-improved","title":"Rethinking Space-Time Networks with Improved Memory Coverage for Efficient Video Object Segmentation","date":"2021-06-09","arxiv_id":"2106.05210","repositories_listed":3,"syntology":{"n":10,"n_ran":6,"n_unverified":4,"n_pointer_only":2}},{"url":"/paper/physarum-powered-differentiable-linear","title":"Physarum Powered Differentiable Linear Programming Layers and Applications","date":"2020-04-30","arxiv_id":"2004.14539","repositories_listed":3,"syntology":null},{"url":"/paper/exploiting-geometric-constraints-on-dense","title":"EpO-Net: Exploiting Geometric Constraints on Dense Trajectories for Motion Saliency","date":"2019-09-29","arxiv_id":"1909.13258","repositories_listed":3,"syntology":null},{"url":"/paper/video-object-segmentation-using-space-time","title":"Video Object Segmentation using Space-Time Memory Networks","date":"2019-04-01","arxiv_id":"1904.00607","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/feelvos-fast-end-to-end-embedding-learning","title":"FEELVOS: Fast End-to-End Embedding Learning for Video Object Segmentation","date":"2019-02-25","arxiv_id":"1902.09513","repositories_listed":3,"syntology":{"n":9,"n_ran":1,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/fast-online-object-tracking-and-segmentation","title":"Fast Online Object Tracking and Segmentation: A Unifying Approach","date":"2018-12-12","arxiv_id":"1812.05050","repositories_listed":3,"syntology":{"n":10,"n_ran":1,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/video-object-segmentation-with-re","title":"Video Object Segmentation with Re-identification","date":"2017-08-01","arxiv_id":"1708.00197","repositories_listed":3,"syntology":null},{"url":"/paper/accelerating-volumetric-medical-image","title":"Accelerating Volumetric Medical Image Annotation via Short-Long Memory SAM 2","date":"2025-05-03","arxiv_id":"2505.01854","repositories_listed":2,"syntology":null},{"url":"/paper/stable-mean-teacher-for-semi-supervised-video","title":"Stable Mean Teacher for Semi-supervised Video Action Detection","date":"2024-12-10","arxiv_id":"2412.07072","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/visa-reasoning-video-object-segmentation-via","title":"VISA: Reasoning Video Object Segmentation via Large Language Models","date":"2024-07-16","arxiv_id":"2407.11325","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/pvuw-2024-challenge-on-complex-video","title":"PVUW 2024 Challenge on Complex Video Understanding: Methods and Results","date":"2024-06-24","arxiv_id":"2406.17005","repositories_listed":2,"syntology":null},{"url":"/paper/uniref-segment-every-reference-object-in","title":"UniRef++: Segment Every Reference Object in Spatial and Temporal Spaces","date":"2023-12-25","arxiv_id":"2312.15715","repositories_listed":2,"syntology":{"n":9,"n_ran":9,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/xmem-production-level-video-segmentation-from","title":"XMem++: Production-level Video Segmentation From Few Annotated Frames","date":"2023-07-29","arxiv_id":"2307.15958","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/online-unsupervised-video-object-segmentation","title":"Online Unsupervised Video Object Segmentation via Contrastive Motion Clustering","date":"2023-06-21","arxiv_id":"2306.12048","repositories_listed":2,"syntology":null},{"url":"/paper/video-object-segmentation-in-panoptic-wild","title":"Video Object Segmentation in Panoptic Wild Scenes","date":"2023-05-08","arxiv_id":"2305.04470","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/dsec-mos-segment-any-moving-object-with","title":"Event-Free Moving Object Segmentation from Moving Ego Vehicle","date":"2023-04-28","arxiv_id":"2305.00126","repositories_listed":2,"syntology":null}],"syntology_records":18,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}