{"url":"/dataset/mose","name":"MOSE","full_name":"Complex Video Object Segmentation","description_markdown":"**CoMplex video Object SEgmentation (MOSE)** is a dataset to study the tracking and segmenting objects in complex environments. MOSE contains 2,149 video clips and 5,200 objects from 36 categories, with 431,725 high-quality object segmentation masks. The most notable feature of MOSE dataset is complex scenes with crowded and occluded objects.\r\n\r\nSource: [MOSE: A New Dataset for Video Object Segmentation in Complex Scenes](https://arxiv.org/pdf/2302.01872.pdf)","description_withheld":null,"homepage":"https://henghuiding.github.io/MOSE","introduced_date":"2023-02-03","introduced_date_note":null,"introduced_by":{"paper":"/paper/mose-a-new-dataset-for-video-object","title":"MOSE: A New Dataset for Video Object Segmentation in Complex Scenes","first_author":"Henghui Ding","url":null},"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Video Object Segmentation","url":"/task/video-object-segmentation","datasets_with_task":"/datasets/task/video-object-segmentation"},{"name":"Semi-Supervised Video Object Segmentation","url":"/task/semi-supervised-video-object-segmentation","datasets_with_task":"/datasets/task/semi-supervised-video-object-segmentation"},{"name":"Unsupervised Video Object Segmentation","url":"/task/unsupervised-video-object-segmentation","datasets_with_task":"/datasets/task/unsupervised-video-object-segmentation"},{"name":"Video Semantic Segmentation","url":"/task/video-semantic-segmentation","datasets_with_task":"/datasets/task/video-semantic-segmentation"},{"name":"Interactive Video Object Segmentation","url":"/task/interactive-video-object-segmentation","datasets_with_task":"/datasets/task/interactive-video-object-segmentation"}],"languages":[],"variants":["MOSE"],"data_loaders":[{"repo":"https://github.com/henghuiding/MOSE-api","url":"https://github.com/henghuiding/MOSE-api","frameworks":["pytorch"]}],"num_papers_in_archive":49,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/semi-supervised-video-object-segmentation-on-21","task":"Semi-Supervised Video Object Segmentation","dataset_variant":"MOSE","rows":17,"metrics":["J&F","J","F","FPS"],"first_row_in_archive_order":{"model":"SAM2","paper":"/paper/2408-00714","metrics":{"J&F":"77.9"},"code_links":[{"title":"facebookresearch/segment-anything","url":"https://github.com/facebookresearch/segment-anything"},{"title":"facebookresearch/sam2","url":"https://github.com/facebookresearch/sam2"},{"title":"yangchris11/samurai","url":"https://github.com/yangchris11/samurai"},{"title":"idea-research/grounded-sam-2","url":"https://github.com/idea-research/grounded-sam-2"},{"title":"ibaiGorordo/ONNX-SAM2-Segment-Anything","url":"https://github.com/ibaiGorordo/ONNX-SAM2-Segment-Anything"},{"title":"bowang-lab/medsam2","url":"https://github.com/bowang-lab/medsam2"},{"title":"TripleJoy/SAM2MOT","url":"https://github.com/TripleJoy/SAM2MOT"},{"title":"louisfinner/him2sam","url":"https://github.com/louisfinner/him2sam"},{"title":"MindCode-4/code-4","url":"https://github.com/MindCode-4/code-4/tree/main/sam"},{"title":"dcnieho/segment-anything-2","url":"https://github.com/dcnieho/segment-anything-2"},{"title":"MindSpore-scientific/code-7","url":"https://github.com/MindSpore-scientific/code-7/tree/main/SAM"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-object-segmentation-on-mose","task":"Video Object Segmentation","dataset_variant":"MOSE","rows":1,"metrics":["J&F"],"first_row_in_archive_order":{"model":"Cutie","paper":"/paper/putting-the-object-back-into-video-object","metrics":{"J&F":"68.3"},"code_links":[{"title":"hkchengrex/Cutie","url":"https://github.com/hkchengrex/Cutie"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/2408-00714","title":"SAM 2: Segment Anything in Images and Videos","date":"2024-08-01","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":49,"samples_ran":37,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/putting-the-object-back-into-video-object","title":"Putting the Object Back into Video Object Segmentation","date":"2023-10-19","rows_on_this_dataset":9,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tracking-anything-with-decoupled-video","title":"Tracking Anything with Decoupled Video Segmentation","date":"2023-09-07","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":10,"samples_ran":7,"samples_unverified":3,"pointer_only_for_licence":10,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/decoupling-features-in-hierarchical","title":"Decoupling Features in Hierarchical Propagation for Video Object Segmentation","date":"2022-10-18","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/swem-towards-real-time-video-object-1","title":"SWEM: Towards Real-Time Video Object Segmentation with Sequential Weighted Expectation-Maximization","date":"2022-08-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/xmem-long-term-video-object-segmentation-with","title":"XMem: Long-Term Video Object Segmentation with an Atkinson-Shiffrin Memory Model","date":"2022-07-14","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/recurrent-dynamic-embedding-for-video-object","title":"Recurrent Dynamic Embedding for Video Object Segmentation","date":"2022-05-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rethinking-space-time-networks-with-improved","title":"Rethinking Space-Time Networks with Improved Memory Coverage for Efficient Video Object Segmentation","date":"2021-06-09","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":10,"samples_ran":6,"samples_unverified":4,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/associating-objects-with-transformers-for","title":"Associating Objects with Transformers for Video Object Segmentation","date":"2021-06-04","rows_on_this_dataset":1,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":5,"samples_harvested":77,"samples_ran":54,"samples_unverified":23,"pointer_only_for_licence":14,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}