{"url":"/dataset/ovis","name":"OVIS","full_name":"Occluded Video Instance Segmentation","description_markdown":"OVIS is a new large scale benchmark dataset for video instance segmentation task. It is designed with the philosophy of perceiving object occlusions in videos, which could reveal the complexity and the diversity of real-world scenes. OVIS consists of:\r\n\r\n* 296k high-quality instance masks\r\n* 25 commonly seen semantic categories\r\n* 901 videos with severe object occlusions\r\n* 5,223 unique instances\r\n\r\nIf the description or image is from a different paper, please refer to it as follows:\r\nSource: [http://songbai.site/ovis/](http://songbai.site/ovis/)","description_withheld":null,"homepage":"http://songbai.site/ovis/","introduced_date":"2021-02-02","introduced_date_note":null,"introduced_by":{"paper":"/paper/occluded-video-instance-segmentation","title":"Occluded Video Instance Segmentation: A Benchmark","first_author":"Jiyang Qi","url":null},"license":{"name":"Creative Commons Attribution-NonCommercial-ShareAlike 4.0 License","url":"https://creativecommons.org/licenses/by-nc-sa/4.0/"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Video Instance Segmentation","url":"/task/video-instance-segmentation","datasets_with_task":"/datasets/task/video-instance-segmentation"}],"languages":[],"variants":["OVIS","OVIS validation"],"data_loaders":[{"repo":"https://github.com/qjy981010/CMaskTrack-RCNN","url":"https://github.com/qjy981010/CMaskTrack-RCNN","frameworks":[]}],"num_papers_in_archive":75,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-instance-segmentation-on-ovis-1","task":"Video Instance Segmentation","dataset_variant":"OVIS validation","rows":44,"metrics":["mask AP","AP50","AP75","APho","APmo","AR1","APso","AR10"],"first_row_in_archive_order":{"model":"DVIS-DAQ(VIT-L, Offline)","paper":"/paper/dvis-daq-improving-video-segmentation-via","metrics":{"AP50":"83.8","AP75":"62.9","mask AP":"57.1"},"code_links":[{"title":"zhang-tao-whu/DVIS","url":"https://github.com/zhang-tao-whu/DVIS"},{"title":"zhang-tao-whu/DVIS_Plus","url":"https://github.com/zhang-tao-whu/DVIS_Plus"},{"title":"skyworkai/daq-vs","url":"https://github.com/skyworkai/daq-vs"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/context-aware-video-instance-segmentation","title":"Context-Aware Video Instance Segmentation","date":"2024-07-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dvis-daq-improving-video-segmentation-via","title":"DVIS-DAQ: Improving Video Segmentation via Dynamic Anchor Queries","date":"2024-03-29","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":1,"samples_unverified":4,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/univs-unified-and-universal-video","title":"UniVS: Unified and Universal Video Segmentation with Prompts as Queries","date":"2024-02-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":12,"samples_unverified":2,"pointer_only_for_licence":14,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dvis-improved-decoupled-framework-for","title":"DVIS++: Improved Decoupled Framework for Universal Video Segmentation","date":"2023-12-20","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/general-object-foundation-model-for-images","title":"General Object Foundation Model for Images and Videos at Scale","date":"2023-12-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":8,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/novis-a-case-for-end-to-end-near-online-video","title":"NOVIS: A Case for End-to-End Near-Online Video Instance Segmentation","date":"2023-08-29","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/ctvis-consistent-training-for-online-video","title":"CTVIS: Consistent Training for Online Video Instance Segmentation","date":"2023-07-24","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":4,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/refinevis-video-instance-segmentation-with","title":"RefineVIS: Video Instance Segmentation with Temporal Attention Refinement","date":"2023-06-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dvis-decoupled-video-instance-segmentation","title":"DVIS: Decoupled Video Instance Segmentation Framework","date":"2023-06-06","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/gratt-vis-gated-residual-attention-for-auto","title":"GRAtt-VIS: Gated Residual Attention for Auto Rectifying Video Instance Segmentation","date":"2023-05-26","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/boxvis-video-instance-segmentation-with-box","title":"BoxVIS: Video Instance Segmentation with Box Annotations","date":"2023-03-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mdqe-mining-discriminative-query-embeddings","title":"MDQE: Mining Discriminative Query Embeddings to Segment Occluded Instances on Challenging Videos","date":"2023-03-25","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tube-link-a-flexible-cross-tube-baseline-for","title":"Tube-Link: A Flexible Cross Tube Framework for Universal Video Segmentation","date":"2023-03-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/universal-instance-perception-as-object","title":"Universal Instance Perception as Object Discovery and Retrieval","date":"2023-03-12","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tarvis-a-unified-approach-for-target-based","title":"TarViS: A Unified Approach for Target-based Video Segmentation","date":"2023-01-06","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/robust-online-video-instance-segmentation","title":"Robust Online Video Instance Segmentation with Track Queries","date":"2022-11-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-generalized-framework-for-video-instance","title":"A Generalized Framework for Video Instance Segmentation","date":"2022-11-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/instanceformer-an-online-video-instance","title":"InstanceFormer: An Online Video Instance Segmentation Framework","date":"2022-08-22","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/minvis-a-minimal-video-instance-segmentation","title":"MinVIS: A Minimal Video Instance Segmentation Framework without Video-based Training","date":"2022-08-03","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/devis-making-deformable-transformers-work-for","title":"DeVIS: Making Deformable Transformers Work for Video Instance Segmentation","date":"2022-07-22","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":2,"samples_unverified":9,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/in-defense-of-online-models-for-video","title":"In Defense of Online Models for Video Instance Segmentation","date":"2022-07-21","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vita-video-instance-segmentation-via-object","title":"VITA: Video Instance Segmentation via Object Token Association","date":"2022-06-09","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":0,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/temporally-efficient-vision-transformer-for","title":"Temporally Efficient Vision Transformer for Video Instance Segmentation","date":"2022-04-18","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/stc-spatio-temporal-contrastive-learning-for","title":"STC: Spatio-Temporal Contrastive Learning for Video Instance Segmentation","date":"2022-02-08","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/mask2former-for-video-instance-segmentation","title":"Mask2Former for Video Instance Segmentation","date":"2021-12-20","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/d2conv3d-dynamic-dilated-convolutions-for","title":"D2Conv3D: Dynamic Dilated Convolutions for Object Segmentation in Videos","date":"2021-11-15","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/crossover-learning-for-fast-online-video","title":"Crossover Learning for Fast Online Video Instance Segmentation","date":"2021-04-13","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/spatial-feature-calibration-and-temporal","title":"Spatial Feature Calibration and Temporal Fusion for Effective One-stage Video Instance Segmentation","date":"2021-04-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/occluded-video-instance-segmentation","title":"Occluded Video Instance Segmentation: A Benchmark","date":"2021-02-02","rows_on_this_dataset":2,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":12,"samples_harvested":85,"samples_ran":38,"samples_unverified":47,"pointer_only_for_licence":32,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}