{"url":"/dataset/mevis","name":"MeViS","full_name":"Motion expressions Video Segmentation","description_markdown":"MeViS is a large-scale dataset for motion expressions guided video segmentation, which focuses on segmenting objects in video content based on a sentence describing the motion of the objects. The dataset contains numerous motion expressions to indicate target objects in complex environments.","description_withheld":null,"homepage":"https://henghuiding.github.io/MeViS/","introduced_date":"2023-08-16","introduced_date_note":null,"introduced_by":{"paper":"/paper/mevis-a-large-scale-benchmark-for-video","title":"MeViS: A Large-scale Benchmark for Video Segmentation with Motion Expressions","first_author":"Henghui Ding","url":null},"license":null,"modalities":[],"tasks":[{"name":"Referring Video Object Segmentation","url":"/task/referring-video-object-segmentation","datasets_with_task":"/datasets/task/referring-video-object-segmentation"}],"languages":[],"variants":["MeViS"],"data_loaders":[],"num_papers_in_archive":46,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/referring-video-object-segmentation-on-mevis","task":"Referring Video Object Segmentation","dataset_variant":"MeViS","rows":16,"metrics":["J&F","J","F"],"first_row_in_archive_order":{"model":"MPG-SAM 2","paper":"/paper/mpg-sam-2-adapting-sam-2-with-mask-priors-and","metrics":{"F":"56.7","J":"50.7","J&F":"53.7"},"code_links":[{"title":"rongfu-dsb/MPG-SAM2","url":"https://github.com/rongfu-dsb/MPG-SAM2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/glus-global-local-reasoning-unified-into-a","title":"GLUS: Global-Local Reasoning Unified into A Single Large Language Model for Video Segmentation","date":"2025-04-10","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":3,"samples_unverified":9,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/find-first-track-next-decoupling","title":"Find First, Track Next: Decoupling Identification and Propagation in Referring Video Object Segmentation","date":"2025-03-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/referdino-referring-video-object-segmentation","title":"ReferDINO: Referring Video Object Segmentation with Visual Grounding Foundations","date":"2025-01-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/mpg-sam-2-adapting-sam-2-with-mask-priors-and","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","date":"2025-01-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":5,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/internvideo2-5-empowering-video-mllms-with","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","date":"2025-01-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/the-devil-is-in-temporal-token-high-quality","title":"The Devil is in Temporal Token: High Quality Video Reasoning Segmentation","date":"2025-01-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multi-context-temporal-consistent-modeling","title":"Multi-Context Temporal Consistent Modeling for Referring Video Object Segmentation","date":"2025-01-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/samwise-infusing-wisdom-in-sam2-for-text","title":"SAMWISE: Infusing Wisdom in SAM2 for Text-Driven Video Segmentation","date":"2024-11-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/decoupling-static-and-hierarchical-motion","title":"Decoupling Static and Hierarchical Motion Perception for Referring Video Segmentation","date":"2024-04-04","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-temporally-consistent-referring-video","title":"Temporally Consistent Referring Video Object Segmentation with Hybrid Memory","date":"2024-03-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":14,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mevis-a-large-scale-benchmark-for-video","title":"MeViS: A Large-scale Benchmark for Video Segmentation with Motion Expressions","date":"2023-08-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vlt-vision-language-transformer-and-query","title":"VLT: Vision-Language Transformer and Query Generation for Referring Segmentation","date":"2022-10-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/language-bridged-spatial-temporal-interaction-1","title":"Language-Bridged Spatial-Temporal Interaction for Referring Video Object Segmentation","date":"2022-06-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/language-as-queries-for-referring-video","title":"Language as Queries for Referring Video Object Segmentation","date":"2022-01-03","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":7,"samples_unverified":1,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/end-to-end-referring-video-object","title":"End-to-End Referring Video Object Segmentation with Multimodal Transformers","date":"2021-11-29","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":6,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/urvos-unified-referring-video-object","title":"URVOS: Unified Referring Video Object Segmentation Network with a Large-Scale Benchmark","date":"2020-08-01","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":9,"samples_harvested":88,"samples_ran":43,"samples_unverified":45,"pointer_only_for_licence":27,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}