{"url":"/task/referring-video-object-segmentation","name":"Referring Video Object Segmentation","slug":"referring-video-object-segmentation","description_markdown":"Referring video object segmentation aims at segmenting an object in video with language expressions. Unlike the previous video object segmentation, the task exploits a different type of supervision, language expressions, to identify and segment an object referred by the given language expressions in a video.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":74,"papers_with_code":50,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/referring-video-object-segmentation-on-refer","slug":"referring-video-object-segmentation-on-refer","dataset":"Refer-YouTube-VOS","dataset_url":"/dataset/refer-youtube-vos","rows_in_archive":18,"metrics":["J&F","J","F"],"first_row_in_archive_order":{"model":"FindTrack","paper_title":"Find First, Track Next: Decoupling Identification and Propagation in Referring Video Object Segmentation","paper_url":"/paper/find-first-track-next-decoupling","paper_date":"2025-03-05","arxiv_id":"2503.03492","code_links":[{"title":"suhwan-cho/FindTrack","url":"https://github.com/suhwan-cho/FindTrack"}],"syntology":null}},{"leaderboard":"/sota/referring-video-object-segmentation-on-mevis","slug":"referring-video-object-segmentation-on-mevis","dataset":"MeViS","dataset_url":"/dataset/mevis","rows_in_archive":16,"metrics":["J&F","J","F"],"first_row_in_archive_order":{"model":"MPG-SAM 2","paper_title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","paper_url":"/paper/mpg-sam-2-adapting-sam-2-with-mask-priors-and","paper_date":"2025-01-23","arxiv_id":"2501.13667","code_links":[{"title":"rongfu-dsb/MPG-SAM2","url":"https://github.com/rongfu-dsb/MPG-SAM2"}],"syntology":{"n":16,"n_ran":5,"n_unverified":11,"n_pointer_only":0}}},{"leaderboard":"/sota/referring-video-object-segmentation-on-ref","slug":"referring-video-object-segmentation-on-ref","dataset":"Ref-DAVIS17","dataset_url":null,"rows_in_archive":11,"metrics":["J&F","J","F"],"first_row_in_archive_order":{"model":"FindTrack","paper_title":"Find First, Track Next: Decoupling Identification and Propagation in Referring Video Object Segmentation","paper_url":"/paper/find-first-track-next-decoupling","paper_date":"2025-03-05","arxiv_id":"2503.03492","code_links":[{"title":"suhwan-cho/FindTrack","url":"https://github.com/suhwan-cho/FindTrack"}],"syntology":null}},{"leaderboard":"/sota/referring-video-object-segmentation-on-revos","slug":"referring-video-object-segmentation-on-revos","dataset":"ReVOS","dataset_url":"/dataset/revos","rows_in_archive":9,"metrics":["J&F","J","F","R"],"first_row_in_archive_order":{"model":"VRS-HQ (Chat-UniVi-13B)","paper_title":"The Devil is in Temporal Token: High Quality Video Reasoning Segmentation","paper_url":"/paper/the-devil-is-in-temporal-token-high-quality","paper_date":"2025-01-15","arxiv_id":"2501.08549","code_links":[{"title":"sitonggong/vrs-hq","url":"https://github.com/sitonggong/vrs-hq"}],"syntology":null}},{"leaderboard":"/sota/referring-video-object-segmentation-on-long","slug":"referring-video-object-segmentation-on-long","dataset":"Long-RVOS","dataset_url":"/dataset/long-rvos","rows_in_archive":7,"metrics":["J&F","tIoU","vIoU"],"first_row_in_archive_order":{"model":"ReferMo","paper_title":"Long-RVOS: A Comprehensive Benchmark for Long-term Referring Video Object Segmentation","paper_url":"/paper/long-rvos-a-comprehensive-benchmark-for-long","paper_date":"2025-05-19","arxiv_id":"2505.12702","code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/refer-youtube-vos","name":"Refer-YouTube-VOS","full_name":"","num_papers_in_archive":52},{"url":"/dataset/mevis","name":"MeViS","full_name":"Motion expressions Video Segmentation","num_papers_in_archive":46},{"url":"/dataset/revos","name":"ReVOS","full_name":"","num_papers_in_archive":18},{"url":"/dataset/long-rvos","name":"Long-RVOS","full_name":"","num_papers_in_archive":7}],"subtasks":[],"parent_tasks":[{"url":"/task/video-object-segmentation","name":"Video Object Segmentation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":50,"tagged_in_all":74,"items":[{"url":"/paper/visa-reasoning-video-object-segmentation-via","title":"VISA: Reasoning Video Object Segmentation via Large Language Models","date":"2024-07-16","arxiv_id":"2407.11325","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/uniref-segment-every-reference-object-in","title":"UniRef++: Segment Every Reference Object in Spatial and Temporal Spaces","date":"2023-12-25","arxiv_id":"2312.15715","repositories_listed":2,"syntology":{"n":9,"n_ran":9,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/lisa-reasoning-segmentation-via-large","title":"LISA: Reasoning Segmentation via Large Language Model","date":"2023-08-01","arxiv_id":"2308.00692","repositories_listed":2,"syntology":null},{"url":"/paper/end-to-end-referring-video-object","title":"End-to-End Referring Video Object Segmentation with Multimodal Transformers","date":"2021-11-29","arxiv_id":"2111.14821","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/videomolmo-spatio-temporal-grounding-meets","title":"VideoMolmo: Spatio-Temporal Grounding Meets Pointing","date":"2025-06-05","arxiv_id":"2506.05336","repositories_listed":1,"syntology":null},{"url":"/paper/few-shot-referring-video-single-and-multi","title":"Few-Shot Referring Video Single- and Multi-Object Segmentation via Cross-Modal Affinity with Instance Sequence Matching","date":"2025-04-18","arxiv_id":"2504.13710","repositories_listed":1,"syntology":null},{"url":"/paper/glus-global-local-reasoning-unified-into-a","title":"GLUS: Global-Local Reasoning Unified into A Single Large Language Model for Video Segmentation","date":"2025-04-10","arxiv_id":"2504.07962","repositories_listed":1,"syntology":{"n":12,"n_ran":3,"n_unverified":9,"n_pointer_only":12}},{"url":"/paper/the-1st-solution-for-4th-pvuw-mevis-challenge","title":"The 1st Solution for 4th PVUW MeViS Challenge: Unleashing the Potential of Large Multimodal Models for Referring Video Segmentation","date":"2025-04-07","arxiv_id":"2504.05178","repositories_listed":1,"syntology":null},{"url":"/paper/4th-pvuw-mevis-3rd-place-report-sa2va","title":"4th PVUW MeViS 3rd Place Report: Sa2VA","date":"2025-04-01","arxiv_id":"2504.00476","repositories_listed":1,"syntology":null},{"url":"/paper/referdino-plus-2nd-solution-for-4th-pvuw","title":"ReferDINO-Plus: 2nd Solution for 4th PVUW MeViS Challenge at CVPR 2025","date":"2025-03-30","arxiv_id":"2503.23509","repositories_listed":1,"syntology":null},{"url":"/paper/find-first-track-next-decoupling","title":"Find First, Track Next: Decoupling Identification and Propagation in Referring Video Object Segmentation","date":"2025-03-05","arxiv_id":"2503.03492","repositories_listed":1,"syntology":null},{"url":"/paper/mpg-sam-2-adapting-sam-2-with-mask-priors-and","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","date":"2025-01-23","arxiv_id":"2501.13667","repositories_listed":1,"syntology":{"n":16,"n_ran":5,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/internvideo2-5-empowering-video-mllms-with","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","date":"2025-01-21","arxiv_id":"2501.12386","repositories_listed":1,"syntology":null},{"url":"/paper/the-devil-is-in-temporal-token-high-quality","title":"The Devil is in Temporal Token: High Quality Video Reasoning Segmentation","date":"2025-01-15","arxiv_id":"2501.08549","repositories_listed":1,"syntology":null},{"url":"/paper/multi-context-temporal-consistent-modeling","title":"Multi-Context Temporal Consistent Modeling for Referring Video Object Segmentation","date":"2025-01-09","arxiv_id":"2501.04939","repositories_listed":1,"syntology":null},{"url":"/paper/sa2va-marrying-sam2-with-llava-for-dense","title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","date":"2025-01-07","arxiv_id":"2501.04001","repositories_listed":1,"syntology":null},{"url":"/paper/dtos-dynamic-time-object-sensing-with-large","title":"DTOS: Dynamic Time Object Sensing with Large Multimodal Model","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/referring-video-object-segmentation-via","title":"Referring Video Object Segmentation via Language-aligned Track Selection","date":"2024-12-02","arxiv_id":"2412.01136","repositories_listed":1,"syntology":null},{"url":"/paper/samwise-infusing-wisdom-in-sam2-for-text","title":"SAMWISE: Infusing Wisdom in SAM2 for Text-Driven Video Segmentation","date":"2024-11-26","arxiv_id":"2411.17646","repositories_listed":1,"syntology":null},{"url":"/paper/hyperseg-towards-universal-visual","title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","date":"2024-11-26","arxiv_id":"2411.17606","repositories_listed":1,"syntology":{"n":17,"n_ran":7,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/one-token-to-seg-them-all-language-instructed","title":"One Token to Seg Them All: Language Instructed Reasoning Segmentation in Videos","date":"2024-09-29","arxiv_id":"2409.19603","repositories_listed":1,"syntology":{"n":16,"n_ran":7,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/actionvos-actions-as-prompts-for-video-object","title":"ActionVOS: Actions as Prompts for Video Object Segmentation","date":"2024-07-10","arxiv_id":"2407.07402","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_unverified":1,"n_pointer_only":15}},{"url":"/paper/1st-place-solution-for-mevis-track-in-cvpr","title":"1st Place Solution for MeViS Track in CVPR 2024 PVUW Workshop: Motion Expression guided Video Segmentation","date":"2024-06-11","arxiv_id":"2406.07043","repositories_listed":1,"syntology":null},{"url":"/paper/improving-referring-image-segmentation-using","title":"Vision-Aware Text Features in Referring Image Segmentation: From Object Understanding to Context Understanding","date":"2024-04-12","arxiv_id":"2404.08590","repositories_listed":1,"syntology":null},{"url":"/paper/decoupling-static-and-hierarchical-motion","title":"Decoupling Static and Hierarchical Motion Perception for Referring Video Segmentation","date":"2024-04-04","arxiv_id":"2404.03645","repositories_listed":1,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":7}},{"url":"/paper/towards-temporally-consistent-referring-video","title":"Temporally Consistent Referring Video Object Segmentation with Hybrid Memory","date":"2024-03-28","arxiv_id":"2403.19407","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/exploring-pre-trained-text-to-video-diffusion","title":"Exploring Pre-trained Text-to-Video Diffusion Models for Referring Video Object Segmentation","date":"2024-03-18","arxiv_id":"2403.12042","repositories_listed":1,"syntology":{"n":20,"n_ran":16,"n_unverified":4,"n_pointer_only":20}},{"url":"/paper/univs-unified-and-universal-video","title":"UniVS: Unified and Universal Video Segmentation with Prompts as Queries","date":"2024-02-28","arxiv_id":"2402.18115","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_unverified":2,"n_pointer_only":14}},{"url":"/paper/1st-place-solution-for-5th-lsvos-challenge","title":"1st Place Solution for 5th LSVOS Challenge: Referring Video Object Segmentation","date":"2024-01-01","arxiv_id":"2401.00663","repositories_listed":1,"syntology":null},{"url":"/paper/tracking-with-human-intent-reasoning","title":"Tracking with Human-Intent Reasoning","date":"2023-12-29","arxiv_id":"2312.17448","repositories_listed":1,"syntology":null}],"syntology_records":12,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}