{"url":"/sota/referring-expression-segmentation-on-davis","task":{"name":"Referring Expression Segmentation","url":"/task/referring-expression-segmentation","note":null},"dataset":{"name":"DAVIS 2017 (val)","url":"/dataset/davis-2017"},"category":"Computer Vision","categories":["Computer Vision"],"category_note":null,"description":"The task aims at labeling the pixels of an image or video that represent an object instance referred by a linguistic expression. In particular, the referring expression (RE) must allow the identification of an individual object in a discourse or scene (the referent). REs unambiguously identify the target instance.","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["J&F 1st frame","J&F Full video","Zero-Shot Transfer","J&F score"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"J&F 1st frame":null,"J&F Full video":null,"Zero-Shot Transfer":null,"J&F score":"higher"}},"counts":{"rows":18,"rows_with_code":15,"rows_with_paper_page":18,"rows_dated":15,"rows_using_additional_data":11},"rows":[{"rank_in_archive_order":1,"model":"UNINEXT-H","metrics":{"J&F 1st frame":"72.5"},"uses_additional_data":true,"paper_date":"2023-03-12","paper":"/paper/universal-instance-perception-as-object","paper_url":"https://arxiv.org/abs/2303.06674v2","paper_title":"Universal Instance Perception as Object Discovery and Retrieval","code":"https://github.com/MasterBin-IIAU/UNINEXT","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":1,"n_samples":4,"n_pointer_only_licence":0}},{"rank_in_archive_order":2,"model":"HyperSeg","metrics":{"J&F 1st frame":"71.2"},"uses_additional_data":true,"paper_date":"2024-11-26","paper":"/paper/hyperseg-towards-universal-visual","paper_url":"https://arxiv.org/abs/2411.17606v2","paper_title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","code":"https://github.com/congvvc/HyperSeg","n_code_links":1,"syntology":{"n_ran":7,"n_unverified":10,"n_samples":17,"n_pointer_only_licence":0}},{"rank_in_archive_order":3,"model":"DEVA (ReferFormer)","metrics":{"J&F 1st frame":"66.3"},"uses_additional_data":true,"paper_date":"2023-09-07","paper":"/paper/tracking-anything-with-decoupled-video","paper_url":"https://arxiv.org/abs/2309.03903v1","paper_title":"Tracking Anything with Decoupled Video Segmentation","code":"https://github.com/hkchengrex/Tracking-Anything-with-DEVA","n_code_links":1,"syntology":{"n_ran":7,"n_unverified":3,"n_samples":10,"n_pointer_only_licence":10}},{"rank_in_archive_order":4,"model":"HTR","metrics":{"J&F 1st frame":"65.6"},"uses_additional_data":true,"paper_date":"2024-03-28","paper":"/paper/towards-temporally-consistent-referring-video","paper_url":"https://arxiv.org/abs/2403.19407v2","paper_title":"Temporally Consistent Referring Video Object Segmentation with Hybrid Memory","code":"https://github.com/bo-miao/HTR","n_code_links":1,"syntology":{"n_ran":14,"n_unverified":1,"n_samples":15,"n_pointer_only_licence":0}},{"rank_in_archive_order":5,"model":"SgMg","metrics":{"J&F 1st frame":"63.3"},"uses_additional_data":true,"paper_date":"2023-07-25","paper":"/paper/spectrum-guided-multi-granularity-referring","paper_url":"https://arxiv.org/abs/2307.13537v1","paper_title":"Spectrum-guided Multi-granularity Referring Video Object Segmentation","code":"https://github.com/bo-miao/sgmg","n_code_links":1,"syntology":{"n_ran":6,"n_unverified":3,"n_samples":9,"n_pointer_only_licence":9}},{"rank_in_archive_order":6,"model":"SafaRi-B","metrics":{"J&F 1st frame":"61.3","Zero-Shot Transfer":"true"},"uses_additional_data":false,"paper_date":"2024-07-02","paper":"/paper/safari-adaptive-sequence-transformer-for","paper_url":"https://arxiv.org/abs/2407.02389v1","paper_title":"SafaRi:Adaptive Sequence Transformer for Weakly Supervised Referring Expression Segmentation","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":7,"model":"ReferFormer","metrics":{"J&F 1st frame":"61.1"},"uses_additional_data":true,"paper_date":"2022-01-03","paper":"/paper/language-as-queries-for-referring-video","paper_url":"https://arxiv.org/abs/2201.00487v2","paper_title":"Language as Queries for Referring Video Object Segmentation","code":"https://github.com/wjn922/referformer","n_code_links":1,"syntology":{"n_ran":7,"n_unverified":1,"n_samples":8,"n_pointer_only_licence":8}},{"rank_in_archive_order":8,"model":"PolyFormer-B","metrics":{"J&F 1st frame":"60.9","Zero-Shot Transfer":"true"},"uses_additional_data":true,"paper_date":"2023-02-14","paper":"/paper/polyformer-referring-image-segmentation-as","paper_url":"https://arxiv.org/abs/2302.07387v2","paper_title":"PolyFormer: Referring Image Segmentation as Sequential Polygon Generation","code":"https://github.com/amazon-science/polygon-transformer","n_code_links":1,"syntology":{"n_ran":5,"n_unverified":1,"n_samples":6,"n_pointer_only_licence":6}},{"rank_in_archive_order":9,"model":"URVOS + Refer-Youtube-VOS + ft. DAVIS","metrics":{"J&F 1st frame":"51.63"},"uses_additional_data":true,"paper_date":null,"paper":"/paper/urvos-unified-referring-video-object","paper_url":"https://www.ecva.net/papers/eccv_2020/papers_ECCV/html/2327_ECCV_2020_paper.php","paper_title":"URVOS: Unified Referring Video Object Segmentation Network with a Large-Scale Benchmark","code":"https://github.com/skynbe/Refer-Youtube-VOS","n_code_links":1,"syntology":null},{"rank_in_archive_order":10,"model":"HINet","metrics":{"J&F 1st frame":"50.2","J&F Full video":"47.9"},"uses_additional_data":false,"paper_date":"2021-11-22","paper":"/paper/hierarchical-interaction-network-for-video","paper_url":"https://www.bmvc2021-virtualconference.com/conference/papers/paper_0386.html","paper_title":"Hierarchical interaction network for video object segmentation from referring expressions","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":11,"model":"URVOS + Refer-Youtube-VOS","metrics":{"J&F 1st frame":"46.85"},"uses_additional_data":true,"paper_date":null,"paper":"/paper/urvos-unified-referring-video-object","paper_url":"https://www.ecva.net/papers/eccv_2020/papers_ECCV/html/2327_ECCV_2020_paper.php","paper_title":"URVOS: Unified Referring Video Object Segmentation Network with a Large-Scale Benchmark","code":"https://github.com/skynbe/Refer-Youtube-VOS","n_code_links":1,"syntology":null},{"rank_in_archive_order":12,"model":"RefVOS + SynthRef-YouTube-VIS","metrics":{"J&F 1st frame":"45.3","J&F Full video":"44.8"},"uses_additional_data":true,"paper_date":"2021-06-08","paper":"/paper/synthref-generation-of-synthetic-referring","paper_url":"https://arxiv.org/abs/2106.04403v2","paper_title":"SynthRef: Generation of Synthetic Referring Expressions for Object Segmentation","code":"https://github.com/miriambellver/refvos","n_code_links":2,"syntology":null},{"rank_in_archive_order":13,"model":"RefVOS","metrics":{"J&F 1st frame":"45.1"},"uses_additional_data":false,"paper_date":"2021-06-08","paper":"/paper/synthref-generation-of-synthetic-referring","paper_url":"https://arxiv.org/abs/2106.04403v2","paper_title":"SynthRef: Generation of Synthetic Referring Expressions for Object Segmentation","code":"https://github.com/miriambellver/refvos","n_code_links":2,"syntology":null},{"rank_in_archive_order":14,"model":"RefVOS","metrics":{"J&F 1st frame":"44.5","J&F Full video":"45.1"},"uses_additional_data":false,"paper_date":"2020-10-01","paper":"/paper/refvos-a-closer-look-at-referring-expressions","paper_url":"https://arxiv.org/abs/2010.00263v1","paper_title":"RefVOS: A Closer Look at Referring Expressions for Video Object Segmentation","code":"https://github.com/miriambellver/refvos","n_code_links":2,"syntology":null},{"rank_in_archive_order":15,"model":"URVOS","metrics":{"J&F 1st frame":"44.1"},"uses_additional_data":false,"paper_date":null,"paper":"/paper/urvos-unified-referring-video-object","paper_url":"https://www.ecva.net/papers/eccv_2020/papers_ECCV/html/2327_ECCV_2020_paper.php","paper_title":"URVOS: Unified Referring Video Object Segmentation Network with a Large-Scale Benchmark","code":"https://github.com/skynbe/Refer-Youtube-VOS","n_code_links":1,"syntology":null},{"rank_in_archive_order":16,"model":"Khoreva et al.","metrics":{"J&F 1st frame":"39.3","J&F Full video":"37.1"},"uses_additional_data":false,"paper_date":"2018-03-21","paper":"/paper/video-object-segmentation-with-language","paper_url":"http://arxiv.org/abs/1803.08006v3","paper_title":"Video Object Segmentation with Language Referring Expressions","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":17,"model":"UniVS(Swin-L)","metrics":{"J&F 1st frame":"59.4?","J&F Full video":"59.4"},"uses_additional_data":true,"paper_date":"2024-02-28","paper":"/paper/univs-unified-and-universal-video","paper_url":"https://arxiv.org/abs/2402.18115v2","paper_title":"UniVS: Unified and Universal Video Segmentation with Prompts as Queries","code":"https://github.com/minghanli/univs","n_code_links":1,"syntology":{"n_ran":12,"n_unverified":2,"n_samples":14,"n_pointer_only_licence":14}},{"rank_in_archive_order":18,"model":"VATEX","metrics":{"J&F score":"65.4"},"uses_additional_data":false,"paper_date":"2024-04-12","paper":"/paper/improving-referring-image-segmentation-using","paper_url":"https://arxiv.org/abs/2404.08590v2","paper_title":"Vision-Aware Text Features in Referring Image Segmentation: From Object Understanding to Context Understanding","code":"https://github.com/nero1342/VATEX","n_code_links":1,"syntology":null}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":8,"rows_with_any_sample_ran":8,"distinct_papers_with_graph_line":8,"distinct_papers_with_any_sample_ran":8,"samples_over_distinct_papers":{"n_ran":61,"n_unverified":22,"n_samples":83,"n_pointer_only_licence":47,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":61,"n_unverified":22,"n_samples":83,"n_pointer_only_licence":47,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}