{"url":"/sota/referring-expression-segmentation-on-refcoco-9","task":{"name":"Referring Expression Segmentation","url":"/task/referring-expression-segmentation","note":null},"dataset":{"name":"RefCOCO testB","url":"/dataset/refcoco"},"category":"Computer Vision","categories":["Computer Vision"],"category_note":null,"description":"The task aims at labeling the pixels of an image or video that represent an object instance referred by a linguistic expression. In particular, the referring expression (RE) must allow the identification of an individual object in a discourse or scene (the referent). REs unambiguously identify the target instance.","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["Overall IoU","mIoU","Mean IoU"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"Overall IoU":"higher","mIoU":null,"Mean IoU":"higher"}},"counts":{"rows":13,"rows_with_code":12,"rows_with_paper_page":13,"rows_dated":13,"rows_using_additional_data":3},"rows":[{"rank_in_archive_order":1,"model":"HyperSeg","metrics":{"Overall IoU":"83.4"},"uses_additional_data":true,"paper_date":"2024-11-26","paper":"/paper/hyperseg-towards-universal-visual","paper_url":"https://arxiv.org/abs/2411.17606v2","paper_title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","code":"https://github.com/congvvc/HyperSeg","n_code_links":1,"syntology":{"n_ran":7,"n_unverified":10,"n_samples":17,"n_pointer_only_licence":0}},{"rank_in_archive_order":2,"model":"DeRIS-L","metrics":{"Mean IoU":"84.52","Overall IoU":"82.87"},"uses_additional_data":false,"paper_date":"2025-07-02","paper":"/paper/deris-decoupling-perception-and-cognition-for","paper_url":"https://arxiv.org/abs/2507.01738v1","paper_title":"DeRIS: Decoupling Perception and Cognition for Enhanced Referring Image Segmentation through Loopback Synergy","code":"https://github.com/Dmmm1997/DeRIS","n_code_links":1,"syntology":null},{"rank_in_archive_order":3,"model":"MLCD-Seg-7B","metrics":{"Overall IoU":"81.5"},"uses_additional_data":true,"paper_date":"2024-07-24","paper":"/paper/multi-label-cluster-discrimination-for-visual","paper_url":"https://arxiv.org/abs/2407.17331v2","paper_title":"Multi-label Cluster Discrimination for Visual Representation Learning","code":"https://github.com/deepglint/unicom","n_code_links":1,"syntology":{"n_ran":7,"n_unverified":4,"n_samples":11,"n_pointer_only_licence":0}},{"rank_in_archive_order":4,"model":"EVF-SAM","metrics":{"Overall IoU":"80.2"},"uses_additional_data":true,"paper_date":"2024-06-28","paper":"/paper/evf-sam-early-vision-language-fusion-for-text","paper_url":"https://arxiv.org/abs/2406.20076v4","paper_title":"EVF-SAM: Early Vision-Language Fusion for Text-Prompted Segment Anything Model","code":"https://github.com/hustvl/evf-sam","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":0,"n_samples":3,"n_pointer_only_licence":2}},{"rank_in_archive_order":5,"model":"DETRIS","metrics":{"Overall IoU":"79.0"},"uses_additional_data":false,"paper_date":"2025-01-15","paper":"/paper/densely-connected-parameter-efficient-tuning","paper_url":"https://arxiv.org/abs/2501.08580v1","paper_title":"Densely Connected Parameter-Efficient Tuning for Referring Image Segmentation","code":"https://github.com/jiaqihuang01/detris","n_code_links":1,"syntology":{"n_ran":7,"n_unverified":10,"n_samples":17,"n_pointer_only_licence":0}},{"rank_in_archive_order":6,"model":"C3VG","metrics":{"Overall IoU":"77.86"},"uses_additional_data":false,"paper_date":"2025-01-12","paper":"/paper/multi-task-visual-grounding-with-coarse-to","paper_url":"https://arxiv.org/abs/2501.06710v1","paper_title":"Multi-task Visual Grounding with Coarse-to-Fine Consistency Constraints","code":"https://github.com/dmmm1997/c3vg","n_code_links":1,"syntology":{"n_ran":5,"n_unverified":8,"n_samples":13,"n_pointer_only_licence":0}},{"rank_in_archive_order":7,"model":"MaskRIS (Swin-B, combined DB)","metrics":{"Overall IoU":"75.1"},"uses_additional_data":false,"paper_date":"2024-11-28","paper":"/paper/maskris-semantic-distortion-aware-data","paper_url":"https://arxiv.org/abs/2411.19067v1","paper_title":"MaskRIS: Semantic Distortion-aware Data Augmentation for Referring Image Segmentation","code":"https://github.com/naver-ai/maskris","n_code_links":1,"syntology":null},{"rank_in_archive_order":8,"model":"MaskRIS (Swin-B)","metrics":{"Mean IoU":"76.06","Overall IoU":"73.96"},"uses_additional_data":false,"paper_date":"2024-11-28","paper":"/paper/maskris-semantic-distortion-aware-data","paper_url":"https://arxiv.org/abs/2411.19067v1","paper_title":"MaskRIS: Semantic Distortion-aware Data Augmentation for Referring Image Segmentation","code":"https://github.com/naver-ai/maskris","n_code_links":1,"syntology":null},{"rank_in_archive_order":9,"model":"EVP","metrics":{"Overall IoU":"72.94"},"uses_additional_data":false,"paper_date":"2023-12-13","paper":"/paper/evp-enhanced-visual-perception-using-inverse","paper_url":"https://arxiv.org/abs/2312.08548v1","paper_title":"EVP: Enhanced Visual Perception using Inverse Multi-Attentive Feature Refinement and Regularized Image-Text Alignment","code":"https://github.com/lavreniuk/evp","n_code_links":1,"syntology":null},{"rank_in_archive_order":10,"model":"MagNet","metrics":{"Overall IoU":"71.05"},"uses_additional_data":false,"paper_date":"2023-12-19","paper":"/paper/mask-grounding-for-referring-image","paper_url":"https://arxiv.org/abs/2312.12198v2","paper_title":"Mask Grounding for Referring Image Segmentation","code":"https://github.com/yxchng/mask-grounding","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":3,"n_samples":13,"n_pointer_only_licence":13}},{"rank_in_archive_order":11,"model":"SafaRi","metrics":{"Overall IoU":"70.71"},"uses_additional_data":false,"paper_date":"2024-07-02","paper":"/paper/safari-adaptive-sequence-transformer-for","paper_url":"https://arxiv.org/abs/2407.02389v1","paper_title":"SafaRi:Adaptive Sequence Transformer for Weakly Supervised Referring Expression Segmentation","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":12,"model":"SeqTR","metrics":{"Overall IoU":" 64.12"},"uses_additional_data":false,"paper_date":"2022-03-30","paper":"/paper/seqtr-a-simple-yet-universal-network-for","paper_url":"https://arxiv.org/abs/2203.16265v2","paper_title":"SeqTR: A Simple yet Universal Network for Visual Grounding","code":"https://github.com/seanzhuh/seqtr","n_code_links":3,"syntology":{"n_ran":1,"n_unverified":3,"n_samples":4,"n_pointer_only_licence":0}},{"rank_in_archive_order":13,"model":"VATEX","metrics":{"mIoU":"75.64"},"uses_additional_data":false,"paper_date":"2024-04-12","paper":"/paper/improving-referring-image-segmentation-using","paper_url":"https://arxiv.org/abs/2404.08590v2","paper_title":"Vision-Aware Text Features in Referring Image Segmentation: From Object Understanding to Context Understanding","code":"https://github.com/nero1342/VATEX","n_code_links":1,"syntology":null}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":7,"rows_with_any_sample_ran":7,"distinct_papers_with_graph_line":7,"distinct_papers_with_any_sample_ran":7,"samples_over_distinct_papers":{"n_ran":40,"n_unverified":38,"n_samples":78,"n_pointer_only_licence":15,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":40,"n_unverified":38,"n_samples":78,"n_pointer_only_licence":15,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}