{"url":"/task/generalized-referring-expression-segmentation","name":"Generalized Referring Expression Segmentation","slug":"generalized-referring-expression-segmentation","description_markdown":"Generalized Referring Expression Segmentation (GRES), introduced by [Liu et al in CVPR 2023](https://henghuiding.github.io/GRES/), allows expressions indicating any number of target objects. GRES takes an image and a referring expression as input, and requires mask prediction of the target object(s).","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":15,"papers_with_code":10,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/generalized-referring-expression-segmentation","slug":"generalized-referring-expression-segmentation","dataset":"gRefCOCO","dataset_url":"/dataset/grefcoco","rows_in_archive":13,"metrics":["gIoU","cIoU"],"first_row_in_archive_order":{"model":"DeRIS-L","paper_title":"DeRIS: Decoupling Perception and Cognition for Enhanced Referring Image Segmentation through Loopback Synergy","paper_url":"/paper/deris-decoupling-perception-and-cognition-for","paper_date":"2025-07-02","arxiv_id":"2507.01738","code_links":[{"title":"Dmmm1997/DeRIS","url":"https://github.com/Dmmm1997/DeRIS"}],"syntology":null}}],"datasets":[{"url":"/dataset/grefcoco","name":"gRefCOCO","full_name":"","num_papers_in_archive":43}],"subtasks":[],"parent_tasks":[{"url":"/task/referring-expression-segmentation","name":"Referring Expression Segmentation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":10,"of":10,"tagged_in_all":15,"items":[{"url":"/paper/hdc-hierarchical-semantic-decoding-with-1","title":"CoHD: A Counting-Aware Hierarchical Decoding Framework for Generalized Referring Expression Segmentation","date":"2024-05-24","arxiv_id":"2405.15658","repositories_listed":2,"syntology":null},{"url":"/paper/gres-generalized-referring-expression-1","title":"GRES: Generalized Referring Expression Segmentation","date":"2023-06-01","arxiv_id":"2306.00968","repositories_listed":2,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/deris-decoupling-perception-and-cognition-for","title":"DeRIS: Decoupling Perception and Cognition for Enhanced Referring Image Segmentation through Loopback Synergy","date":"2025-07-02","arxiv_id":"2507.01738","repositories_listed":1,"syntology":null},{"url":"/paper/bring-adaptive-binding-prototypes-to","title":"Bring Adaptive Binding Prototypes to Generalized Referring Expression Segmentation","date":"2024-05-24","arxiv_id":"2405.15169","repositories_listed":1,"syntology":null},{"url":"/paper/psalm-pixelwise-segmentation-with-large-multi","title":"PSALM: Pixelwise SegmentAtion with Large Multi-Modal Model","date":"2024-03-21","arxiv_id":"2403.14598","repositories_listed":1,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/gsva-generalized-segmentation-via-multimodal","title":"GSVA: Generalized Segmentation via Multimodal Large Language Models","date":"2023-12-15","arxiv_id":"2312.10103","repositories_listed":1,"syntology":{"n":10,"n_ran":2,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/lavt-language-aware-vision-transformer-for","title":"LAVT: Language-Aware Vision Transformer for Referring Image Segmentation","date":"2021-12-04","arxiv_id":"2112.02244","repositories_listed":1,"syntology":null},{"url":"/paper/cris-clip-driven-referring-image-segmentation","title":"CRIS: CLIP-Driven Referring Image Segmentation","date":"2021-11-30","arxiv_id":"2111.15174","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/vision-language-transformer-and-query","title":"Vision-Language Transformer and Query Generation for Referring Segmentation","date":"2021-08-12","arxiv_id":"2108.05565","repositories_listed":1,"syntology":null},{"url":"/paper/mattnet-modular-attention-network-for","title":"MAttNet: Modular Attention Network for Referring Expression Comprehension","date":"2018-01-24","arxiv_id":"1801.08186","repositories_listed":1,"syntology":null}],"syntology_records":4,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}