{"url":"/dataset/grefcoco","name":"gRefCOCO","full_name":null,"description_markdown":"gRefCOCO is the first large-scale Generalized Referring Expression Segmentation dataset that contains multi-target, no-target, and single-target expressions.","description_withheld":null,"homepage":"https://henghuiding.github.io/GRES/","introduced_date":"2023-01-01","introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Generalized Referring Expression Segmentation","url":"/task/generalized-referring-expression-segmentation","datasets_with_task":"/datasets/task/generalized-referring-expression-segmentation"},{"name":"Generalized Referring Expression Comprehension","url":"/task/generalized-referring-expression","datasets_with_task":"/datasets/task/generalized-referring-expression"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["gRefCOCO"],"data_loaders":[],"num_papers_in_archive":43,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/generalized-referring-expression-segmentation","task":"Generalized Referring Expression Segmentation","dataset_variant":"gRefCOCO","rows":13,"metrics":["gIoU","cIoU"],"first_row_in_archive_order":{"model":"DeRIS-L","paper":"/paper/deris-decoupling-perception-and-cognition-for","metrics":{"cIoU":"72.00","gIoU":"77.67"},"code_links":[{"title":"Dmmm1997/DeRIS","url":"https://github.com/Dmmm1997/DeRIS"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/generalized-referring-expression","task":"Generalized Referring Expression Comprehension","dataset_variant":"gRefCOCO","rows":5,"metrics":["Precision@(F1=1, IoU≥0.5)","N-acc."],"first_row_in_archive_order":{"model":"SimVG-DB","paper":"/paper/simvg-a-simple-framework-for-visual-grounding","metrics":{"N-acc.":"54.7","Precision@(F1=1, IoU≥0.5)":"62.1"},"code_links":[{"title":"dmmm1997/simvg","url":"https://github.com/dmmm1997/simvg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/deris-decoupling-perception-and-cognition-for","title":"DeRIS: Decoupling Perception and Cognition for Enhanced Referring Image Segmentation through Loopback Synergy","date":"2025-07-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/simvg-a-simple-framework-for-visual-grounding","title":"SimVG: A Simple Framework for Visual Grounding with Decoupled Multi-modal Fusion","date":"2024-09-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":7,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hdc-hierarchical-semantic-decoding-with-1","title":"CoHD: A Counting-Aware Hierarchical Decoding Framework for Generalized Referring Expression Segmentation","date":"2024-05-24","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/bring-adaptive-binding-prototypes-to","title":"Bring Adaptive Binding Prototypes to Generalized Referring Expression Segmentation","date":"2024-05-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/groundhog-grounding-large-language-models-to","title":"GROUNDHOG: Grounding Large Language Models to Holistic Segmentation","date":"2024-02-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/gsva-generalized-segmentation-via-multimodal","title":"GSVA: Generalized Segmentation via Multimodal Large Language Models","date":"2023-12-15","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":2,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gres-generalized-referring-expression-1","title":"GRES: Generalized Referring Expression Segmentation","date":"2023-06-01","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/universal-instance-perception-as-object","title":"Universal Instance Perception as Object Discovery and Retrieval","date":"2023-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lavt-language-aware-vision-transformer-for","title":"LAVT: Language-Aware Vision Transformer for Referring Image Segmentation","date":"2021-12-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/cris-clip-driven-referring-image-segmentation","title":"CRIS: CLIP-Driven Referring Image Segmentation","date":"2021-11-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":7,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vision-language-transformer-and-query","title":"Vision-Language Transformer and Query Generation for Referring Segmentation","date":"2021-08-12","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/mdetr-modulated-detection-for-end-to-end","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-04-26","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":6,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/locate-then-segment-a-strong-pipeline-for","title":"Locate then Segment: A Strong Pipeline for Referring Image Segmentation","date":"2021-03-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/multi-task-collaborative-network-for-joint","title":"Multi-task Collaborative Network for Joint Referring Expression Comprehension and Segmentation","date":"2020-03-19","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":0,"samples_unverified":13,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mattnet-modular-attention-network-for","title":"MAttNet: Modular Attention Network for Referring Expression Comprehension","date":"2018-01-24","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":63,"samples_ran":27,"samples_unverified":36,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}