{"url":"/dataset/google-refexp","name":"Google Refexp","full_name":null,"description_markdown":"A new large-scale dataset for referring expressions, based on MS-COCO.\r\n\r\nSource: [Generation and Comprehension of Unambiguous Object Descriptions](/paper/generation-and-comprehension-of-unambiguous)","description_withheld":null,"homepage":"https://github.com/mjhucla/Google_Refexp_toolbox","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/generation-and-comprehension-of-unambiguous","title":"Generation and Comprehension of Unambiguous Object Descriptions","first_author":"Junhua Mao","url":null},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[],"tasks":[{"name":"Image Captioning","url":"/task/image-captioning","datasets_with_task":"/datasets/task/image-captioning"},{"name":"Referring Expression Segmentation","url":"/task/referring-expression-segmentation","datasets_with_task":"/datasets/task/referring-expression-segmentation"},{"name":"Referring Expression Comprehension","url":"/task/referring-expression-comprehension","datasets_with_task":"/datasets/task/referring-expression-comprehension"},{"name":"Zero-Shot Region Description","url":"/task/zero-shot-region-description","datasets_with_task":"/datasets/task/zero-shot-region-description"},{"name":"Natural Language Visual Grounding","url":"/task/natural-language-visual-grounding","datasets_with_task":"/datasets/task/natural-language-visual-grounding"},{"name":"Deep Attention","url":"/task/deep-attention","datasets_with_task":"/datasets/task/deep-attention"}],"languages":[],"variants":["Google Refexp","RefCOCOg-val","RefCOCOg-test"],"data_loaders":[{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/gref","frameworks":["tf","jax"]},{"repo":"https://github.com/mjhucla/Google_Refexp_toolbox","url":"https://github.com/mjhucla/Google_Refexp_toolbox","frameworks":[]}],"num_papers_in_archive":46,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/referring-expression-segmentation-on-refcocog","task":"Referring Expression Segmentation","dataset_variant":"RefCOCOg-val","rows":23,"metrics":["Overall IoU","Mean IoU","IoU","mIoU"],"first_row_in_archive_order":{"model":"MLCD-Seg-7B","paper":"/paper/multi-label-cluster-discrimination-for-visual","metrics":{"Overall IoU":"79.9"},"code_links":[{"title":"deepglint/unicom","url":"https://github.com/deepglint/unicom"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/referring-expression-segmentation-on-refcocog-1","task":"Referring Expression Segmentation","dataset_variant":"RefCOCOg-test","rows":18,"metrics":["Overall IoU","Mean IoU","mIoU"],"first_row_in_archive_order":{"model":"UniLSeg-100","paper":"/paper/universal-segmentation-at-arbitrary","metrics":{"Overall IoU":"80.54"},"code_links":[{"title":"yongliu20/UniLSeg","url":"https://github.com/yongliu20/UniLSeg"},{"title":"workforai/UniLSeg","url":"https://github.com/workforai/UniLSeg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/deris-decoupling-perception-and-cognition-for","title":"DeRIS: Decoupling Perception and Cognition for Enhanced Referring Image Segmentation through Loopback Synergy","date":"2025-07-02","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/segagent-exploring-pixel-understanding","title":"SegAgent: Exploring Pixel Understanding Capabilities in MLLMs by Imitating Human Annotator Trajectories","date":"2025-03-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/densely-connected-parameter-efficient-tuning","title":"Densely Connected Parameter-Efficient Tuning for Referring Image Segmentation","date":"2025-01-15","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":17,"samples_ran":7,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-task-visual-grounding-with-coarse-to","title":"Multi-task Visual Grounding with Coarse-to-Fine Consistency Constraints","date":"2025-01-12","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":13,"samples_ran":5,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/maskris-semantic-distortion-aware-data","title":"MaskRIS: Semantic Distortion-aware Data Augmentation for Referring Image Segmentation","date":"2024-11-28","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/hyperseg-towards-universal-visual","title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","date":"2024-11-26","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":17,"samples_ran":7,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-label-cluster-discrimination-for-visual","title":"Multi-label Cluster Discrimination for Visual Representation Learning","date":"2024-07-24","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":11,"samples_ran":7,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/safari-adaptive-sequence-transformer-for","title":"SafaRi:Adaptive Sequence Transformer for Weakly Supervised Referring Expression Segmentation","date":"2024-07-02","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/evf-sam-early-vision-language-fusion-for-text","title":"EVF-SAM: Early Vision-Language Fusion for Text-Prompted Segment Anything Model","date":"2024-06-28","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-referring-image-segmentation-using","title":"Vision-Aware Text Features in Referring Image Segmentation: From Object Understanding to Context Understanding","date":"2024-04-12","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/groundhog-grounding-large-language-models-to","title":"GROUNDHOG: Grounding Large Language Models to Holistic Segmentation","date":"2024-02-26","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/mask-grounding-for-referring-image","title":"Mask Grounding for Referring Image Segmentation","date":"2023-12-19","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":13,"samples_ran":10,"samples_unverified":3,"pointer_only_for_licence":13,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/general-object-foundation-model-for-images","title":"General Object Foundation Model for Images and Videos at Scale","date":"2023-12-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":13,"samples_ran":8,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/universal-segmentation-at-arbitrary","title":"Universal Segmentation at Arbitrary Granularity with Language Instruction","date":"2023-12-04","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":16,"samples_ran":13,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/polyformer-referring-image-segmentation-as","title":"PolyFormer: Referring Image Segmentation as Sequential Polygon Generation","date":"2023-02-14","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/generalized-decoding-for-pixel-image-and","title":"Generalized Decoding for Pixel, Image, and Language","date":"2022-12-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vlt-vision-language-transformer-and-query","title":"VLT: Vision-Language Transformer and Query Generation for Referring Segmentation","date":"2022-10-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lavt-language-aware-vision-transformer-for","title":"LAVT: Language-Aware Vision Transformer for Referring Image Segmentation","date":"2021-12-04","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/vision-language-transformer-and-query","title":"Vision-Language Transformer and Query Generation for Referring Segmentation","date":"2021-08-12","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/comprehensive-multi-modal-interactions-for","title":"Comprehensive Multi-Modal Interactions for Referring Image Segmentation","date":"2021-04-21","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":10,"samples_harvested":115,"samples_ran":68,"samples_unverified":47,"pointer_only_for_licence":21,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}