{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/referring-expression-segmentation/papers/2","list_of":"/task/referring-expression-segmentation","task":"Referring Expression Segmentation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,145],"of":145,"counts":{"archive_papers_tagged":145,"with_a_code_link":97,"where_syntology_ran_a_sample":48,"not_listed_spam_title":0,"listed":145,"listed_where_code_ran":48,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":42,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":42,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/referring-expression-segmentation","prev":"/task/referring-expression-segmentation","next":null,"papers":[{"url":null,"slug":"3drest-a-strong-baseline-for-semi-supervised","title":"3DResT: A Strong Baseline for Semi-Supervised 3D Referring Expression Segmentation","date":"2025-04-17","arxiv_id":"2504.12599","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-unified-referring-expression","title":"Towards Unified Referring Expression Segmentation Across Omni-Level Visual Target Granularities","date":"2025-04-02","arxiv_id":"2504.01954","repositories_listed":0,"syntology":null},{"url":"/paper/referdino-referring-video-object-segmentation","slug":"referdino-referring-video-object-segmentation","title":"ReferDINO: Referring Video Object Segmentation with Visual Grounding Foundations","date":"2025-01-24","arxiv_id":"2501.14607","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-alignment-enhanced-adaptive","title":"Hierarchical Alignment-enhanced Adaptive Grounding Network for Generalized Referring Expression Comprehension","date":"2025-01-02","arxiv_id":"2501.01416","repositories_listed":0,"syntology":null},{"url":null,"slug":"dvin-dynamic-visual-routing-network-for","title":"DViN: Dynamic Visual Routing Network for Weakly Supervised Referring Expression Comprehension","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"task-aware-cross-modal-feature-refinement","title":"Task-aware Cross-modal Feature Refinement Transformer with Large Language Models for Visual Grounding","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"instance-aware-generalized-referring","title":"Instance-Aware Generalized Referring Expression Segmentation","date":"2024-11-22","arxiv_id":"2411.15087","repositories_listed":0,"syntology":null},{"url":null,"slug":"segllm-multi-round-reasoning-segmentation","title":"SegLLM: Multi-round Reasoning Segmentation","date":"2024-10-24","arxiv_id":"2410.18923","repositories_listed":0,"syntology":null},{"url":"/paper/safari-adaptive-sequence-transformer-for","slug":"safari-adaptive-sequence-transformer-for","title":"SafaRi:Adaptive Sequence Transformer for Weakly Supervised Referring Expression Segmentation","date":"2024-07-02","arxiv_id":"2407.02389","repositories_listed":0,"syntology":null},{"url":"/paper/groprompt-efficient-grounded-prompting-and","slug":"groprompt-efficient-grounded-prompting-and","title":"GroPrompt: Efficient Grounded Prompting and Adaptation for Referring Video Object Segmentation","date":"2024-06-18","arxiv_id":"2406.12834","repositories_listed":0,"syntology":null},{"url":null,"slug":"goi-find-3d-gaussians-of-interest-with-an","title":"GOI: Find 3D Gaussians of Interest with an Optimizable Open-vocabulary Semantic-space Hyperplane","date":"2024-05-27","arxiv_id":"2405.17596","repositories_listed":0,"syntology":null},{"url":"/paper/driving-referring-video-object-segmentation","slug":"driving-referring-video-object-segmentation","title":"Harnessing Vision-Language Pretrained Models with Temporal-Aware Adaptation for Referring Video Object Segmentation","date":"2024-05-17","arxiv_id":"2405.10610","repositories_listed":0,"syntology":null},{"url":"/paper/groundhog-grounding-large-language-models-to","slug":"groundhog-grounding-large-language-models-to","title":"GROUNDHOG: Grounding Large Language Models to Holistic Segmentation","date":"2024-02-26","arxiv_id":"2402.16846","repositories_listed":0,"syntology":null},{"url":null,"slug":"resmatch-referring-expression-segmentation-in","title":"RESMatch: Referring Expression Segmentation in a Semi-Supervised Manner","date":"2024-02-08","arxiv_id":"2402.05589","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizable-entity-grounding-via-assistance","title":"Generalizable Entity Grounding via Assistance of Large Language Model","date":"2024-02-04","arxiv_id":"2402.02555","repositories_listed":0,"syntology":null},{"url":"/paper/lyrics-boosting-fine-grained-language-vision","slug":"lyrics-boosting-fine-grained-language-vision","title":"Lyrics: Boosting Fine-grained Language-Vision Alignment and Comprehension via Semantic-aware Visual Objects","date":"2023-12-08","arxiv_id":"2312.05278","repositories_listed":0,"syntology":null},{"url":null,"slug":"instructseq-unifying-vision-tasks-with","title":"InstructSeq: Unifying Vision Tasks with Instruction-conditioned Multi-modal Sequence Generation","date":"2023-11-30","arxiv_id":"2311.18835","repositories_listed":0,"syntology":null},{"url":null,"slug":"clipunetr-assisting-human-robot-interface-for","title":"CLIPUNetr: Assisting Human-robot Interface for Uncalibrated Visual Servoing Control with CLIP-driven Referring Expression Segmentation","date":"2023-09-17","arxiv_id":"2309.09183","repositories_listed":0,"syntology":null},{"url":null,"slug":"eavl-explicitly-align-vision-and-language-for","title":"EAVL: Explicitly Align Vision and Language for Referring Image Segmentation","date":"2023-08-18","arxiv_id":"2308.09779","repositories_listed":0,"syntology":null},{"url":"/paper/epcformer-expression-prompt-collaboration","slug":"epcformer-expression-prompt-collaboration","title":"Expression Prompt Collaboration Transformer for Universal Referring Video Object Segmentation","date":"2023-08-08","arxiv_id":"2308.04162","repositories_listed":0,"syntology":null},{"url":null,"slug":"wico-win-win-cooperation-of-bottom-up-and-top","title":"WiCo: Win-win Cooperation of Bottom-up and Top-down Referring Image Segmentation","date":"2023-06-19","arxiv_id":"2306.10750","repositories_listed":0,"syntology":null},{"url":null,"slug":"risclip-referring-image-segmentation","title":"Extending CLIP's Image-Text Alignment to Referring Image Segmentation","date":"2023-06-14","arxiv_id":"2306.08498","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-compositional-referring-expression","title":"Meta Compositional Referring Expression Segmentation","date":"2023-04-10","arxiv_id":"2304.04415","repositories_listed":0,"syntology":null},{"url":"/paper/segment-every-reference-object-in-spatial-and","slug":"segment-every-reference-object-in-spatial-and","title":"Segment Every Reference Object in Spatial and Temporal Spaces","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-and-weakly-supervised-referring","title":"Fully and Weakly Supervised Referring Expression Segmentation with End-to-End Learning","date":"2022-12-17","arxiv_id":"2212.10278","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-mutual-supervision-framework-for","title":"A Unified Mutual Supervision Framework for Referring Expression Segmentation and Generation","date":"2022-11-15","arxiv_id":"2211.07919","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-segmentation-of-referring","title":"Weakly-supervised segmentation of referring expressions","date":"2022-05-10","arxiv_id":"2205.04725","repositories_listed":0,"syntology":null},{"url":null,"slug":"restr-convolution-free-referring-image","title":"ReSTR: Convolution-free Referring Image Segmentation Using Transformers","date":"2022-03-31","arxiv_id":"2203.16768","repositories_listed":0,"syntology":null},{"url":"/paper/deeply-interleaved-two-stream-encoder-for","slug":"deeply-interleaved-two-stream-encoder-for","title":"Deeply Interleaved Two-Stream Encoder for Referring Video Segmentation","date":"2022-03-30","arxiv_id":"2203.15969","repositories_listed":0,"syntology":null},{"url":"/paper/multi-level-representation-learning-with","slug":"multi-level-representation-learning-with","title":"Multi-Level Representation Learning With Semantic Alignment for Referring Video Object Segmentation","date":"2022-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/hierarchical-interaction-network-for-video","slug":"hierarchical-interaction-network-for-video","title":"Hierarchical interaction network for video object segmentation from referring expressions","date":"2021-11-22","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/mail-a-unified-mask-image-language-trimodal","slug":"mail-a-unified-mask-image-language-trimodal","title":"MaIL: A Unified Mask-Image-Language Trimodal Network for Referring Image Segmentation","date":"2021-11-21","arxiv_id":"2111.10747","repositories_listed":0,"syntology":null},{"url":"/paper/collaborative-spatial-temporal-modeling-for","slug":"collaborative-spatial-temporal-modeling-for","title":"Collaborative Spatial-Temporal Modeling for Language-Queried Video Actor Segmentation","date":"2021-05-14","arxiv_id":"2105.06818","repositories_listed":0,"syntology":null},{"url":"/paper/clawcranenet-leveraging-object-level-relation","slug":"clawcranenet-leveraging-object-level-relation","title":"ClawCraneNet: Leveraging Object-level Relation for Text-based Video Segmentation","date":"2021-03-19","arxiv_id":"2103.10702","repositories_listed":0,"syntology":null},{"url":"/paper/referring-segmentation-in-images-and-videos","slug":"referring-segmentation-in-images-and-videos","title":"Referring Segmentation in Images and Videos with Cross-Modal Self-Attention Network","date":"2021-02-09","arxiv_id":"2102.04762","repositories_listed":0,"syntology":null},{"url":"/paper/actor-and-action-modular-network-for-text","slug":"actor-and-action-modular-network-for-text","title":"Actor and Action Modular Network for Text-based Video Segmentation","date":"2020-11-02","arxiv_id":"2011.00786","repositories_listed":0,"syntology":null},{"url":"/paper/polar-relative-positional-encoding-for-video","slug":"polar-relative-positional-encoding-for-video","title":"Polar Relative Positional Encoding for Video-Language Segmentation","date":"2020-07-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/bi-directional-relationship-inferring-network","slug":"bi-directional-relationship-inferring-network","title":"Bi-Directional Relationship Inferring Network for Referring Image Segmentation","date":"2020-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/visual-textual-capsule-routing-for-text-based","slug":"visual-textual-capsule-routing-for-text-based","title":"Visual-Textual Capsule Routing for Text-Based Video Segmentation","date":"2020-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"referring-image-segmentation-by-generative","title":"Referring Image Segmentation by Generative Adversarial Learning","date":"2020-04-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/context-modulated-dynamic-networks-for-actor","slug":"context-modulated-dynamic-networks-for-actor","title":"Context Modulated Dynamic Networks for Actor and Action Video Segmentation with Language Queries","date":"2020-04-03","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"recurrent-instance-segmentation-using","title":"Recurrent Instance Segmentation using Sequences of Referring Expressions","date":"2019-11-05","arxiv_id":"1911.02103","repositories_listed":0,"syntology":null},{"url":"/paper/see-through-text-grouping-for-referring-image","slug":"see-through-text-grouping-for-referring-image","title":"See-Through-Text Grouping for Referring Image Segmentation","date":"2019-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/video-object-segmentation-with-language","slug":"video-object-segmentation-with-language","title":"Video Object Segmentation with Language Referring Expressions","date":"2018-03-21","arxiv_id":"1803.08006","repositories_listed":0,"syntology":null},{"url":"/paper/tracking-by-natural-language-specification","slug":"tracking-by-natural-language-specification","title":"Tracking by Natural Language Specification","date":"2017-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"4c8f6a60731c1fb3acfabfc99dc9bac84728ae440a597e026f55eaf1f9664222","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}