{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/referring-expression/papers/2","list_of":"/task/referring-expression","task":"Referring Expression","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":364,"counts":{"archive_papers_tagged":364,"with_a_code_link":166,"where_syntology_ran_a_sample":57,"not_listed_spam_title":0,"listed":364,"listed_where_code_ran":57,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":51,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":51,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/referring-expression","prev":"/task/referring-expression","next":"/task/referring-expression/papers/3","papers":[{"url":"/paper/advancing-referring-expression-segmentation","slug":"advancing-referring-expression-segmentation","title":"Advancing Referring Expression Segmentation Beyond Single Image","date":"2023-05-21","arxiv_id":"2305.12452","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-referring-image-segmentation-with","slug":"zero-shot-referring-image-segmentation-with","title":"Zero-shot Referring Image Segmentation with Global-Local Context Features","date":"2023-03-31","arxiv_id":"2303.17811","repositories_listed":1,"syntology":null},{"url":"/paper/ns3d-neuro-symbolic-grounding-of-3d-objects","slug":"ns3d-neuro-symbolic-grounding-of-3d-objects","title":"NS3D: Neuro-Symbolic Grounding of 3D Objects and Relations","date":"2023-03-23","arxiv_id":"2303.13483","repositories_listed":1,"syntology":null},{"url":"/paper/universal-instance-perception-as-object","slug":"universal-instance-perception-as-object","title":"Universal Instance Perception as Object Discovery and Retrieval","date":"2023-03-12","arxiv_id":"2303.06674","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/universal-instance-perception-as-object#ran","syntology_url":"https://syntology.ai/paper/2303.06674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06674"}},"official":{"repos":["MasterBin-IIAU/UNINEXT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ck-transformer-commonsense-knowledge-enhanced","slug":"ck-transformer-commonsense-knowledge-enhanced","title":"CK-Transformer: Commonsense Knowledge Enhanced Transformers for Referring Expression Comprehension","date":"2023-02-17","arxiv_id":"2302.09027","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-segment-every-referring-object","slug":"learning-to-segment-every-referring-object","title":"Learning To Segment Every Referring Object Point by Point","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/layout-aware-dreamer-for-embodied-referring","slug":"layout-aware-dreamer-for-embodied-referring","title":"Layout-aware Dreamer for Embodied Referring Expression Grounding","date":"2022-11-30","arxiv_id":"2212.00171","repositories_listed":1,"syntology":null},{"url":"/paper/scene-text-oriented-reffering-expression","slug":"scene-text-oriented-reffering-expression","title":"Scene-Text Oriented Reffering Expression Comprehension","date":"2022-11-04","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/toist-task-oriented-instance-segmentation","slug":"toist-task-oriented-instance-segmentation","title":"TOIST: Task Oriented Instance Segmentation Transformer with Noun-Pronoun Distillation","date":"2022-10-19","arxiv_id":"2210.10775","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/toist-task-oriented-instance-segmentation#ran","syntology_url":"https://syntology.ai/paper/2210.10775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10775"}},"official":{"repos":["air-discover/toist"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/sqa3d-situated-question-answering-in-3d","slug":"sqa3d-situated-question-answering-in-3d","title":"SQA3D: Situated Question Answering in 3D Scenes","date":"2022-10-14","arxiv_id":"2210.07474","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":6,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sqa3d-situated-question-answering-in-3d#ran","syntology_url":"https://syntology.ai/paper/2210.07474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07474"}},"official":{"repos":["SilongYong/SQA3D"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/volta-vision-language-transformer-with-weakly","slug":"volta-vision-language-transformer-with-weakly","title":"VoLTA: Vision-Language Transformer with Weakly-Supervised Local-Feature Alignment","date":"2022-10-09","arxiv_id":"2210.04135","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/volta-vision-language-transformer-with-weakly#ran","syntology_url":"https://syntology.ai/paper/2210.04135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.04135"}},"official":{"repos":["ShramanPramanick/VoLTA"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/a-hybrid-compositional-reasoning-approach-for","slug":"a-hybrid-compositional-reasoning-approach-for","title":"Enhancing Interpretability and Interactivity in Robot Manipulation: A Neurosymbolic Approach","date":"2022-10-03","arxiv_id":"2210.00858","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-modulated-detection-transformer-as","slug":"exploring-modulated-detection-transformer-as","title":"Exploring Modulated Detection Transformer as a Tool for Action Recognition in Videos","date":"2022-09-21","arxiv_id":"2209.10126","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-evaluate-performance-of-multi","slug":"learning-to-evaluate-performance-of-multi","title":"Learning to Evaluate Performance of Multi-modal Semantic Localization","date":"2022-09-14","arxiv_id":"2209.06515","repositories_listed":1,"syntology":null},{"url":"/paper/correspondence-matters-for-video-referring","slug":"correspondence-matters-for-video-referring","title":"Correspondence Matters for Video Referring Expression Comprehension","date":"2022-07-21","arxiv_id":"2207.10400","repositories_listed":1,"syntology":null},{"url":"/paper/entity-enhanced-adaptive-reconstruction","slug":"entity-enhanced-adaptive-reconstruction","title":"Entity-enhanced Adaptive Reconstruction Network for Weakly Supervised Referring Expression Grounding","date":"2022-07-18","arxiv_id":"2207.08386","repositories_listed":1,"syntology":null},{"url":"/paper/improving-visual-grounding-by-encouraging","slug":"improving-visual-grounding-by-encouraging","title":"Improving Visual Grounding by Encouraging Consistent Gradient-based Explanations","date":"2022-06-30","arxiv_id":"2206.15462","repositories_listed":1,"syntology":null},{"url":"/paper/pevl-position-enhanced-pre-training-and","slug":"pevl-position-enhanced-pre-training-and","title":"PEVL: Position-enhanced Pre-training and Prompt Tuning for Vision-language Models","date":"2022-05-23","arxiv_id":"2205.11169","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pevl-position-enhanced-pre-training-and#ran","syntology_url":"https://syntology.ai/paper/2205.11169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.11169"}},"official":{"repos":["thunlp/pevl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/grit-general-robust-image-task-benchmark","slug":"grit-general-robust-image-task-benchmark","title":"GRIT: General Robust Image Task Benchmark","date":"2022-04-28","arxiv_id":"2204.13653","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grit-general-robust-image-task-benchmark#ran","syntology_url":"https://syntology.ai/paper/2204.13653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.13653"}},"official":{"repos":["allenai/grit_official"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-goes-beyond-multi-modal-fusion-in-one","slug":"what-goes-beyond-multi-modal-fusion-in-one","title":"A Survivor in the Era of Large-Scale Pretraining: An Empirical Study of One-Stage Referring Expression Comprehension","date":"2022-04-17","arxiv_id":"2204.07913","repositories_listed":1,"syntology":null},{"url":"/paper/single-stream-multi-level-alignment-for","slug":"single-stream-multi-level-alignment-for","title":"Single-Stream Multi-Level Alignment for Vision-Language Pretraining","date":"2022-03-27","arxiv_id":"2203.14395","repositories_listed":1,"syntology":null},{"url":"/paper/lavt-language-aware-vision-transformer-for","slug":"lavt-language-aware-vision-transformer-for","title":"LAVT: Language-Aware Vision Transformer for Referring Image Segmentation","date":"2021-12-04","arxiv_id":"2112.02244","repositories_listed":1,"syntology":null},{"url":"/paper/towards-language-guided-visual-recognition","slug":"towards-language-guided-visual-recognition","title":"Towards Language-guided Visual Recognition via Dynamic Convolutions","date":"2021-10-17","arxiv_id":"2110.08797","repositories_listed":1,"syntology":null},{"url":"/paper/does-referent-predictability-affect-the","slug":"does-referent-predictability-affect-the","title":"Does referent predictability affect the choice of referential form? A computational approach using masked coreference resolution","date":"2021-09-27","arxiv_id":"2109.13105","repositories_listed":1,"syntology":null},{"url":"/paper/enriching-the-e2e-dataset","slug":"enriching-the-e2e-dataset","title":"Enriching the E2E dataset","date":"2021-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/room-and-object-aware-knowledge-reasoning-for","slug":"room-and-object-aware-knowledge-reasoning-for","title":"Room-and-Object Aware Knowledge Reasoning for Remote Embodied Referring Expression","date":"2021-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/discriminative-triad-matching-and","slug":"discriminative-triad-matching-and","title":"Discriminative Triad Matching and Reconstruction for Weakly Referring Expression Grounding","date":"2021-06-08","arxiv_id":"2106.04053","repositories_listed":1,"syntology":null},{"url":"/paper/giving-commands-to-a-self-driving-car-how-to","slug":"giving-commands-to-a-self-driving-car-how-to","title":"Giving Commands to a Self-Driving Car: How to Deal with Uncertain Situations?","date":"2021-06-08","arxiv_id":"2106.04232","repositories_listed":1,"syntology":null},{"url":"/paper/referring-transformer-a-one-step-approach-to","slug":"referring-transformer-a-one-step-approach-to","title":"Referring Transformer: A One-step Approach to Multi-task Visual Grounding","date":"2021-06-06","arxiv_id":"2106.03089","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/referring-transformer-a-one-step-approach-to#ran","syntology_url":"https://syntology.ai/paper/2106.03089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03089"}},"official":{"repos":["ubc-vision/RefTR"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-better-visual-dialog-agents-with","slug":"learning-better-visual-dialog-agents-with","title":"Learning Better Visual Dialog Agents with Pretrained Visual-Linguistic Representation","date":"2021-05-24","arxiv_id":"2105.11541","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-synonymous-referring","slug":"understanding-synonymous-referring","title":"Understanding Synonymous Referring Expressions via Contrastive Features","date":"2021-04-20","arxiv_id":"2104.10156","repositories_listed":1,"syntology":null},{"url":"/paper/ocid-ref-a-3d-robotic-dataset-with-embodied","slug":"ocid-ref-a-3d-robotic-dataset-with-embodied","title":"OCID-Ref: A 3D Robotic Dataset with Embodied Language for Clutter Scene Grounding","date":"2021-03-13","arxiv_id":"2103.07679","repositories_listed":1,"syntology":null},{"url":"/paper/iterative-shrinking-for-referring-expression","slug":"iterative-shrinking-for-referring-expression","title":"Iterative Shrinking for Referring Expression Grounding Using Deep Reinforcement Learning","date":"2021-03-09","arxiv_id":"2103.05187","repositories_listed":1,"syntology":null},{"url":"/paper/mdetr-modulated-detection-for-end-to-end-1","slug":"mdetr-modulated-detection-for-end-to-end-1","title":"MDETR - Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/trar-routing-the-attention-spans-in","slug":"trar-routing-the-attention-spans-in","title":"TRAR: Routing the Attention Spans in Transformer for Visual Question Answering","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-recurrent-vision-and-language-bert-for","slug":"a-recurrent-vision-and-language-bert-for","title":"A Recurrent Vision-and-Language BERT for Navigation","date":"2020-11-26","arxiv_id":"2011.13922","repositories_listed":1,"syntology":null},{"url":"/paper/human-centric-spatio-temporal-video-grounding","slug":"human-centric-spatio-temporal-video-grounding","title":"Human-centric Spatio-Temporal Video Grounding With Visual Transformers","date":"2020-11-10","arxiv_id":"2011.05049","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/human-centric-spatio-temporal-video-grounding#ran","syntology_url":"https://syntology.ai/paper/2011.05049","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.05049"}},"official":{"repos":["tzhhhh123/HC-STVG"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-conditioned-feature-pyramids-for","slug":"language-conditioned-feature-pyramids-for","title":"Language-Conditioned Feature Pyramids for Visual Selection Tasks","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ref-nms-breaking-proposal-bottlenecks-in-two","slug":"ref-nms-breaking-proposal-bottlenecks-in-two","title":"Ref-NMS: Breaking Proposal Bottlenecks in Two-Stage Referring Expression Grounding","date":"2020-09-03","arxiv_id":"2009.01449","repositories_listed":1,"syntology":null},{"url":"/paper/urvos-unified-referring-video-object","slug":"urvos-unified-referring-video-object","title":"URVOS: Unified Referring Video Object Segmentation Network with a Large-Scale Benchmark","date":"2020-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-one-stage-vision-and","slug":"weakly-supervised-one-stage-vision-and","title":"Weakly supervised one-stage vision and language disease detection using large scale pneumonia and pneumothorax studies","date":"2020-07-31","arxiv_id":"2007.15778","repositories_listed":1,"syntology":null},{"url":"/paper/refer360-circ-a-referring-expression","slug":"refer360-circ-a-referring-expression","title":"Refer360$^\\circ$: A Referring Expression Recognition Dataset in 360$^\\circ$ Images","date":"2020-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/give-me-something-to-eat-referring-expression","slug":"give-me-something-to-eat-referring-expression","title":"Give Me Something to Eat: Referring Expression Comprehension with Commonsense Knowledge","date":"2020-06-02","arxiv_id":"2006.01629","repositories_listed":1,"syntology":null},{"url":"/paper/words-aren-t-enough-their-order-matters-on","slug":"words-aren-t-enough-their-order-matters-on","title":"Words aren't enough, their order matters: On the Robustness of Grounding Visual Referring Expressions","date":"2020-05-04","arxiv_id":"2005.01655","repositories_listed":1,"syntology":null},{"url":"/paper/graph-structured-referring-expression","slug":"graph-structured-referring-expression","title":"Graph-Structured Referring Expression Reasoning in The Wild","date":"2020-04-19","arxiv_id":"2004.08814","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/graph-structured-referring-expression#ran","syntology_url":"https://syntology.ai/paper/2004.08814","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.08814"}},"official":{"repos":["sibeiyang/sgmn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bilingunet-image-segmentation-by-modulating","slug":"bilingunet-image-segmentation-by-modulating","title":"Modulating Bottom-Up and Top-Down Visual Processing via Language-Conditional Filters","date":"2020-03-28","arxiv_id":"2003.12739","repositories_listed":1,"syntology":null},{"url":"/paper/a-real-time-global-inference-network-for-one","slug":"a-real-time-global-inference-network-for-one","title":"A Real-time Global Inference Network for One-stage Referring Expression Comprehension","date":"2019-12-07","arxiv_id":"1912.03478","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":9,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-real-time-global-inference-network-for-one#ran","syntology_url":"https://syntology.ai/paper/1912.03478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.03478"}},"official":{"repos":["luogen1996/Real-time-Global-Inference-Network"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/referring-expression-object-segmentation-with","slug":"referring-expression-object-segmentation-with","title":"Referring Expression Object Segmentation with Caption-Aware Consistency","date":"2019-10-10","arxiv_id":"1910.04748","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/referring-expression-object-segmentation-with#ran","syntology_url":"https://syntology.ai/paper/1910.04748","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.04748"}},"official":{"repos":["wenz116/lang2seg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/190909986","slug":"190909986","title":"Improving Quality and Efficiency in Plan-based Neural Data-to-Text Generation","date":"2019-09-22","arxiv_id":"1909.09986","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-guided-pairwise-reconstruction","slug":"knowledge-guided-pairwise-reconstruction","title":"Knowledge-guided Pairwise Reconstruction Network for Weakly Supervised Referring Expression Grounding","date":"2019-09-05","arxiv_id":"1909.02860","repositories_listed":1,"syntology":null},{"url":"/paper/referring-expression-generation-using-entity","slug":"referring-expression-generation-using-entity","title":"Referring Expression Generation Using Entity Profiles","date":"2019-09-04","arxiv_id":"1909.01528","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-reconstruction-network-for-weakly","slug":"adaptive-reconstruction-network-for-weakly","title":"Adaptive Reconstruction Network for Weakly Supervised Referring Expression Grounding","date":"2019-08-28","arxiv_id":"1908.10568","repositories_listed":1,"syntology":null},{"url":"/paper/searching-for-ambiguous-objects-in-videos","slug":"searching-for-ambiguous-objects-in-videos","title":"Searching for Ambiguous Objects in Videos using Relational Referring Expressions","date":"2019-08-03","arxiv_id":"1908.01189","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-relationship-inference-for-1","slug":"cross-modal-relationship-inference-for-1","title":"Relationship-Embedded Representation Learning for Grounding Referring Expressions","date":"2019-06-11","arxiv_id":"1906.04464","repositories_listed":1,"syntology":null},{"url":"/paper/rerere-remote-embodied-referring-expressions","slug":"rerere-remote-embodied-referring-expressions","title":"REVERIE: Remote Embodied Visual Referring Expression in Real Indoor Environments","date":"2019-04-23","arxiv_id":"1904.10151","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rerere-remote-embodied-referring-expressions#ran","syntology_url":"https://syntology.ai/paper/1904.10151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.10151"}},"official":null}},{"url":"/paper/cross-modal-self-attention-network-for","slug":"cross-modal-self-attention-network-for","title":"Cross-Modal Self-Attention Network for Referring Image Segmentation","date":"2019-04-09","arxiv_id":"1904.04745","repositories_listed":1,"syntology":null},{"url":"/paper/enriching-the-webnlg-corpus","slug":"enriching-the-webnlg-corpus","title":"Enriching the WebNLG corpus","date":"2018-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/visual-referring-expression-recognition-what","slug":"visual-referring-expression-recognition-what","title":"Visual Referring Expression Recognition: What Do Systems Actually Learn?","date":"2018-05-30","arxiv_id":"1805.11818","repositories_listed":1,"syntology":null},{"url":"/paper/using-syntax-to-ground-referring-expressions","slug":"using-syntax-to-ground-referring-expressions","title":"Using Syntax to Ground Referring Expressions in Natural Images","date":"2018-05-26","arxiv_id":"1805.10547","repositories_listed":1,"syntology":null},{"url":"/paper/neuralreg-an-end-to-end-approach-to-referring","slug":"neuralreg-an-end-to-end-approach-to-referring","title":"NeuralREG: An end-to-end approach to referring expression generation","date":"2018-05-21","arxiv_id":"1805.08093","repositories_listed":1,"syntology":null},{"url":"/paper/mattnet-modular-attention-network-for","slug":"mattnet-modular-attention-network-for","title":"MAttNet: Modular Attention Network for Referring Expression Comprehension","date":"2018-01-24","arxiv_id":"1801.08186","repositories_listed":1,"syntology":null},{"url":"/paper/grounding-referring-expressions-in-images-by","slug":"grounding-referring-expressions-in-images-by","title":"Grounding Referring Expressions in Images by Variational Context","date":"2017-12-05","arxiv_id":"1712.01892","repositories_listed":1,"syntology":null},{"url":"/paper/colors-in-context-a-pragmatic-neural-model","slug":"colors-in-context-a-pragmatic-neural-model","title":"Colors in Context: A Pragmatic Neural Model for Grounded Language Understanding","date":"2017-03-29","arxiv_id":"1703.10186","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/colors-in-context-a-pragmatic-neural-model#ran","syntology_url":"https://syntology.ai/paper/1703.10186","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1703.10186"}},"official":null}},{"url":"/paper/modeling-context-between-objects-for","slug":"modeling-context-between-objects-for","title":"Modeling Context Between Objects for Referring Expression Understanding","date":"2016-08-01","arxiv_id":"1608.00525","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-about-pragmatics-with-neural","slug":"reasoning-about-pragmatics-with-neural","title":"Reasoning About Pragmatics with Neural Listeners and Speakers","date":"2016-04-02","arxiv_id":"1604.00562","repositories_listed":1,"syntology":null},{"url":"/paper/generation-and-comprehension-of-unambiguous","slug":"generation-and-comprehension-of-unambiguous","title":"Generation and Comprehension of Unambiguous Object Descriptions","date":"2015-11-07","arxiv_id":"1511.02283","repositories_listed":1,"syntology":null},{"url":null,"slug":"mask-aware-text-to-image-retrieval-referring","title":"Mask-aware Text-to-Image Retrieval: Referring Expression Segmentation Meets Cross-modal Retrieval","date":"2025-06-28","arxiv_id":"2506.22864","repositories_listed":0,"syntology":null},{"url":null,"slug":"referring-expression-instance-retrieval-and-a","title":"Referring Expression Instance Retrieval and A Strong End-to-End Baseline","date":"2025-06-23","arxiv_id":"2506.18246","repositories_listed":0,"syntology":null},{"url":null,"slug":"gondola-grounded-vision-language-planning-for","title":"Gondola: Grounded Vision Language Planning for Generalizable Robotic Manipulation","date":"2025-06-12","arxiv_id":"2506.11261","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthetic-visual-genome-1","title":"Synthetic Visual Genome","date":"2025-06-09","arxiv_id":"2506.07643","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-objects-to-anywhere-a-holistic-benchmark","title":"From Objects to Anywhere: A Holistic Benchmark for Multi-level Visual Grounding in 3D Scenes","date":"2025-06-05","arxiv_id":"2506.04897","repositories_listed":0,"syntology":null},{"url":null,"slug":"refer-to-anything-with-vision-language","title":"Refer to Anything with Vision-Language Prompts","date":"2025-06-05","arxiv_id":"2506.05342","repositories_listed":0,"syntology":null},{"url":null,"slug":"rex-thinker-grounded-object-referring-via","title":"Rex-Thinker: Grounded Object Referring via Chain-of-Thought Reasoning","date":"2025-06-04","arxiv_id":"2506.04034","repositories_listed":0,"syntology":null},{"url":null,"slug":"refedit-a-benchmark-and-method-for-improving","title":"RefEdit: A Benchmark and Method for Improving Instruction-based Image Editing Model on Referring Expressions","date":"2025-06-03","arxiv_id":"2506.03448","repositories_listed":0,"syntology":null},{"url":null,"slug":"deformable-attentive-visual-enhancement-for","title":"Deformable Attentive Visual Enhancement for Referring Segmentation Using Vision-Language Model","date":"2025-05-25","arxiv_id":"2505.19242","repositories_listed":0,"syntology":null},{"url":"/paper/weakmcn-multi-task-collaborative-network-for","slug":"weakmcn-multi-task-collaborative-network-for","title":"WeakMCN: Multi-task Collaborative Network for Weakly Supervised Referring Expression Comprehension and Segmentation","date":"2025-05-24","arxiv_id":"2505.18686","repositories_listed":0,"syntology":{"n":24,"n_ran":14,"n_constructed":0,"n_ran_checked":8,"n_instrument":6,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":24,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 6 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/weakmcn-multi-task-collaborative-network-for#ran","syntology_url":"https://syntology.ai/paper/2505.18686","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18686"}},"official":null}},{"url":null,"slug":"learning-to-reason-and-navigate-parameter","title":"Learning to Reason and Navigate: Parameter Efficient Action Planning with Large Language Models","date":"2025-05-12","arxiv_id":"2505.07500","repositories_listed":0,"syntology":null},{"url":null,"slug":"resanything-attribute-prompting-for-arbitrary","title":"RESAnything: Attribute Prompting for Arbitrary Referring Segmentation","date":"2025-05-03","arxiv_id":"2505.02867","repositories_listed":0,"syntology":null},{"url":null,"slug":"lgd-leveraging-generative-descriptions-for","title":"LGD: Leveraging Generative Descriptions for Zero-Shot Referring Image Segmentation","date":"2025-04-20","arxiv_id":"2504.14467","repositories_listed":0,"syntology":null},{"url":null,"slug":"3drest-a-strong-baseline-for-semi-supervised","title":"3DResT: A Strong Baseline for Semi-Supervised 3D Referring Expression Segmentation","date":"2025-04-17","arxiv_id":"2504.12599","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-unified-referring-expression","title":"Towards Unified Referring Expression Segmentation Across Omni-Level Visual Target Granularities","date":"2025-04-02","arxiv_id":"2504.01954","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-object-categories-multi-attribute","title":"Beyond Object Categories: Multi-Attribute Reference Understanding for Visual Grounding","date":"2025-03-25","arxiv_id":"2503.19240","repositories_listed":0,"syntology":null},{"url":null,"slug":"georsmllm-a-multimodal-large-language-model","title":"GeoRSMLLM: A Multimodal Large Language Model for Vision-Language Tasks in Geoscience and Remote Sensing","date":"2025-03-16","arxiv_id":"2503.12490","repositories_listed":0,"syntology":null},{"url":null,"slug":"cognitive-disentanglement-for-referring-multi","title":"Cognitive Disentanglement for Referring Multi-Object Tracking","date":"2025-03-14","arxiv_id":"2503.11496","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-spatial-language-grounding-through","title":"Exploring Spatial Language Grounding Through Referring Expressions","date":"2025-02-04","arxiv_id":"2502.04359","repositories_listed":0,"syntology":null},{"url":null,"slug":"implicit-causality-biases-in-humans-and-llms","title":"Implicit Causality-biases in humans and LLMs as a tool for benchmarking LLM discourse capabilities","date":"2025-01-22","arxiv_id":"2501.12980","repositories_listed":0,"syntology":null},{"url":null,"slug":"flora-formal-language-model-enables-robust","title":"FLORA: Formal Language Model Enables Robust Training-free Zero-shot Object Referring Analysis","date":"2025-01-17","arxiv_id":"2501.09887","repositories_listed":0,"syntology":null},{"url":null,"slug":"omni-rgpt-unifying-image-and-video-region","title":"Omni-RGPT: Unifying Image and Video Region-level Understanding via Token Marks","date":"2025-01-14","arxiv_id":"2501.08326","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-alignment-enhanced-adaptive","title":"Hierarchical Alignment-enhanced Adaptive Grounding Network for Generalized Referring Expression Comprehension","date":"2025-01-02","arxiv_id":"2501.01416","repositories_listed":0,"syntology":null},{"url":null,"slug":"dvin-dynamic-visual-routing-network-for","title":"DViN: Dynamic Visual Routing Network for Weakly Supervised Referring Expression Comprehension","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"task-aware-cross-modal-feature-refinement","title":"Task-aware Cross-modal Feature Refinement Transformer with Large Language Models for Visual Grounding","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"harlequin-color-driven-generation-of","title":"Harlequin: Color-driven Generation of Synthetic Data for Referring Expression Comprehension","date":"2024-11-22","arxiv_id":"2411.14807","repositories_listed":0,"syntology":null},{"url":null,"slug":"instance-aware-generalized-referring","title":"Instance-Aware Generalized Referring Expression Segmentation","date":"2024-11-22","arxiv_id":"2411.15087","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-nemo-negative-mined-mosaic","title":"Finding NeMo: Negative-mined Mosaic Augmentation for Referring Image Segmentation","date":"2024-11-03","arxiv_id":"2411.01494","repositories_listed":0,"syntology":null},{"url":null,"slug":"segllm-multi-round-reasoning-segmentation","title":"SegLLM: Multi-round Reasoning Segmentation","date":"2024-10-24","arxiv_id":"2410.18923","repositories_listed":0,"syntology":null},{"url":null,"slug":"make-graph-based-referring-expression","title":"Make Graph-based Referring Expression Comprehension Great Again through Expression-guided Dynamic Gating and Regression","date":"2024-09-05","arxiv_id":"2409.03385","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-multi-modal-llm-evaluation","title":"Revisiting Multi-Modal LLM Evaluation","date":"2024-08-09","arxiv_id":"2408.05334","repositories_listed":0,"syntology":null},{"url":null,"slug":"maskinversion-localized-embeddings-via","title":"MaskInversion: Localized Embeddings via Optimization of Explainability Maps","date":"2024-07-29","arxiv_id":"2407.20034","repositories_listed":0,"syntology":null},{"url":null,"slug":"look-hear-gaze-prediction-for-speech-directed","title":"Look Hear: Gaze Prediction for Speech-directed Human Attention","date":"2024-07-28","arxiv_id":"2407.19605","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-visual-grounding-from-generative","title":"Learning Visual Grounding from Generative Vision and Language Model","date":"2024-07-18","arxiv_id":"2407.14563","repositories_listed":0,"syntology":null}],"record_sha256":"161fa555bab0bcba78753be26cf9834bbd7d7ade2fb471e770bdb6b4d0cdd4cb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}