{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/referring-expression/papers/ran/1","list_of":"/task/referring-expression","task":"Referring Expression","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,57],"of":57,"counts":{"archive_papers_tagged":364,"with_a_code_link":166,"where_syntology_ran_a_sample":57,"not_listed_spam_title":0,"listed":364,"listed_where_code_ran":57,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":51,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":51,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/referring-expression/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/weakmcn-multi-task-collaborative-network-for","slug":"weakmcn-multi-task-collaborative-network-for","title":"WeakMCN: Multi-task Collaborative Network for Weakly Supervised Referring Expression Comprehension and Segmentation","date":"2025-05-24","arxiv_id":"2505.18686","repositories_listed":0,"syntology":{"n":24,"n_ran":14,"n_constructed":0,"n_ran_checked":8,"n_instrument":6,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":24,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 6 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/weakmcn-multi-task-collaborative-network-for#ran","syntology_url":"https://syntology.ai/paper/2505.18686","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18686"}},"official":null}},{"url":"/paper/exploring-contextual-attribute-density-in-1","slug":"exploring-contextual-attribute-density-in-1","title":"Exploring Contextual Attribute Density in Referring Expression Counting","date":"2025-03-16","arxiv_id":"2503.12460","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploring-contextual-attribute-density-in-1#ran","syntology_url":"https://syntology.ai/paper/2503.12460","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.12460"}},"official":{"repos":["xu3xiwang/cad-gd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-task-visual-grounding-with-coarse-to","slug":"multi-task-visual-grounding-with-coarse-to","title":"Multi-task Visual Grounding with Coarse-to-Fine Consistency Constraints","date":"2025-01-12","arxiv_id":"2501.06710","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/multi-task-visual-grounding-with-coarse-to#ran","syntology_url":"https://syntology.ai/paper/2501.06710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.06710"}},"official":{"repos":["dmmm1997/c3vg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-visual-grounding-a-survey","slug":"towards-visual-grounding-a-survey","title":"Towards Visual Grounding: A Survey","date":"2024-12-28","arxiv_id":"2412.20206","repositories_listed":4,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":7,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-visual-grounding-a-survey#ran","syntology_url":"https://syntology.ai/paper/2412.20206","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.20206"}},"official":null}},{"url":"/paper/rg-san-rule-guided-spatial-awareness-network","slug":"rg-san-rule-guided-spatial-awareness-network","title":"RG-SAN: Rule-Guided Spatial Awareness Network for End-to-End 3D Referring Expression Segmentation","date":"2024-12-03","arxiv_id":"2412.02402","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":16,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/rg-san-rule-guided-spatial-awareness-network#ran","syntology_url":"https://syntology.ai/paper/2412.02402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.02402"}},"official":{"repos":["sosppxo/rg-san"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/text4seg-reimagining-image-segmentation-as","slug":"text4seg-reimagining-image-segmentation-as","title":"Text4Seg: Reimagining Image Segmentation as Text Generation","date":"2024-10-13","arxiv_id":"2410.09855","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/text4seg-reimagining-image-segmentation-as#ran","syntology_url":"https://syntology.ai/paper/2410.09855","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09855"}},"official":{"repos":["mc-lan/text4seg"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/grounding-language-in-multi-perspective","slug":"grounding-language-in-multi-perspective","title":"Grounding Language in Multi-Perspective Referential Communication","date":"2024-10-04","arxiv_id":"2410.03959","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounding-language-in-multi-perspective#ran","syntology_url":"https://syntology.ai/paper/2410.03959","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03959"}},"official":{"repos":["zinengtang/MulAgentRef"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/finecops-ref-a-new-dataset-and-task-for-fine","slug":"finecops-ref-a-new-dataset-and-task-for-fine","title":"FineCops-Ref: A new Dataset and Task for Fine-Grained Compositional Referring Expression Comprehension","date":"2024-09-23","arxiv_id":"2409.14750","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finecops-ref-a-new-dataset-and-task-for-fine#ran","syntology_url":"https://syntology.ai/paper/2409.14750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.14750"}},"official":{"repos":["liujunzhuo/FineCops-Ref"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/mapper-multimodal-prior-guided-parameter","slug":"mapper-multimodal-prior-guided-parameter","title":"MaPPER: Multimodal Prior-guided Parameter Efficient Tuning for Referring Expression Comprehension","date":"2024-09-20","arxiv_id":"2409.13609","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mapper-multimodal-prior-guided-parameter#ran","syntology_url":"https://syntology.ai/paper/2409.13609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.13609"}},"official":{"repos":["liuting20/mapper"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-wrapper-black-box-semantic-aware","slug":"llm-wrapper-black-box-semantic-aware","title":"LLM-wrapper: Black-Box Semantic-Aware Adaptation of Vision-Language Models for Referring Expression Comprehension","date":"2024-09-18","arxiv_id":"2409.11919","repositories_listed":1,"syntology":{"n":18,"n_ran":9,"n_constructed":0,"n_ran_checked":1,"n_instrument":8,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/llm-wrapper-black-box-semantic-aware#ran","syntology_url":"https://syntology.ai/paper/2409.11919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.11919"}},"official":{"repos":["valeoai/LLM_wrapper"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/3d-gres-generalized-3d-referring-expression","slug":"3d-gres-generalized-3d-referring-expression","title":"3D-GRES: Generalized 3D Referring Expression Segmentation","date":"2024-07-30","arxiv_id":"2407.20664","repositories_listed":2,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/3d-gres-generalized-3d-referring-expression#ran","syntology_url":"https://syntology.ai/paper/2407.20664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.20664"}},"official":{"repos":["sosppxo/MDIN"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/evf-sam-early-vision-language-fusion-for-text","slug":"evf-sam-early-vision-language-fusion-for-text","title":"EVF-SAM: Early Vision-Language Fusion for Text-Prompted Segment Anything Model","date":"2024-06-28","arxiv_id":"2406.20076","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evf-sam-early-vision-language-fusion-for-text#ran","syntology_url":"https://syntology.ai/paper/2406.20076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.20076"}},"official":{"repos":["hustvl/evf-sam"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/revisiting-referring-expression-comprehension","slug":"revisiting-referring-expression-comprehension","title":"Revisiting Referring Expression Comprehension Evaluation in the Era of Large Multimodal Models","date":"2024-06-24","arxiv_id":"2406.16866","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/revisiting-referring-expression-comprehension#ran","syntology_url":"https://syntology.ai/paper/2406.16866","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16866"}},"official":{"repos":["jierunchen/ref-l4"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarial-robustness-for-visual-grounding","slug":"adversarial-robustness-for-visual-grounding","title":"Adversarial Robustness for Visual Grounding of Multimodal Large Language Models","date":"2024-05-16","arxiv_id":"2405.09981","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adversarial-robustness-for-visual-grounding#ran","syntology_url":"https://syntology.ai/paper/2405.09981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.09981"}},"official":{"repos":["KuofengGao/MLLM-Grounding-Robustness"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/decoupling-static-and-hierarchical-motion","slug":"decoupling-static-and-hierarchical-motion","title":"Decoupling Static and Hierarchical Motion Perception for Referring Video Segmentation","date":"2024-04-04","arxiv_id":"2404.03645","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/decoupling-static-and-hierarchical-motion#ran","syntology_url":"https://syntology.ai/paper/2404.03645","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03645"}},"official":{"repos":["heshuting555/dshmp"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/elysium-exploring-object-level-perception-in","slug":"elysium-exploring-object-level-perception-in","title":"Elysium: Exploring Object-level Perception in Videos via MLLM","date":"2024-03-25","arxiv_id":"2403.16558","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/elysium-exploring-object-level-perception-in#ran","syntology_url":"https://syntology.ai/paper/2403.16558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16558"}},"official":{"repos":["hon-wong/elysium"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/psalm-pixelwise-segmentation-with-large-multi","slug":"psalm-pixelwise-segmentation-with-large-multi","title":"PSALM: Pixelwise SegmentAtion with Large Multi-Modal Model","date":"2024-03-21","arxiv_id":"2403.14598","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/psalm-pixelwise-segmentation-with-large-multi#ran","syntology_url":"https://syntology.ai/paper/2403.14598","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.14598"}},"official":{"repos":["zamling/psalm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-modal-instruction-tuned-llms-with-fine","slug":"multi-modal-instruction-tuned-llms-with-fine","title":"Multi-modal Instruction Tuned LLMs with Fine-grained Visual Perception","date":"2024-03-05","arxiv_id":"2403.02969","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multi-modal-instruction-tuned-llms-with-fine#ran","syntology_url":"https://syntology.ai/paper/2403.02969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02969"}},"official":{"repos":["jwh97nn/anyref"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/an-open-and-comprehensive-pipeline-for","slug":"an-open-and-comprehensive-pipeline-for","title":"An Open and Comprehensive Pipeline for Unified Object Grounding and Detection","date":"2024-01-04","arxiv_id":"2401.02361","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/an-open-and-comprehensive-pipeline-for#ran","syntology_url":"https://syntology.ai/paper/2401.02361","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02361"}},"official":{"repos":["open-mmlab/mmdetection"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/gsva-generalized-segmentation-via-multimodal","slug":"gsva-generalized-segmentation-via-multimodal","title":"GSVA: Generalized Segmentation via Multimodal Large Language Models","date":"2023-12-15","arxiv_id":"2312.10103","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/gsva-generalized-segmentation-via-multimodal#ran","syntology_url":"https://syntology.ai/paper/2312.10103","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10103"}},"official":{"repos":["leaplabthu/gsva"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/localized-symbolic-knowledge-distillation-for-1","slug":"localized-symbolic-knowledge-distillation-for-1","title":"Localized Symbolic Knowledge Distillation for Visual Commonsense Models","date":"2023-12-08","arxiv_id":"2312.04837","repositories_listed":2,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/localized-symbolic-knowledge-distillation-for-1#ran","syntology_url":"https://syntology.ai/paper/2312.04837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.04837"}},"official":{"repos":["jamespark3922/localized-skd","jamespark3922/lskd"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-referring-expression-comprehension","slug":"zero-shot-referring-expression-comprehension","title":"Zero-shot Referring Expression Comprehension via Structural Similarity Between Images and Captions","date":"2023-11-28","arxiv_id":"2311.17048","repositories_listed":1,"syntology":{"n":20,"n_ran":14,"n_constructed":0,"n_ran_checked":10,"n_instrument":4,"n_unverified":6,"n_honours":0,"n_violates":3,"n_no_contract":7,"n_pointer_only":10,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 3 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/zero-shot-referring-expression-comprehension#ran","syntology_url":"https://syntology.ai/paper/2311.17048","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17048"}},"official":{"repos":["show-han/zeroshot_rec"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/next-chat-an-lmm-for-chat-detection-and","slug":"next-chat-an-lmm-for-chat-detection-and","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","date":"2023-11-08","arxiv_id":"2311.04498","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/next-chat-an-lmm-for-chat-detection-and#ran","syntology_url":"https://syntology.ai/paper/2311.04498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04498"}},"official":{"repos":["next-chatv/next-chat"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/genome-generative-neuro-symbolic-visual","slug":"genome-generative-neuro-symbolic-visual","title":"GENOME: GenerativE Neuro-symbOlic visual reasoning by growing and reusing ModulEs","date":"2023-11-08","arxiv_id":"2311.04901","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/genome-generative-neuro-symbolic-visual#ran","syntology_url":"https://syntology.ai/paper/2311.04901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04901"}},"official":null}},{"url":"/paper/glamm-pixel-grounding-large-multimodal-model","slug":"glamm-pixel-grounding-large-multimodal-model","title":"GLaMM: Pixel Grounding Large Multimodal Model","date":"2023-11-06","arxiv_id":"2311.03356","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/glamm-pixel-grounding-large-multimodal-model#ran","syntology_url":"https://syntology.ai/paper/2311.03356","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.03356"}},"official":{"repos":["mbzuai-oryx/groundingLMM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pink-unveiling-the-power-of-referential","slug":"pink-unveiling-the-power-of-referential","title":"Pink: Unveiling the Power of Referential Comprehension for Multi-modal LLMs","date":"2023-10-01","arxiv_id":"2310.00582","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pink-unveiling-the-power-of-referential#ran","syntology_url":"https://syntology.ai/paper/2310.00582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00582"}},"official":{"repos":["sy-xuan/pink"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/3d-stmn-dependency-driven-superpoint-text","slug":"3d-stmn-dependency-driven-superpoint-text","title":"3D-STMN: Dependency-Driven Superpoint-Text Matching Network for End-to-End 3D Referring Expression Segmentation","date":"2023-08-31","arxiv_id":"2308.16632","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/3d-stmn-dependency-driven-superpoint-text#ran","syntology_url":"https://syntology.ai/paper/2308.16632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.16632"}},"official":{"repos":["sosppxo/3d-stmn"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-unified-framework-for-3d-point-cloud-visual","slug":"a-unified-framework-for-3d-point-cloud-visual","title":"A Unified Framework for 3D Point Cloud Visual Grounding","date":"2023-08-23","arxiv_id":"2308.11887","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-framework-for-3d-point-cloud-visual#ran","syntology_url":"https://syntology.ai/paper/2308.11887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11887"}},"official":{"repos":["leon1207/3dreftr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/described-object-detection-liberating-object-1","slug":"described-object-detection-liberating-object-1","title":"Described Object Detection: Liberating Object Detection with Flexible Expressions","date":"2023-07-24","arxiv_id":"2307.12813","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/described-object-detection-liberating-object-1#ran","syntology_url":"https://syntology.ai/paper/2307.12813","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12813"}},"official":{"repos":["shikras/d-cube"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gres-generalized-referring-expression-1","slug":"gres-generalized-referring-expression-1","title":"GRES: Generalized Referring Expression Segmentation","date":"2023-06-01","arxiv_id":"2306.00968","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gres-generalized-referring-expression-1#ran","syntology_url":"https://syntology.ai/paper/2306.00968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00968"}},"official":{"repos":["henghuiding/ReLA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/universal-instance-perception-as-object","slug":"universal-instance-perception-as-object","title":"Universal Instance Perception as Object Discovery and Retrieval","date":"2023-03-12","arxiv_id":"2303.06674","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/universal-instance-perception-as-object#ran","syntology_url":"https://syntology.ai/paper/2303.06674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06674"}},"official":{"repos":["MasterBin-IIAU/UNINEXT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grounding-dino-marrying-dino-with-grounded","slug":"grounding-dino-marrying-dino-with-grounded","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","date":"2023-03-09","arxiv_id":"2303.05499","repositories_listed":10,"syntology":{"n":5,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/grounding-dino-marrying-dino-with-grounded#ran","syntology_url":"https://syntology.ai/paper/2303.05499","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.05499"}},"official":{"repos":["idea-research/groundingdino"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/sqa3d-situated-question-answering-in-3d","slug":"sqa3d-situated-question-answering-in-3d","title":"SQA3D: Situated Question Answering in 3D Scenes","date":"2022-10-14","arxiv_id":"2210.07474","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":6,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sqa3d-situated-question-answering-in-3d#ran","syntology_url":"https://syntology.ai/paper/2210.07474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07474"}},"official":{"repos":["SilongYong/SQA3D"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/volta-vision-language-transformer-with-weakly","slug":"volta-vision-language-transformer-with-weakly","title":"VoLTA: Vision-Language Transformer with Weakly-Supervised Local-Feature Alignment","date":"2022-10-09","arxiv_id":"2210.04135","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/volta-vision-language-transformer-with-weakly#ran","syntology_url":"https://syntology.ai/paper/2210.04135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.04135"}},"official":{"repos":["ShramanPramanick/VoLTA"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/pevl-position-enhanced-pre-training-and","slug":"pevl-position-enhanced-pre-training-and","title":"PEVL: Position-enhanced Pre-training and Prompt Tuning for Vision-language Models","date":"2022-05-23","arxiv_id":"2205.11169","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pevl-position-enhanced-pre-training-and#ran","syntology_url":"https://syntology.ai/paper/2205.11169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.11169"}},"official":{"repos":["thunlp/pevl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/grit-general-robust-image-task-benchmark","slug":"grit-general-robust-image-task-benchmark","title":"GRIT: General Robust Image Task Benchmark","date":"2022-04-28","arxiv_id":"2204.13653","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grit-general-robust-image-task-benchmark#ran","syntology_url":"https://syntology.ai/paper/2204.13653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.13653"}},"official":{"repos":["allenai/grit_official"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reclip-a-strong-zero-shot-baseline-for-1","slug":"reclip-a-strong-zero-shot-baseline-for-1","title":"ReCLIP: A Strong Zero-Shot Baseline for Referring Expression Comprehension","date":"2022-04-12","arxiv_id":"2204.05991","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/reclip-a-strong-zero-shot-baseline-for-1#ran","syntology_url":"https://syntology.ai/paper/2204.05991","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.05991"}},"official":{"repos":["allenai/reclip"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/seqtr-a-simple-yet-universal-network-for","slug":"seqtr-a-simple-yet-universal-network-for","title":"SeqTR: A Simple yet Universal Network for Visual Grounding","date":"2022-03-30","arxiv_id":"2203.16265","repositories_listed":3,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/seqtr-a-simple-yet-universal-network-for#ran","syntology_url":"https://syntology.ai/paper/2203.16265","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16265"}},"official":{"repos":["sean-zhuh/seqtr"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/unifying-architectures-tasks-and-modalities","slug":"unifying-architectures-tasks-and-modalities","title":"OFA: Unifying Architectures, Tasks, and Modalities Through a Simple Sequence-to-Sequence Learning Framework","date":"2022-02-07","arxiv_id":"2202.03052","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-architectures-tasks-and-modalities#ran","syntology_url":"https://syntology.ai/paper/2202.03052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03052"}},"official":{"repos":["ofa-sys/ofa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prompt-based-multi-modal-image-segmentation","slug":"prompt-based-multi-modal-image-segmentation","title":"Image Segmentation Using Text and Image Prompts","date":"2021-12-18","arxiv_id":"2112.10003","repositories_listed":6,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/prompt-based-multi-modal-image-segmentation#ran","syntology_url":"https://syntology.ai/paper/2112.10003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.10003"}},"official":{"repos":["timojl/clipseg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/airbert-in-domain-pretraining-for-vision-and","slug":"airbert-in-domain-pretraining-for-vision-and","title":"Airbert: In-domain Pretraining for Vision-and-Language Navigation","date":"2021-08-20","arxiv_id":"2108.09105","repositories_listed":2,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/airbert-in-domain-pretraining-for-vision-and#ran","syntology_url":"https://syntology.ai/paper/2108.09105","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.09105"}},"official":{"repos":["airbert-vln/airbert"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/referring-transformer-a-one-step-approach-to","slug":"referring-transformer-a-one-step-approach-to","title":"Referring Transformer: A One-step Approach to Multi-task Visual Grounding","date":"2021-06-06","arxiv_id":"2106.03089","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/referring-transformer-a-one-step-approach-to#ran","syntology_url":"https://syntology.ai/paper/2106.03089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03089"}},"official":{"repos":["ubc-vision/RefTR"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mdetr-modulated-detection-for-end-to-end","slug":"mdetr-modulated-detection-for-end-to-end","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-04-26","arxiv_id":"2104.12763","repositories_listed":5,"syntology":{"n":11,"n_ran":7,"n_constructed":4,"n_ran_checked":6,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/mdetr-modulated-detection-for-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2104.12763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.12763"}},"official":{"repos":["ashkamath/mdetr"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/unifying-vision-and-language-tasks-via-text","slug":"unifying-vision-and-language-tasks-via-text","title":"Unifying Vision-and-Language Tasks via Text Generation","date":"2021-02-04","arxiv_id":"2102.02779","repositories_listed":2,"syntology":{"n":12,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 10 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/unifying-vision-and-language-tasks-via-text#ran","syntology_url":"https://syntology.ai/paper/2102.02779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.02779"}},"official":{"repos":["j-min/VL-T5"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/human-centric-spatio-temporal-video-grounding","slug":"human-centric-spatio-temporal-video-grounding","title":"Human-centric Spatio-Temporal Video Grounding With Visual Transformers","date":"2020-11-10","arxiv_id":"2011.05049","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/human-centric-spatio-temporal-video-grounding#ran","syntology_url":"https://syntology.ai/paper/2011.05049","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.05049"}},"official":{"repos":["tzhhhh123/HC-STVG"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-scale-adversarial-training-for-vision","slug":"large-scale-adversarial-training-for-vision","title":"Large-Scale Adversarial Training for Vision-and-Language Representation Learning","date":"2020-06-11","arxiv_id":"2006.06195","repositories_listed":2,"syntology":{"n":20,"n_ran":12,"n_constructed":5,"n_ran_checked":9,"n_instrument":3,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"12 ran (of which 5 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/large-scale-adversarial-training-for-vision#ran","syntology_url":"https://syntology.ai/paper/2006.06195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06195"}},"official":{"repos":["zhegan27/LXMERT-AdvTrain","zhegan27/VILLA"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":5,"n_ran_no_instrument_failure":9,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/graph-structured-referring-expression","slug":"graph-structured-referring-expression","title":"Graph-Structured Referring Expression Reasoning in The Wild","date":"2020-04-19","arxiv_id":"2004.08814","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/graph-structured-referring-expression#ran","syntology_url":"https://syntology.ai/paper/2004.08814","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.08814"}},"official":{"repos":["sibeiyang/sgmn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-task-collaborative-network-for-joint","slug":"multi-task-collaborative-network-for-joint","title":"Multi-task Collaborative Network for Joint Referring Expression Comprehension and Segmentation","date":"2020-03-19","arxiv_id":"2003.08813","repositories_listed":2,"syntology":{"n":13,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/multi-task-collaborative-network-for-joint#ran","syntology_url":"https://syntology.ai/paper/2003.08813","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.08813"}},"official":{"repos":["luogen1996/MCN"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/a-real-time-global-inference-network-for-one","slug":"a-real-time-global-inference-network-for-one","title":"A Real-time Global Inference Network for One-stage Referring Expression Comprehension","date":"2019-12-07","arxiv_id":"1912.03478","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":9,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-real-time-global-inference-network-for-one#ran","syntology_url":"https://syntology.ai/paper/1912.03478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.03478"}},"official":{"repos":["luogen1996/Real-time-Global-Inference-Network"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/referring-expression-object-segmentation-with","slug":"referring-expression-object-segmentation-with","title":"Referring Expression Object Segmentation with Caption-Aware Consistency","date":"2019-10-10","arxiv_id":"1910.04748","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/referring-expression-object-segmentation-with#ran","syntology_url":"https://syntology.ai/paper/1910.04748","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.04748"}},"official":{"repos":["wenz116/lang2seg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uniter-learning-universal-image-text-1","slug":"uniter-learning-universal-image-text-1","title":"UNITER: UNiversal Image-TExt Representation Learning","date":"2019-09-25","arxiv_id":"1909.11740","repositories_listed":7,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uniter-learning-universal-image-text-1#ran","syntology_url":"https://syntology.ai/paper/1909.11740","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.11740"}},"official":{"repos":["ChenRocks/UNITER"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/vl-bert-pre-training-of-generic-visual","slug":"vl-bert-pre-training-of-generic-visual","title":"VL-BERT: Pre-training of Generic Visual-Linguistic Representations","date":"2019-08-22","arxiv_id":"1908.08530","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vl-bert-pre-training-of-generic-visual#ran","syntology_url":"https://syntology.ai/paper/1908.08530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.08530"}},"official":{"repos":["jackroos/VL-BERT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-fast-and-accurate-one-stage-approach-to","slug":"a-fast-and-accurate-one-stage-approach-to","title":"A Fast and Accurate One-Stage Approach to Visual Grounding","date":"2019-08-18","arxiv_id":"1908.06354","repositories_listed":2,"syntology":{"n":19,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":15,"n_pointer_only":3,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 1 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-fast-and-accurate-one-stage-approach-to#ran","syntology_url":"https://syntology.ai/paper/1908.06354","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.06354"}},"official":{"repos":["zyang-ur/onestage_grounding"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rerere-remote-embodied-referring-expressions","slug":"rerere-remote-embodied-referring-expressions","title":"REVERIE: Remote Embodied Visual Referring Expression in Real Indoor Environments","date":"2019-04-23","arxiv_id":"1904.10151","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rerere-remote-embodied-referring-expressions#ran","syntology_url":"https://syntology.ai/paper/1904.10151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.10151"}},"official":null}},{"url":"/paper/clevr-ref-diagnosing-visual-reasoning-with","slug":"clevr-ref-diagnosing-visual-reasoning-with","title":"CLEVR-Ref+: Diagnosing Visual Reasoning with Referring Expressions","date":"2019-01-03","arxiv_id":"1901.00850","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/clevr-ref-diagnosing-visual-reasoning-with#ran","syntology_url":"https://syntology.ai/paper/1901.00850","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.00850"}},"official":null}},{"url":"/paper/colors-in-context-a-pragmatic-neural-model","slug":"colors-in-context-a-pragmatic-neural-model","title":"Colors in Context: A Pragmatic Neural Model for Grounded Language Understanding","date":"2017-03-29","arxiv_id":"1703.10186","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/colors-in-context-a-pragmatic-neural-model#ran","syntology_url":"https://syntology.ai/paper/1703.10186","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1703.10186"}},"official":null}},{"url":"/paper/modeling-context-in-referring-expressions","slug":"modeling-context-in-referring-expressions","title":"Modeling Context in Referring Expressions","date":"2016-07-31","arxiv_id":"1608.00272","repositories_listed":4,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/modeling-context-in-referring-expressions#ran","syntology_url":"https://syntology.ai/paper/1608.00272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1608.00272"}},"official":{"repos":["lichengunc/refer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}}],"record_sha256":"65e97a806be0f1a5792e0cd23fa8de2066755697838d5cbfabdd96d5de781dbc","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}