{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/referring-expression-segmentation/papers/ran/1","list_of":"/task/referring-expression-segmentation","task":"Referring Expression Segmentation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,48],"of":48,"counts":{"archive_papers_tagged":145,"with_a_code_link":97,"where_syntology_ran_a_sample":48,"not_listed_spam_title":0,"listed":145,"listed_where_code_ran":48,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":42,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":42,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/referring-expression-segmentation/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/visionreasoner-unified-visual-perception-and","slug":"visionreasoner-unified-visual-perception-and","title":"VisionReasoner: Unified Visual Perception and Reasoning via Reinforcement Learning","date":"2025-05-17","arxiv_id":"2505.12081","repositories_listed":3,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visionreasoner-unified-visual-perception-and#ran","syntology_url":"https://syntology.ai/paper/2505.12081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12081"}},"official":{"repos":["dvlab-research/VisionReasoner","hiyouga/easyr1","dvlab-research/Seg-Zero"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mpg-sam-2-adapting-sam-2-with-mask-priors-and","slug":"mpg-sam-2-adapting-sam-2-with-mask-priors-and","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","date":"2025-01-23","arxiv_id":"2501.13667","repositories_listed":1,"syntology":{"n":16,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":6,"n_honours":1,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 1 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/mpg-sam-2-adapting-sam-2-with-mask-priors-and#ran","syntology_url":"https://syntology.ai/paper/2501.13667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.13667"}},"official":{"repos":["rongfu-dsb/MPG-SAM2"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/densely-connected-parameter-efficient-tuning","slug":"densely-connected-parameter-efficient-tuning","title":"Densely Connected Parameter-Efficient Tuning for Referring Image Segmentation","date":"2025-01-15","arxiv_id":"2501.08580","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":8,"n_instrument":5,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/densely-connected-parameter-efficient-tuning#ran","syntology_url":"https://syntology.ai/paper/2501.08580","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.08580"}},"official":{"repos":["jiaqihuang01/detris"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-task-visual-grounding-with-coarse-to","slug":"multi-task-visual-grounding-with-coarse-to","title":"Multi-task Visual Grounding with Coarse-to-Fine Consistency Constraints","date":"2025-01-12","arxiv_id":"2501.06710","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/multi-task-visual-grounding-with-coarse-to#ran","syntology_url":"https://syntology.ai/paper/2501.06710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.06710"}},"official":{"repos":["dmmm1997/c3vg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/rg-san-rule-guided-spatial-awareness-network","slug":"rg-san-rule-guided-spatial-awareness-network","title":"RG-SAN: Rule-Guided Spatial Awareness Network for End-to-End 3D Referring Expression Segmentation","date":"2024-12-03","arxiv_id":"2412.02402","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":16,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/rg-san-rule-guided-spatial-awareness-network#ran","syntology_url":"https://syntology.ai/paper/2412.02402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.02402"}},"official":{"repos":["sosppxo/rg-san"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/hyperseg-towards-universal-visual","slug":"hyperseg-towards-universal-visual","title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","date":"2024-11-26","arxiv_id":"2411.17606","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":9,"n_instrument":4,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":7,"n_pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 1 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/hyperseg-towards-universal-visual#ran","syntology_url":"https://syntology.ai/paper/2411.17606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17606"}},"official":{"repos":["congvvc/HyperSeg"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/text4seg-reimagining-image-segmentation-as","slug":"text4seg-reimagining-image-segmentation-as","title":"Text4Seg: Reimagining Image Segmentation as Text Generation","date":"2024-10-13","arxiv_id":"2410.09855","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/text4seg-reimagining-image-segmentation-as#ran","syntology_url":"https://syntology.ai/paper/2410.09855","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09855"}},"official":{"repos":["mc-lan/text4seg"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/3d-gres-generalized-3d-referring-expression","slug":"3d-gres-generalized-3d-referring-expression","title":"3D-GRES: Generalized 3D Referring Expression Segmentation","date":"2024-07-30","arxiv_id":"2407.20664","repositories_listed":2,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/3d-gres-generalized-3d-referring-expression#ran","syntology_url":"https://syntology.ai/paper/2407.20664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.20664"}},"official":{"repos":["sosppxo/MDIN"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-label-cluster-discrimination-for-visual","slug":"multi-label-cluster-discrimination-for-visual","title":"Multi-label Cluster Discrimination for Visual Representation Learning","date":"2024-07-24","arxiv_id":"2407.17331","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":7,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 7 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 7 samples that ran constructed an object rather than computing a result","sample_list":"/paper/multi-label-cluster-discrimination-for-visual#ran","syntology_url":"https://syntology.ai/paper/2407.17331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.17331"}},"official":{"repos":["deepglint/unicom"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":7,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/evf-sam-early-vision-language-fusion-for-text","slug":"evf-sam-early-vision-language-fusion-for-text","title":"EVF-SAM: Early Vision-Language Fusion for Text-Prompted Segment Anything Model","date":"2024-06-28","arxiv_id":"2406.20076","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evf-sam-early-vision-language-fusion-for-text#ran","syntology_url":"https://syntology.ai/paper/2406.20076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.20076"}},"official":{"repos":["hustvl/evf-sam"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/decoupling-static-and-hierarchical-motion","slug":"decoupling-static-and-hierarchical-motion","title":"Decoupling Static and Hierarchical Motion Perception for Referring Video Segmentation","date":"2024-04-04","arxiv_id":"2404.03645","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/decoupling-static-and-hierarchical-motion#ran","syntology_url":"https://syntology.ai/paper/2404.03645","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03645"}},"official":{"repos":["heshuting555/dshmp"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-temporally-consistent-referring-video","slug":"towards-temporally-consistent-referring-video","title":"Temporally Consistent Referring Video Object Segmentation with Hybrid Memory","date":"2024-03-28","arxiv_id":"2403.19407","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":9,"n_instrument":5,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":7,"n_pointer_only":4,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 1 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-temporally-consistent-referring-video#ran","syntology_url":"https://syntology.ai/paper/2403.19407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19407"}},"official":{"repos":["bo-miao/HTR"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/psalm-pixelwise-segmentation-with-large-multi","slug":"psalm-pixelwise-segmentation-with-large-multi","title":"PSALM: Pixelwise SegmentAtion with Large Multi-Modal Model","date":"2024-03-21","arxiv_id":"2403.14598","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/psalm-pixelwise-segmentation-with-large-multi#ran","syntology_url":"https://syntology.ai/paper/2403.14598","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.14598"}},"official":{"repos":["zamling/psalm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/univs-unified-and-universal-video","slug":"univs-unified-and-universal-video","title":"UniVS: Unified and Universal Video Segmentation with Prompts as Queries","date":"2024-02-28","arxiv_id":"2402.18115","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":14,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/univs-unified-and-universal-video#ran","syntology_url":"https://syntology.ai/paper/2402.18115","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18115"}},"official":{"repos":["minghanli/univs"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/uniref-segment-every-reference-object-in","slug":"uniref-segment-every-reference-object-in","title":"UniRef++: Segment Every Reference Object in Spatial and Temporal Spaces","date":"2023-12-25","arxiv_id":"2312.15715","repositories_listed":2,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uniref-segment-every-reference-object-in#ran","syntology_url":"https://syntology.ai/paper/2312.15715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15715"}},"official":{"repos":["foundationvision/uniref"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mask-grounding-for-referring-image","slug":"mask-grounding-for-referring-image","title":"Mask Grounding for Referring Image Segmentation","date":"2023-12-19","arxiv_id":"2312.12198","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":5,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mask-grounding-for-referring-image#ran","syntology_url":"https://syntology.ai/paper/2312.12198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.12198"}},"official":{"repos":["yxchng/mask-grounding"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gsva-generalized-segmentation-via-multimodal","slug":"gsva-generalized-segmentation-via-multimodal","title":"GSVA: Generalized Segmentation via Multimodal Large Language Models","date":"2023-12-15","arxiv_id":"2312.10103","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/gsva-generalized-segmentation-via-multimodal#ran","syntology_url":"https://syntology.ai/paper/2312.10103","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10103"}},"official":{"repos":["leaplabthu/gsva"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/general-object-foundation-model-for-images","slug":"general-object-foundation-model-for-images","title":"General Object Foundation Model for Images and Videos at Scale","date":"2023-12-14","arxiv_id":"2312.09158","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/general-object-foundation-model-for-images#ran","syntology_url":"https://syntology.ai/paper/2312.09158","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.09158"}},"official":{"repos":["FoundationVision/GLEE"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/universal-segmentation-at-arbitrary","slug":"universal-segmentation-at-arbitrary","title":"Universal Segmentation at Arbitrary Granularity with Language Instruction","date":"2023-12-04","arxiv_id":"2312.01623","repositories_listed":2,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":6,"n_instrument":8,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/universal-segmentation-at-arbitrary#ran","syntology_url":"https://syntology.ai/paper/2312.01623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01623"}},"official":{"repos":["workforai/UniLSeg","yongliu20/UniLSeg"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/next-chat-an-lmm-for-chat-detection-and","slug":"next-chat-an-lmm-for-chat-detection-and","title":"NExT-Chat: An LMM for Chat, Detection and Segmentation","date":"2023-11-08","arxiv_id":"2311.04498","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/next-chat-an-lmm-for-chat-detection-and#ran","syntology_url":"https://syntology.ai/paper/2311.04498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04498"}},"official":{"repos":["next-chatv/next-chat"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/glamm-pixel-grounding-large-multimodal-model","slug":"glamm-pixel-grounding-large-multimodal-model","title":"GLaMM: Pixel Grounding Large Multimodal Model","date":"2023-11-06","arxiv_id":"2311.03356","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/glamm-pixel-grounding-large-multimodal-model#ran","syntology_url":"https://syntology.ai/paper/2311.03356","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.03356"}},"official":{"repos":["mbzuai-oryx/groundingLMM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tracking-anything-with-decoupled-video","slug":"tracking-anything-with-decoupled-video","title":"Tracking Anything with Decoupled Video Segmentation","date":"2023-09-07","arxiv_id":"2309.03903","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/tracking-anything-with-decoupled-video#ran","syntology_url":"https://syntology.ai/paper/2309.03903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.03903"}},"official":{"repos":["hkchengrex/Tracking-Anything-with-DEVA"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/3d-stmn-dependency-driven-superpoint-text","slug":"3d-stmn-dependency-driven-superpoint-text","title":"3D-STMN: Dependency-Driven Superpoint-Text Matching Network for End-to-End 3D Referring Expression Segmentation","date":"2023-08-31","arxiv_id":"2308.16632","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/3d-stmn-dependency-driven-superpoint-text#ran","syntology_url":"https://syntology.ai/paper/2308.16632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.16632"}},"official":{"repos":["sosppxo/3d-stmn"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/referring-image-segmentation-using-text","slug":"referring-image-segmentation-using-text","title":"Referring Image Segmentation Using Text Supervision","date":"2023-08-28","arxiv_id":"2308.14575","repositories_listed":1,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/referring-image-segmentation-using-text#ran","syntology_url":"https://syntology.ai/paper/2308.14575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14575"}},"official":{"repos":["fawnliu/tris"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/spectrum-guided-multi-granularity-referring","slug":"spectrum-guided-multi-granularity-referring","title":"Spectrum-guided Multi-granularity Referring Video Object Segmentation","date":"2023-07-25","arxiv_id":"2307.13537","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/spectrum-guided-multi-granularity-referring#ran","syntology_url":"https://syntology.ai/paper/2307.13537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.13537"}},"official":{"repos":["bo-miao/sgmg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-vision-and-language-encoders","slug":"bridging-vision-and-language-encoders","title":"Bridging Vision and Language Encoders: Parameter-Efficient Tuning for Referring Image Segmentation","date":"2023-07-21","arxiv_id":"2307.11545","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/bridging-vision-and-language-encoders#ran","syntology_url":"https://syntology.ai/paper/2307.11545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11545"}},"official":{"repos":["kkakkkka/etris"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/onlinerefer-a-simple-online-baseline-for","slug":"onlinerefer-a-simple-online-baseline-for","title":"OnlineRefer: A Simple Online Baseline for Referring Video Object Segmentation","date":"2023-07-18","arxiv_id":"2307.09356","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/onlinerefer-a-simple-online-baseline-for#ran","syntology_url":"https://syntology.ai/paper/2307.09356","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.09356"}},"official":{"repos":["wudongming97/onlinerefer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-open-vocabulary-universal-image-1","slug":"hierarchical-open-vocabulary-universal-image-1","title":"Hierarchical Open-vocabulary Universal Image Segmentation","date":"2023-07-03","arxiv_id":"2307.00764","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hierarchical-open-vocabulary-universal-image-1#ran","syntology_url":"https://syntology.ai/paper/2307.00764","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.00764"}},"official":{"repos":["berkeley-hipie/hipie"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/shikra-unleashing-multimodal-llm-s","slug":"shikra-unleashing-multimodal-llm-s","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","date":"2023-06-27","arxiv_id":"2306.15195","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shikra-unleashing-multimodal-llm-s#ran","syntology_url":"https://syntology.ai/paper/2306.15195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.15195"}},"official":{"repos":["shikras/shikra"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gres-generalized-referring-expression-1","slug":"gres-generalized-referring-expression-1","title":"GRES: Generalized Referring Expression Segmentation","date":"2023-06-01","arxiv_id":"2306.00968","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gres-generalized-referring-expression-1#ran","syntology_url":"https://syntology.ai/paper/2306.00968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00968"}},"official":{"repos":["henghuiding/ReLA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/universal-instance-perception-as-object","slug":"universal-instance-perception-as-object","title":"Universal Instance Perception as Object Discovery and Retrieval","date":"2023-03-12","arxiv_id":"2303.06674","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/universal-instance-perception-as-object#ran","syntology_url":"https://syntology.ai/paper/2303.06674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06674"}},"official":{"repos":["MasterBin-IIAU/UNINEXT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unleashing-text-to-image-diffusion-models-for-1","slug":"unleashing-text-to-image-diffusion-models-for-1","title":"Unleashing Text-to-Image Diffusion Models for Visual Perception","date":"2023-03-03","arxiv_id":"2303.02153","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":2,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 2 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unleashing-text-to-image-diffusion-models-for-1#ran","syntology_url":"https://syntology.ai/paper/2303.02153","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.02153"}},"official":{"repos":["wl-zhao/VPD"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/polyformer-referring-image-segmentation-as","slug":"polyformer-referring-image-segmentation-as","title":"PolyFormer: Referring Image Segmentation as Sequential Polygon Generation","date":"2023-02-14","arxiv_id":"2302.07387","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/polyformer-referring-image-segmentation-as#ran","syntology_url":"https://syntology.ai/paper/2302.07387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.07387"}},"official":{"repos":["amazon-science/polygon-transformer"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vlt-vision-language-transformer-and-query","slug":"vlt-vision-language-transformer-and-query","title":"VLT: Vision-Language Transformer and Query Generation for Referring Segmentation","date":"2022-10-28","arxiv_id":"2210.15871","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vlt-vision-language-transformer-and-query#ran","syntology_url":"https://syntology.ai/paper/2210.15871","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.15871"}},"official":{"repos":["henghuiding/Vision-Language-Transformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/glipv2-unifying-localization-and-vision","slug":"glipv2-unifying-localization-and-vision","title":"GLIPv2: Unifying Localization and Vision-Language Understanding","date":"2022-06-12","arxiv_id":"2206.05836","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/glipv2-unifying-localization-and-vision#ran","syntology_url":"https://syntology.ai/paper/2206.05836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05836"}},"official":{"repos":["microsoft/GLIP"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seqtr-a-simple-yet-universal-network-for","slug":"seqtr-a-simple-yet-universal-network-for","title":"SeqTR: A Simple yet Universal Network for Visual Grounding","date":"2022-03-30","arxiv_id":"2203.16265","repositories_listed":3,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/seqtr-a-simple-yet-universal-network-for#ran","syntology_url":"https://syntology.ai/paper/2203.16265","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16265"}},"official":{"repos":["sean-zhuh/seqtr"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/local-global-context-aware-transformer-for","slug":"local-global-context-aware-transformer-for","title":"Local-Global Context Aware Transformer for Language-Guided Video Segmentation","date":"2022-03-18","arxiv_id":"2203.09773","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/local-global-context-aware-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2203.09773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09773"}},"official":{"repos":["leonnnop/locater"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-as-queries-for-referring-video","slug":"language-as-queries-for-referring-video","title":"Language as Queries for Referring Video Object Segmentation","date":"2022-01-03","arxiv_id":"2201.00487","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-as-queries-for-referring-video#ran","syntology_url":"https://syntology.ai/paper/2201.00487","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.00487"}},"official":{"repos":["wjn922/referformer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/prompt-based-multi-modal-image-segmentation","slug":"prompt-based-multi-modal-image-segmentation","title":"Image Segmentation Using Text and Image Prompts","date":"2021-12-18","arxiv_id":"2112.10003","repositories_listed":6,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/prompt-based-multi-modal-image-segmentation#ran","syntology_url":"https://syntology.ai/paper/2112.10003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.10003"}},"official":{"repos":["timojl/clipseg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cris-clip-driven-referring-image-segmentation","slug":"cris-clip-driven-referring-image-segmentation","title":"CRIS: CLIP-Driven Referring Image Segmentation","date":"2021-11-30","arxiv_id":"2111.15174","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/cris-clip-driven-referring-image-segmentation#ran","syntology_url":"https://syntology.ai/paper/2111.15174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.15174"}},"official":{"repos":["DerrickWang005/CRIS.pytorch"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-referring-video-object","slug":"end-to-end-referring-video-object","title":"End-to-End Referring Video Object Segmentation with Multimodal Transformers","date":"2021-11-29","arxiv_id":"2111.14821","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":4,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-referring-video-object#ran","syntology_url":"https://syntology.ai/paper/2111.14821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.14821"}},"official":{"repos":["mttr2021/MTTR"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-grained-vision-language-pre-training","slug":"multi-grained-vision-language-pre-training","title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","date":"2021-11-16","arxiv_id":"2111.08276","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-grained-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2111.08276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.08276"}},"official":{"repos":["zengyan-97/x-vlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/referring-transformer-a-one-step-approach-to","slug":"referring-transformer-a-one-step-approach-to","title":"Referring Transformer: A One-step Approach to Multi-task Visual Grounding","date":"2021-06-06","arxiv_id":"2106.03089","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/referring-transformer-a-one-step-approach-to#ran","syntology_url":"https://syntology.ai/paper/2106.03089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03089"}},"official":{"repos":["ubc-vision/RefTR"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mdetr-modulated-detection-for-end-to-end","slug":"mdetr-modulated-detection-for-end-to-end","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-04-26","arxiv_id":"2104.12763","repositories_listed":5,"syntology":{"n":11,"n_ran":7,"n_constructed":4,"n_ran_checked":6,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/mdetr-modulated-detection-for-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2104.12763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.12763"}},"official":{"repos":["ashkamath/mdetr"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/referring-image-segmentation-via-cross-modal-1","slug":"referring-image-segmentation-via-cross-modal-1","title":"Referring Image Segmentation via Cross-Modal Progressive Comprehension","date":"2020-10-01","arxiv_id":"2010.00514","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/referring-image-segmentation-via-cross-modal-1#ran","syntology_url":"https://syntology.ai/paper/2010.00514","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.00514"}},"official":{"repos":["spyflying/CMPC-Refseg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-task-collaborative-network-for-joint","slug":"multi-task-collaborative-network-for-joint","title":"Multi-task Collaborative Network for Joint Referring Expression Comprehension and Segmentation","date":"2020-03-19","arxiv_id":"2003.08813","repositories_listed":2,"syntology":{"n":13,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/multi-task-collaborative-network-for-joint#ran","syntology_url":"https://syntology.ai/paper/2003.08813","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.08813"}},"official":{"repos":["luogen1996/MCN"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/referring-expression-object-segmentation-with","slug":"referring-expression-object-segmentation-with","title":"Referring Expression Object Segmentation with Caption-Aware Consistency","date":"2019-10-10","arxiv_id":"1910.04748","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/referring-expression-object-segmentation-with#ran","syntology_url":"https://syntology.ai/paper/1910.04748","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.04748"}},"official":{"repos":["wenz116/lang2seg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/clevr-ref-diagnosing-visual-reasoning-with","slug":"clevr-ref-diagnosing-visual-reasoning-with","title":"CLEVR-Ref+: Diagnosing Visual Reasoning with Referring Expressions","date":"2019-01-03","arxiv_id":"1901.00850","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/clevr-ref-diagnosing-visual-reasoning-with#ran","syntology_url":"https://syntology.ai/paper/1901.00850","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.00850"}},"official":null}}],"record_sha256":"793f6ca98e20dea0914e7019ab6dbfcf2728e137ab91bf792ae31699ec335b9e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}