{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/image-retrieval/papers/ran/1","list_of":"/task/image-retrieval","task":"Image Retrieval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":3,"rows_per_page":100,"rows":[1,100],"of":218,"counts":{"archive_papers_tagged":2239,"with_a_code_link":835,"where_syntology_ran_a_sample":218,"not_listed_spam_title":0,"listed":2239,"listed_where_code_ran":218,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":185,"every_run_a_failure_of_syntologys_instrument":33,"listed_with_a_run_with_no_instrument_failure":185,"listed_every_run_a_failure_of_syntologys_instrument":33,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/image-retrieval/papers/ran/1","prev":null,"next":"/task/image-retrieval/papers/ran/2","papers":[{"url":"/paper/ms-dpps-multi-source-determinantal-point","slug":"ms-dpps-multi-source-determinantal-point","title":"MS-DPPs: Multi-Source Determinantal Point Processes for Contextual Diversity Refinement of Composite Attributes in Text to Image Retrieval","date":"2025-07-09","arxiv_id":"2507.06654","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ms-dpps-multi-source-determinantal-point#ran","syntology_url":"https://syntology.ai/paper/2507.06654","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.06654"}},"official":{"repos":["nec-n-sogi/msdpp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/focus-on-local-finding-reliable-1","slug":"focus-on-local-finding-reliable-1","title":"Focus on Local: Finding Reliable Discriminative Regions for Visual Place Recognition","date":"2025-04-14","arxiv_id":"2504.09881","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/focus-on-local-finding-reliable-1#ran","syntology_url":"https://syntology.ai/paper/2504.09881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.09881"}},"official":{"repos":["chenshunpeng/FoL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scale-efficient-training-for-large-datasets-1","slug":"scale-efficient-training-for-large-datasets-1","title":"Scale Efficient Training for Large Datasets","date":"2025-03-17","arxiv_id":"2503.13385","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/scale-efficient-training-for-large-datasets-1#ran","syntology_url":"https://syntology.ai/paper/2503.13385","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13385"}},"official":{"repos":["mrazhou/seta"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/visualwebinstruct-scaling-up-multimodal","slug":"visualwebinstruct-scaling-up-multimodal","title":"VisualWebInstruct: Scaling up Multimodal Instruction Data through Web Search","date":"2025-03-13","arxiv_id":"2503.10582","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/visualwebinstruct-scaling-up-multimodal#ran","syntology_url":"https://syntology.ai/paper/2503.10582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10582"}},"official":null}},{"url":"/paper/re-align-aligning-vision-language-models-via","slug":"re-align-aligning-vision-language-models-via","title":"Re-Align: Aligning Vision Language Models via Retrieval-Augmented Direct Preference Optimization","date":"2025-02-18","arxiv_id":"2502.13146","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/re-align-aligning-vision-language-models-via#ran","syntology_url":"https://syntology.ai/paper/2502.13146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13146"}},"official":{"repos":["taco-group/re-align"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ilias-instance-level-image-retrieval-at-scale","slug":"ilias-instance-level-image-retrieval-at-scale","title":"ILIAS: Instance-Level Image retrieval At Scale","date":"2025-02-17","arxiv_id":"2502.11748","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ilias-instance-level-image-retrieval-at-scale#ran","syntology_url":"https://syntology.ai/paper/2502.11748","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11748"}},"official":null}},{"url":"/paper/mvc-vpr-mutual-learning-of-viewpoint","slug":"mvc-vpr-mutual-learning-of-viewpoint","title":"MVC-VPR: Mutual Learning of Viewpoint Classification and Visual Place Recognition","date":"2024-12-12","arxiv_id":"2412.09199","repositories_listed":0,"syntology":{"n":15,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":15,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/mvc-vpr-mutual-learning-of-viewpoint#ran","syntology_url":"https://syntology.ai/paper/2412.09199","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09199"}},"official":null}},{"url":"/paper/hopfield-fenchel-young-networks-a-unified","slug":"hopfield-fenchel-young-networks-a-unified","title":"Hopfield-Fenchel-Young Networks: A Unified Framework for Associative Memory Retrieval","date":"2024-11-13","arxiv_id":"2411.08590","repositories_listed":1,"syntology":{"n":8,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/hopfield-fenchel-young-networks-a-unified#ran","syntology_url":"https://syntology.ai/paper/2411.08590","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.08590"}},"official":{"repos":["deep-spin/HFYN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/inquire-a-natural-world-text-to-image","slug":"inquire-a-natural-world-text-to-image","title":"INQUIRE: A Natural World Text-to-Image Retrieval Benchmark","date":"2024-11-04","arxiv_id":"2411.02537","repositories_listed":1,"syntology":{"n":25,"n_ran":20,"n_constructed":0,"n_ran_checked":16,"n_instrument":4,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":15,"n_pointer_only":3,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 1 violated, 15 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/inquire-a-natural-world-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2411.02537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02537"}},"official":{"repos":["inquire-benchmark/INQUIRE"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/beyond-text-optimizing-rag-with-multimodal","slug":"beyond-text-optimizing-rag-with-multimodal","title":"Beyond Text: Optimizing RAG with Multimodal Inputs for Industrial Applications","date":"2024-10-29","arxiv_id":"2410.21943","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-text-optimizing-rag-with-multimodal#ran","syntology_url":"https://syntology.ai/paper/2410.21943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21943"}},"official":{"repos":["riedlerm/multimodal_rag_for_industry"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chatsearch-a-dataset-and-a-generative","slug":"chatsearch-a-dataset-and-a-generative","title":"ChatSearch: a Dataset and a Generative Retrieval Model for General Conversational Image Retrieval","date":"2024-10-24","arxiv_id":"2410.18715","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chatsearch-a-dataset-and-a-generative#ran","syntology_url":"https://syntology.ai/paper/2410.18715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18715"}},"official":{"repos":["joez17/chatsearch"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vlad-buff-burst-aware-fast-feature","slug":"vlad-buff-burst-aware-fast-feature","title":"VLAD-BuFF: Burst-aware Fast Feature Aggregation for Visual Place Recognition","date":"2024-09-28","arxiv_id":"2409.19293","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vlad-buff-burst-aware-fast-feature#ran","syntology_url":"https://syntology.ai/paper/2409.19293","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.19293"}},"official":{"repos":["ahmedest61/vlad-buff"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/training-free-zs-cir-via-weighted-modality","slug":"training-free-zs-cir-via-weighted-modality","title":"Training-free Zero-shot Composed Image Retrieval via Weighted Modality Fusion and Similarity","date":"2024-09-07","arxiv_id":"2409.04918","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-free-zs-cir-via-weighted-modality#ran","syntology_url":"https://syntology.ai/paper/2409.04918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.04918"}},"official":{"repos":["whats2000/WeiMoCIR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimizing-clip-models-for-image-retrieval","slug":"optimizing-clip-models-for-image-retrieval","title":"Optimizing CLIP Models for Image Retrieval with Maintained Joint-Embedding Alignment","date":"2024-09-03","arxiv_id":"2409.01936","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":3,"n_no_contract":8,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 3 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/optimizing-clip-models-for-image-retrieval#ran","syntology_url":"https://syntology.ai/paper/2409.01936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.01936"}},"official":{"repos":["Visual-Computing/MCIP"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/unifashion-a-unified-vision-language-model","slug":"unifashion-a-unified-vision-language-model","title":"UniFashion: A Unified Vision-Language Model for Multimodal Fashion Retrieval and Generation","date":"2024-08-21","arxiv_id":"2408.11305","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":5,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/unifashion-a-unified-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2408.11305","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11305"}},"official":{"repos":["xiangyu-mm/unifashion"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-view-image-geo-localization-with","slug":"cross-view-image-geo-localization-with","title":"Cross-view image geo-localization with Panorama-BEV Co-Retrieval Network","date":"2024-08-10","arxiv_id":"2408.05475","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cross-view-image-geo-localization-with#ran","syntology_url":"https://syntology.ai/paper/2408.05475","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.05475"}},"official":{"repos":["yejy53/ep-bev"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-03282","slug":"2408-03282","title":"AMES: Asymmetric and Memory-Efficient Similarity Estimation for Instance-level Retrieval","date":"2024-08-06","arxiv_id":"2408.03282","repositories_listed":1,"syntology":{"n":13,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2408-03282#ran","syntology_url":"https://syntology.ai/paper/2408.03282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03282"}},"official":{"repos":["pavelsuma/ames"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-haystacks-answering-harder-questions","slug":"visual-haystacks-answering-harder-questions","title":"Visual Haystacks: A Vision-Centric Needle-In-A-Haystack Benchmark","date":"2024-07-18","arxiv_id":"2407.13766","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/visual-haystacks-answering-harder-questions#ran","syntology_url":"https://syntology.ai/paper/2407.13766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13766"}},"official":{"repos":["visual-haystacks/vhs_benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/no-train-all-gain-self-supervised-gradients","slug":"no-train-all-gain-self-supervised-gradients","title":"No Train, all Gain: Self-Supervised Gradients Improve Deep Frozen Representations","date":"2024-07-15","arxiv_id":"2407.10964","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/no-train-all-gain-self-supervised-gradients#ran","syntology_url":"https://syntology.ai/paper/2407.10964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10964"}},"official":{"repos":["waltersimoncini/fungivision"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-group-proportional-representation","slug":"multi-group-proportional-representation","title":"Multi-Group Proportional Representation in Retrieval","date":"2024-07-11","arxiv_id":"2407.08571","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-group-proportional-representation#ran","syntology_url":"https://syntology.ai/paper/2407.08571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08571"}},"official":{"repos":["alex-oesterling/multigroup-proportional-representation"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"url":"/paper/learning-from-memory-non-parametric-memory","slug":"learning-from-memory-non-parametric-memory","title":"Learning from Memory: Non-Parametric Memory Augmented Self-Supervised Learning of Visual Features","date":"2024-07-03","arxiv_id":"2407.17486","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-from-memory-non-parametric-memory#ran","syntology_url":"https://syntology.ai/paper/2407.17486","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.17486"}},"official":{"repos":["sthalles/massl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wv-net-a-foundation-model-for-sar-wv-mode","slug":"wv-net-a-foundation-model-for-sar-wv-mode","title":"WV-Net: A foundation model for SAR WV-mode satellite imagery trained using contrastive self-supervised learning on 10 million images","date":"2024-06-26","arxiv_id":"2406.18765","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wv-net-a-foundation-model-for-sar-wv-mode#ran","syntology_url":"https://syntology.ai/paper/2406.18765","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18765"}},"official":{"repos":["hawaii-ai/wvnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/breaking-the-frame-image-retrieval-by-visual","slug":"breaking-the-frame-image-retrieval-by-visual","title":"Breaking the Frame: Visual Place Recognition by Overlap Prediction","date":"2024-06-23","arxiv_id":"2406.16204","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/breaking-the-frame-image-retrieval-by-visual#ran","syntology_url":"https://syntology.ai/paper/2406.16204","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16204"}},"official":{"repos":["weitong8591/vop"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-needle-in-a-haystack-benchmarking","slug":"multimodal-needle-in-a-haystack-benchmarking","title":"Multimodal Needle in a Haystack: Benchmarking Long-Context Capability of Multimodal Large Language Models","date":"2024-06-17","arxiv_id":"2406.11230","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-needle-in-a-haystack-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2406.11230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11230"}},"official":{"repos":["wang-ml-lab/multimodal-needle-in-a-haystack"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bivlc-extending-vision-language","slug":"bivlc-extending-vision-language","title":"BiVLC: Extending Vision-Language Compositionality Evaluation with Text-to-Image Retrieval","date":"2024-06-14","arxiv_id":"2406.09952","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bivlc-extending-vision-language#ran","syntology_url":"https://syntology.ai/paper/2406.09952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09952"}},"official":{"repos":["imirandam/bivlc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/denoisereid-denoising-model-for","slug":"denoisereid-denoising-model-for","title":"DenoiseRep: Denoising Model for Representation Learning","date":"2024-06-13","arxiv_id":"2406.08773","repositories_listed":1,"syntology":{"n":22,"n_ran":17,"n_constructed":2,"n_ran_checked":17,"n_instrument":0,"n_unverified":5,"n_honours":2,"n_violates":1,"n_no_contract":14,"n_pointer_only":6,"phrase":"17 ran (of which 2 constructed an object rather than computing a result; 17 with no instrument failure: 2 honoured, 1 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/denoisereid-denoising-model-for#ran","syntology_url":"https://syntology.ai/paper/2406.08773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08773"}},"official":{"repos":["wangguanan/denoiserep"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":2,"n_ran_no_instrument_failure":17,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/common-and-rare-fundus-diseases","slug":"common-and-rare-fundus-diseases","title":"Enhancing Diagnostic Accuracy in Rare and Common Fundus Diseases with a Knowledge-Rich Vision-Language Model","date":"2024-06-13","arxiv_id":"2406.09317","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/common-and-rare-fundus-diseases#ran","syntology_url":"https://syntology.ai/paper/2406.09317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09317"}},"official":{"repos":["LooKing9218/RetiZero"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/image-textualization-an-automatic-framework","slug":"image-textualization-an-automatic-framework","title":"Image Textualization: An Automatic Framework for Creating Accurate and Detailed Image Descriptions","date":"2024-06-11","arxiv_id":"2406.07502","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/image-textualization-an-automatic-framework#ran","syntology_url":"https://syntology.ai/paper/2406.07502","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07502"}},"official":{"repos":["sterzhang/image-textualization"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/interactive-text-to-image-retrieval-with","slug":"interactive-text-to-image-retrieval-with","title":"Interactive Text-to-Image Retrieval with Large Language Models: A Plug-and-Play Approach","date":"2024-06-05","arxiv_id":"2406.03411","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interactive-text-to-image-retrieval-with#ran","syntology_url":"https://syntology.ai/paper/2406.03411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03411"}},"official":{"repos":["saehyung-lee/plugir"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/decomposing-and-interpreting-image","slug":"decomposing-and-interpreting-image","title":"Decomposing and Interpreting Image Representations via Text in ViTs Beyond CLIP","date":"2024-06-03","arxiv_id":"2406.01583","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/decomposing-and-interpreting-image#ran","syntology_url":"https://syntology.ai/paper/2406.01583","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.01583"}},"official":{"repos":["sriramb-98/vit-decompose"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/emr-merging-tuning-free-high-performance","slug":"emr-merging-tuning-free-high-performance","title":"EMR-Merging: Tuning-Free High-Performance Model Merging","date":"2024-05-23","arxiv_id":"2405.17461","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":5,"n_instrument":8,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":15,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/emr-merging-tuning-free-high-performance#ran","syntology_url":"https://syntology.ai/paper/2405.17461","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17461"}},"official":{"repos":["harveyhuang18/emr_merging"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/isearle-improving-textual-inversion-for-zero","slug":"isearle-improving-textual-inversion-for-zero","title":"iSEARLE: Improving Textual Inversion for Zero-Shot Composed Image Retrieval","date":"2024-05-05","arxiv_id":"2405.02951","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/isearle-improving-textual-inversion-for-zero#ran","syntology_url":"https://syntology.ai/paper/2405.02951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.02951"}},"official":{"repos":["miccunifi/circo"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]}}},{"url":"/paper/spherical-linear-interpolation-and-text","slug":"spherical-linear-interpolation-and-text","title":"Spherical Linear Interpolation and Text-Anchoring for Zero-shot Composed Image Retrieval","date":"2024-05-01","arxiv_id":"2405.00571","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spherical-linear-interpolation-and-text#ran","syntology_url":"https://syntology.ai/paper/2405.00571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.00571"}},"official":null}},{"url":"/paper/efficient-remote-sensing-with-harmonized","slug":"efficient-remote-sensing-with-harmonized","title":"Efficient Remote Sensing with Harmonized Transfer Learning and Modality Alignment","date":"2024-04-28","arxiv_id":"2404.18253","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/efficient-remote-sensing-with-harmonized#ran","syntology_url":"https://syntology.ai/paper/2404.18253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18253"}},"official":{"repos":["seekerhuang/harma"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-composed-image-retrieval-via","slug":"improving-composed-image-retrieval-via","title":"Improving Composed Image Retrieval via Contrastive Learning with Scaling Positives and Negatives","date":"2024-04-17","arxiv_id":"2404.11317","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-composed-image-retrieval-via#ran","syntology_url":"https://syntology.ai/paper/2404.11317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.11317"}},"official":{"repos":["BUAADreamer/SPN4CIR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/soft-prompting-with-graph-of-thought-for","slug":"soft-prompting-with-graph-of-thought-for","title":"Soft-Prompting with Graph-of-Thought for Multi-modal Representation Learning","date":"2024-04-06","arxiv_id":"2404.04538","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/soft-prompting-with-graph-of-thought-for#ran","syntology_url":"https://syntology.ai/paper/2404.04538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04538"}},"official":{"repos":["shishicode/agot"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/on-train-test-class-overlap-and-detection-for","slug":"on-train-test-class-overlap-and-detection-for","title":"On Train-Test Class Overlap and Detection for Image Retrieval","date":"2024-04-01","arxiv_id":"2404.01524","repositories_listed":0,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/on-train-test-class-overlap-and-detection-for#ran","syntology_url":"https://syntology.ai/paper/2404.01524","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01524"}},"official":null}},{"url":"/paper/do-vision-language-models-understand-compound","slug":"do-vision-language-models-understand-compound","title":"Do Vision-Language Models Understand Compound Nouns?","date":"2024-03-30","arxiv_id":"2404.00419","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/do-vision-language-models-understand-compound#ran","syntology_url":"https://syntology.ai/paper/2404.00419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00419"}},"official":{"repos":["sonalkum/compun"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/magiclens-self-supervised-image-retrieval","slug":"magiclens-self-supervised-image-retrieval","title":"MagicLens: Self-Supervised Image Retrieval with Open-Ended Instructions","date":"2024-03-28","arxiv_id":"2403.19651","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/magiclens-self-supervised-image-retrieval#ran","syntology_url":"https://syntology.ai/paper/2403.19651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19651"}},"official":{"repos":["google-deepmind/magiclens"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-enhanced-dual-stream-zero-shot","slug":"knowledge-enhanced-dual-stream-zero-shot","title":"Knowledge-Enhanced Dual-stream Zero-shot Composed Image Retrieval","date":"2024-03-24","arxiv_id":"2403.16005","repositories_listed":1,"syntology":{"n":17,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/knowledge-enhanced-dual-stream-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2403.16005","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16005"}},"official":{"repos":["suoych/keds"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/long-clip-unlocking-the-long-text-capability","slug":"long-clip-unlocking-the-long-text-capability","title":"Long-CLIP: Unlocking the Long-Text Capability of CLIP","date":"2024-03-22","arxiv_id":"2403.15378","repositories_listed":1,"syntology":{"n":15,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":7,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/long-clip-unlocking-the-long-text-capability#ran","syntology_url":"https://syntology.ai/paper/2403.15378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15378"}},"official":{"repos":["beichenzbc/long-clip"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/mindeye2-shared-subject-models-enable-fmri-to","slug":"mindeye2-shared-subject-models-enable-fmri-to","title":"MindEye2: Shared-Subject Models Enable fMRI-To-Image With 1 Hour of Data","date":"2024-03-17","arxiv_id":"2403.11207","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":11,"n_pointer_only":2,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 3 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mindeye2-shared-subject-models-enable-fmri-to#ran","syntology_url":"https://syntology.ai/paper/2403.11207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.11207"}},"official":{"repos":["medarc-ai/mindeyev2"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/paperclip-associating-astronomical","slug":"paperclip-associating-astronomical","title":"PAPERCLIP: Associating Astronomical Observations and Natural Language with Multi-Modal Models","date":"2024-03-13","arxiv_id":"2403.08851","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/paperclip-associating-astronomical#ran","syntology_url":"https://syntology.ai/paper/2403.08851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.08851"}},"official":{"repos":["smsharma/paperclip-hubble"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/earthloc-astronaut-photography-localization","slug":"earthloc-astronaut-photography-localization","title":"EarthLoc: Astronaut Photography Localization by Indexing Earth from Space","date":"2024-03-11","arxiv_id":"2403.06758","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/earthloc-astronaut-photography-localization#ran","syntology_url":"https://syntology.ai/paper/2403.06758","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.06758"}},"official":{"repos":["gmberton/earthloc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-learned-sparse-retrieval-with","slug":"multimodal-learned-sparse-retrieval-with","title":"Multimodal Learned Sparse Retrieval with Probabilistic Expansion Control","date":"2024-02-27","arxiv_id":"2402.17535","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-learned-sparse-retrieval-with#ran","syntology_url":"https://syntology.ai/paper/2402.17535","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17535"}},"official":{"repos":["thongnt99/lsr-multimodal"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-tuning-clip-text-encoders-with-two-step","slug":"fine-tuning-clip-text-encoders-with-two-step","title":"Fine-tuning CLIP Text Encoders with Two-step Paraphrasing","date":"2024-02-23","arxiv_id":"2402.15120","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fine-tuning-clip-text-encoders-with-two-step#ran","syntology_url":"https://syntology.ai/paper/2402.15120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15120"}},"official":null}},{"url":"/paper/learning-semantic-proxies-from-visual-prompts","slug":"learning-semantic-proxies-from-visual-prompts","title":"Learning Semantic Proxies from Visual Prompts for Parameter-Efficient Fine-Tuning in Deep Metric Learning","date":"2024-02-04","arxiv_id":"2402.02340","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/learning-semantic-proxies-from-visual-prompts#ran","syntology_url":"https://syntology.ai/paper/2402.02340","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02340"}},"official":{"repos":["noahsark/parameterefficient-dml"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/internvl-scaling-up-vision-foundation-models","slug":"internvl-scaling-up-vision-foundation-models","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","date":"2023-12-21","arxiv_id":"2312.14238","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/internvl-scaling-up-vision-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2312.14238","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14238"}},"official":{"repos":["opengvlab/internvl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/let-all-be-whitened-multi-teacher","slug":"let-all-be-whitened-multi-teacher","title":"Let All be Whitened: Multi-teacher Distillation for Efficient Visual Retrieval","date":"2023-12-15","arxiv_id":"2312.09716","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/let-all-be-whitened-multi-teacher#ran","syntology_url":"https://syntology.ai/paper/2312.09716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.09716"}},"official":{"repos":["maryeon/whiten_mtd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/contextually-affinitive-neighborhood-refinery-1","slug":"contextually-affinitive-neighborhood-refinery-1","title":"Contextually Affinitive Neighborhood Refinery for Deep Clustering","date":"2023-12-12","arxiv_id":"2312.07806","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contextually-affinitive-neighborhood-refinery-1#ran","syntology_url":"https://syntology.ai/paper/2312.07806","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07806"}},"official":{"repos":["cly234/deepclustering-connr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lite-mind-towards-efficient-and-versatile","slug":"lite-mind-towards-efficient-and-versatile","title":"Lite-Mind: Towards Efficient and Robust Brain Representation Network","date":"2023-12-06","arxiv_id":"2312.03781","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lite-mind-towards-efficient-and-versatile#ran","syntology_url":"https://syntology.ai/paper/2312.03781","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03781"}},"official":{"repos":["gongzix/lite-mind"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/freestyleret-retrieving-images-from-style","slug":"freestyleret-retrieving-images-from-style","title":"FreestyleRet: Retrieving Images from Style-Diversified Queries","date":"2023-12-05","arxiv_id":"2312.02428","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/freestyleret-retrieving-images-from-style#ran","syntology_url":"https://syntology.ai/paper/2312.02428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02428"}},"official":{"repos":["curisejia/freestyleret"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/language-only-efficient-training-of-zero-shot","slug":"language-only-efficient-training-of-zero-shot","title":"Language-only Efficient Training of Zero-shot Composed Image Retrieval","date":"2023-12-04","arxiv_id":"2312.01998","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/language-only-efficient-training-of-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2312.01998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01998"}},"official":{"repos":["navervision/lincir"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/synthesize-diagnose-and-optimize-towards-fine","slug":"synthesize-diagnose-and-optimize-towards-fine","title":"Synthesize, Diagnose, and Optimize: Towards Fine-Grained Vision-Language Understanding","date":"2023-11-30","arxiv_id":"2312.00081","repositories_listed":1,"syntology":{"n":11,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":11,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/synthesize-diagnose-and-optimize-towards-fine#ran","syntology_url":"https://syntology.ai/paper/2312.00081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.00081"}},"official":{"repos":["wjpoom/spec"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/removing-nsfw-concepts-from-vision-and","slug":"removing-nsfw-concepts-from-vision-and","title":"Safe-CLIP: Removing NSFW Concepts from Vision-and-Language Models","date":"2023-11-27","arxiv_id":"2311.16254","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/removing-nsfw-concepts-from-vision-and#ran","syntology_url":"https://syntology.ai/paper/2311.16254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.16254"}},"official":{"repos":["aimagelab/safe-clip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/pretrain-like-you-inference-masked-tuning","slug":"pretrain-like-you-inference-masked-tuning","title":"Pretrain like Your Inference: Masked Tuning Improves Zero-Shot Composed Image Retrieval","date":"2023-11-13","arxiv_id":"2311.07622","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pretrain-like-you-inference-masked-tuning#ran","syntology_url":"https://syntology.ai/paper/2311.07622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07622"}},"official":{"repos":["Chen-Junyang-cn/PLI"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lipsim-a-provably-robust-perceptual","slug":"lipsim-a-provably-robust-perceptual","title":"LipSim: A Provably Robust Perceptual Similarity Metric","date":"2023-10-27","arxiv_id":"2310.18274","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lipsim-a-provably-robust-perceptual#ran","syntology_url":"https://syntology.ai/paper/2310.18274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18274"}},"official":{"repos":["saraghazanfari/lipsim"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-and-multimodal","slug":"large-language-models-and-multimodal","title":"Large Language Models and Multimodal Retrieval for Visual Word Sense Disambiguation","date":"2023-10-21","arxiv_id":"2310.14025","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-and-multimodal#ran","syntology_url":"https://syntology.ai/paper/2310.14025","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14025"}},"official":{"repos":["anastasiakrith/multimodal-retrieval-for-vwsd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/representation-learning-via-consistent-2","slug":"representation-learning-via-consistent-2","title":"Representation Learning via Consistent Assignment of Views over Random Partitions","date":"2023-10-19","arxiv_id":"2310.12692","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/representation-learning-via-consistent-2#ran","syntology_url":"https://syntology.ai/paper/2310.12692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12692"}},"official":{"repos":["sthalles/carp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-by-language-for-training-free","slug":"vision-by-language-for-training-free","title":"Vision-by-Language for Training-Free Compositional Image Retrieval","date":"2023-10-13","arxiv_id":"2310.09291","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vision-by-language-for-training-free#ran","syntology_url":"https://syntology.ai/paper/2310.09291","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09291"}},"official":{"repos":["explainableml/vision_by_language"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/pairwise-similarity-learning-is-simple-1","slug":"pairwise-similarity-learning-is-simple-1","title":"Pairwise Similarity Learning is SimPLE","date":"2023-10-13","arxiv_id":"2310.09449","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":1,"n_ran_checked":4,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pairwise-similarity-learning-is-simple-1#ran","syntology_url":"https://syntology.ai/paper/2310.09449","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09449"}},"official":{"repos":["ydwen/opensphere"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sentence-level-prompts-benefit-composed-image","slug":"sentence-level-prompts-benefit-composed-image","title":"Sentence-level Prompts Benefit Composed Image Retrieval","date":"2023-10-09","arxiv_id":"2310.05473","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sentence-level-prompts-benefit-composed-image#ran","syntology_url":"https://syntology.ai/paper/2310.05473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05473"}},"official":{"repos":["chunmeifeng/sprc"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cifar-10-warehouse-broad-and-more-realistic","slug":"cifar-10-warehouse-broad-and-more-realistic","title":"CIFAR-10-Warehouse: Broad and More Realistic Testbeds in Model Generalization Analysis","date":"2023-10-06","arxiv_id":"2310.04414","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cifar-10-warehouse-broad-and-more-realistic#ran","syntology_url":"https://syntology.ai/paper/2310.04414","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04414"}},"official":null}},{"url":"/paper/context-i2w-mapping-images-to-context","slug":"context-i2w-mapping-images-to-context","title":"Context-I2W: Mapping Images to Context-dependent Words for Accurate Zero-Shot Composed Image Retrieval","date":"2023-09-28","arxiv_id":"2309.16137","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/context-i2w-mapping-images-to-context#ran","syntology_url":"https://syntology.ai/paper/2309.16137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16137"}},"official":{"repos":["pter61/context-i2w"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/forb-a-flat-object-retrieval-benchmark-for-1","slug":"forb-a-flat-object-retrieval-benchmark-for-1","title":"FORB: A Flat Object Retrieval Benchmark for Universal Image Embedding","date":"2023-09-28","arxiv_id":"2309.16249","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/forb-a-flat-object-retrieval-benchmark-for-1#ran","syntology_url":"https://syntology.ai/paper/2309.16249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16249"}},"official":{"repos":["pxiangwu/forb"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dark-side-augmentation-generating-diverse-1","slug":"dark-side-augmentation-generating-diverse-1","title":"Dark Side Augmentation: Generating Diverse Night Examples for Metric Learning","date":"2023-09-28","arxiv_id":"2309.16351","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dark-side-augmentation-generating-diverse-1#ran","syntology_url":"https://syntology.ai/paper/2309.16351","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16351"}},"official":{"repos":["mohwald/gandtr"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/geoclip-clip-inspired-alignment-between","slug":"geoclip-clip-inspired-alignment-between","title":"GeoCLIP: Clip-Inspired Alignment between Locations and Images for Effective Worldwide Geo-localization","date":"2023-09-27","arxiv_id":"2309.16020","repositories_listed":3,"syntology":{"n":13,"n_ran":5,"n_constructed":2,"n_ran_checked":5,"n_instrument":0,"n_unverified":8,"n_honours":3,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 3 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/geoclip-clip-inspired-alignment-between#ran","syntology_url":"https://syntology.ai/paper/2309.16020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16020"}},"official":{"repos":["VicenteVivan/geo-clip"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/keep-it-simpool-who-said-supervised","slug":"keep-it-simpool-who-said-supervised","title":"Keep It SimPool: Who Said Supervised Transformers Suffer from Attention Deficit?","date":"2023-09-13","arxiv_id":"2309.06891","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/keep-it-simpool-who-said-supervised#ran","syntology_url":"https://syntology.ai/paper/2309.06891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06891"}},"official":{"repos":["billpsomas/simpool"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-content-based-pixel-retrieval-in","slug":"towards-content-based-pixel-retrieval-in","title":"Towards Content-based Pixel Retrieval in Revisited Oxford and Paris","date":"2023-09-11","arxiv_id":"2309.05438","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/towards-content-based-pixel-retrieval-in#ran","syntology_url":"https://syntology.ai/paper/2309.05438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05438"}},"official":{"repos":["anguoyuan/pixel_retrieval-segmented_instance_retrieval"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/nllb-clip-train-performant-multilingual-image","slug":"nllb-clip-train-performant-multilingual-image","title":"NLLB-CLIP -- train performant multilingual image retrieval model on a budget","date":"2023-09-04","arxiv_id":"2309.01859","repositories_listed":4,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/nllb-clip-train-performant-multilingual-image#ran","syntology_url":"https://syntology.ai/paper/2309.01859","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01859"}},"official":null}},{"url":"/paper/covr-learning-composed-video-retrieval-from","slug":"covr-learning-composed-video-retrieval-from","title":"CoVR-2: Automatic Data Construction for Composed Video Retrieval","date":"2023-08-28","arxiv_id":"2308.14746","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/covr-learning-composed-video-retrieval-from#ran","syntology_url":"https://syntology.ai/paper/2308.14746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14746"}},"official":{"repos":["lucas-ventura/CoVR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/composed-image-retrieval-using-contrastive","slug":"composed-image-retrieval-using-contrastive","title":"Composed Image Retrieval using Contrastive Learning and Task-oriented CLIP-based Features","date":"2023-08-22","arxiv_id":"2308.11485","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/composed-image-retrieval-using-contrastive#ran","syntology_url":"https://syntology.ai/paper/2308.11485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11485"}},"official":{"repos":["ABaldrati/CLIP4Cir"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/eigenplaces-training-viewpoint-robust-models","slug":"eigenplaces-training-viewpoint-robust-models","title":"EigenPlaces: Training Viewpoint Robust Models for Visual Place Recognition","date":"2023-08-21","arxiv_id":"2308.10832","repositories_listed":4,"syntology":{"n":13,"n_ran":12,"n_constructed":2,"n_ran_checked":9,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"12 ran (of which 2 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/eigenplaces-training-viewpoint-robust-models#ran","syntology_url":"https://syntology.ai/paper/2308.10832","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10832"}},"official":{"repos":["gmberton/auto_vpr","gmberton/eigenplaces"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":2,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/global-features-are-all-you-need-for-image","slug":"global-features-are-all-you-need-for-image","title":"Global Features are All You Need for Image Retrieval and Reranking","date":"2023-08-14","arxiv_id":"2308.06954","repositories_listed":2,"syntology":{"n":18,"n_ran":14,"n_constructed":9,"n_ran_checked":12,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"14 ran (of which 9 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/global-features-are-all-you-need-for-image#ran","syntology_url":"https://syntology.ai/paper/2308.06954","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06954"}},"official":{"repos":["shihaoshao-gh/superglobal"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":9,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/coarse-to-fine-learning-compact","slug":"coarse-to-fine-learning-compact","title":"Coarse-to-Fine: Learning Compact Discriminative Representation for Single-Stage Image Retrieval","date":"2023-08-08","arxiv_id":"2308.04008","repositories_listed":1,"syntology":{"n":21,"n_ran":18,"n_constructed":0,"n_ran_checked":17,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":11,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/coarse-to-fine-learning-compact#ran","syntology_url":"https://syntology.ai/paper/2308.04008","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04008"}},"official":{"repos":["bassyess/cfcd"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/anyloc-towards-universal-visual-place","slug":"anyloc-towards-universal-visual-place","title":"AnyLoc: Towards Universal Visual Place Recognition","date":"2023-08-01","arxiv_id":"2308.00688","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/anyloc-towards-universal-visual-place#ran","syntology_url":"https://syntology.ai/paper/2308.00688","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.00688"}},"official":{"repos":["AnyLoc/AnyLoc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rs5m-a-large-scale-vision-language-dataset","slug":"rs5m-a-large-scale-vision-language-dataset","title":"RS5M and GeoRSCLIP: A Large Scale Vision-Language Dataset and A Large Vision-Language Model for Remote Sensing","date":"2023-06-20","arxiv_id":"2306.11300","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rs5m-a-large-scale-vision-language-dataset#ran","syntology_url":"https://syntology.ai/paper/2306.11300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11300"}},"official":{"repos":["om-ai-lab/rs5m"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mofi-learning-image-representations-from","slug":"mofi-learning-image-representations-from","title":"MOFI: Learning Image Representations from Noisy Entity Annotated Images","date":"2023-06-13","arxiv_id":"2306.07952","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mofi-learning-image-representations-from#ran","syntology_url":"https://syntology.ai/paper/2306.07952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07952"}},"official":{"repos":["apple/ml-mofi"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crossget-cross-guided-ensemble-of-tokens-for","slug":"crossget-cross-guided-ensemble-of-tokens-for","title":"CrossGET: Cross-Guided Ensemble of Tokens for Accelerating Vision-Language Transformers","date":"2023-05-27","arxiv_id":"2305.17455","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crossget-cross-guided-ensemble-of-tokens-for#ran","syntology_url":"https://syntology.ai/paper/2305.17455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17455"}},"official":{"repos":["sdc17/crossget"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/candidate-set-re-ranking-for-composed-image","slug":"candidate-set-re-ranking-for-composed-image","title":"Candidate Set Re-ranking for Composed Image Retrieval with Dual Multi-modal Encoder","date":"2023-05-25","arxiv_id":"2305.16304","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/candidate-set-re-ranking-for-composed-image#ran","syntology_url":"https://syntology.ai/paper/2305.16304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16304"}},"official":{"repos":["Cuberick-Orion/Candidate-Reranking-CIR"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/edis-entity-driven-image-search-over","slug":"edis-entity-driven-image-search-over","title":"EDIS: Entity-Driven Image Search over Multimodal Web Content","date":"2023-05-23","arxiv_id":"2305.13631","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/edis-entity-driven-image-search-over#ran","syntology_url":"https://syntology.ai/paper/2305.13631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13631"}},"official":{"repos":["emerisly/edis"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-instruction-tuning-1","slug":"visual-instruction-tuning-1","title":"Visual Instruction Tuning","date":"2023-04-17","arxiv_id":"2304.08485","repositories_listed":13,"syntology":{"n":51,"n_ran":16,"n_constructed":6,"n_ran_checked":8,"n_instrument":8,"n_unverified":35,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"16 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 8 where Syntology's instrument failed) · 35 unverified","sample_list":"/paper/visual-instruction-tuning-1#ran","syntology_url":"https://syntology.ai/paper/2304.08485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08485"}},"official":{"repos":["haotian-liu/LLaVA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":8,"ran_from_kinds":["community","listed","named_in_paper","official"]}}},{"url":"/paper/dinov2-learning-robust-visual-features","slug":"dinov2-learning-robust-visual-features","title":"DINOv2: Learning Robust Visual Features without Supervision","date":"2023-04-14","arxiv_id":"2304.07193","repositories_listed":26,"syntology":{"n":46,"n_ran":39,"n_constructed":13,"n_ran_checked":35,"n_instrument":4,"n_unverified":7,"n_honours":2,"n_violates":0,"n_no_contract":33,"n_pointer_only":12,"phrase":"39 ran (of which 13 constructed an object rather than computing a result; 35 with no instrument failure: 2 honoured, 0 violated, 33 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/dinov2-learning-robust-visual-features#ran","syntology_url":"https://syntology.ai/paper/2304.07193","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.07193"}},"official":{"repos":["facebookresearch/dinov2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/unicom-universal-and-compact-representation","slug":"unicom-universal-and-compact-representation","title":"Unicom: Universal and Compact Representation Learning for Image Retrieval","date":"2023-04-12","arxiv_id":"2304.05884","repositories_listed":3,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/unicom-universal-and-compact-representation#ran","syntology_url":"https://syntology.ai/paper/2304.05884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.05884"}},"official":{"repos":["deepglint/unicom"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/mammut-a-simple-architecture-for-joint","slug":"mammut-a-simple-architecture-for-joint","title":"MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks","date":"2023-03-29","arxiv_id":"2303.16839","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mammut-a-simple-architecture-for-joint#ran","syntology_url":"https://syntology.ai/paper/2303.16839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16839"}},"official":null}},{"url":"/paper/zero-shot-everything-sketch-based-image","slug":"zero-shot-everything-sketch-based-image","title":"Zero-Shot Everything Sketch-Based Image Retrieval, and in Explainable Style","date":"2023-03-25","arxiv_id":"2303.14348","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":9,"n_ran_checked":10,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"12 ran (of which 9 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/zero-shot-everything-sketch-based-image#ran","syntology_url":"https://syntology.ai/paper/2303.14348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.14348"}},"official":{"repos":["buptlinfy/zse-sbir"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":9,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/aligning-step-by-step-instructional-diagrams","slug":"aligning-step-by-step-instructional-diagrams","title":"Aligning Step-by-Step Instructional Diagrams to Video Demonstrations","date":"2023-03-24","arxiv_id":"2303.13800","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aligning-step-by-step-instructional-diagrams#ran","syntology_url":"https://syntology.ai/paper/2303.13800","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13800"}},"official":{"repos":["DavidZhang73/AssemblyVideoManualAlignment"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/limitr-leveraging-local-information-for","slug":"limitr-leveraging-local-information-for","title":"LIMITR: Leveraging Local Information for Medical Image-Text Representation","date":"2023-03-21","arxiv_id":"2303.11755","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/limitr-leveraging-local-information-for#ran","syntology_url":"https://syntology.ai/paper/2303.11755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11755"}},"official":null}},{"url":"/paper/compodiff-versatile-composed-image-retrieval","slug":"compodiff-versatile-composed-image-retrieval","title":"CompoDiff: Versatile Composed Image Retrieval With Latent Diffusion","date":"2023-03-21","arxiv_id":"2303.11916","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/compodiff-versatile-composed-image-retrieval#ran","syntology_url":"https://syntology.ai/paper/2303.11916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11916"}},"official":{"repos":["navervision/compodiff"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/gpt-4-technical-report-1","slug":"gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","arxiv_id":"2303.08774","repositories_listed":11,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpt-4-technical-report-1#ran","syntology_url":"https://syntology.ai/paper/2303.08774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08774"}},"official":{"repos":["openai/evals"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/data-free-sketch-based-image-retrieval","slug":"data-free-sketch-based-image-retrieval","title":"Data-Free Sketch-Based Image Retrieval","date":"2023-03-14","arxiv_id":"2303.07775","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/data-free-sketch-based-image-retrieval#ran","syntology_url":"https://syntology.ai/paper/2303.07775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.07775"}},"official":{"repos":["abhrac/data-free-sbir"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pic2word-mapping-pictures-to-words-for-zero","slug":"pic2word-mapping-pictures-to-words-for-zero","title":"Pic2Word: Mapping Pictures to Words for Zero-shot Composed Image Retrieval","date":"2023-02-06","arxiv_id":"2302.03084","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pic2word-mapping-pictures-to-words-for-zero#ran","syntology_url":"https://syntology.ai/paper/2302.03084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.03084"}},"official":{"repos":["google-research/composed_image_retrieval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mplug-2-a-modularized-multi-modal-foundation","slug":"mplug-2-a-modularized-multi-modal-foundation","title":"mPLUG-2: A Modularized Multi-modal Foundation Model Across Text, Image and Video","date":"2023-02-01","arxiv_id":"2302.00402","repositories_listed":4,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":9,"n_instrument":8,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mplug-2-a-modularized-multi-modal-foundation#ran","syntology_url":"https://syntology.ai/paper/2302.00402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00402"}},"official":{"repos":["alibaba/AliceMind"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/grounding-language-models-to-images-for","slug":"grounding-language-models-to-images-for","title":"Grounding Language Models to Images for Multimodal Inputs and Outputs","date":"2023-01-31","arxiv_id":"2301.13823","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/grounding-language-models-to-images-for#ran","syntology_url":"https://syntology.ai/paper/2301.13823","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.13823"}},"official":{"repos":["kohjingyu/fromage"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/blip-2-bootstrapping-language-image-pre","slug":"blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","arxiv_id":"2301.12597","repositories_listed":17,"syntology":{"n":8,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/blip-2-bootstrapping-language-image-pre#ran","syntology_url":"https://syntology.ai/paper/2301.12597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12597"}},"official":{"repos":["salesforce/lavis"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/noise-aware-learning-from-web-crawled-image","slug":"noise-aware-learning-from-web-crawled-image","title":"Noise-aware Learning from Web-crawled Image-Text Data for Image Captioning","date":"2022-12-27","arxiv_id":"2212.13563","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/noise-aware-learning-from-web-crawled-image#ran","syntology_url":"https://syntology.ai/paper/2212.13563","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.13563"}},"official":{"repos":["kakaobrain/noc"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/crepe-can-vision-language-foundation-models","slug":"crepe-can-vision-language-foundation-models","title":"CREPE: Can Vision-Language Foundation Models Reason Compositionally?","date":"2022-12-13","arxiv_id":"2212.07796","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/crepe-can-vision-language-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2212.07796","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.07796"}},"official":{"repos":["raivnlab/crepe"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/roboflow-100-a-rich-multi-domain-object","slug":"roboflow-100-a-rich-multi-domain-object","title":"Roboflow 100: A Rich, Multi-Domain Object Detection Benchmark","date":"2022-11-24","arxiv_id":"2211.13523","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/roboflow-100-a-rich-multi-domain-object#ran","syntology_url":"https://syntology.ai/paper/2211.13523","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.13523"}},"official":{"repos":["roboflow-ai/roboflow-100-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/composed-image-retrieval-with-text-feedback","slug":"composed-image-retrieval-with-text-feedback","title":"Composed Image Retrieval with Text Feedback via Multi-grained Uncertainty Regularization","date":"2022-11-14","arxiv_id":"2211.07394","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":6,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/composed-image-retrieval-with-text-feedback#ran","syntology_url":"https://syntology.ai/paper/2211.07394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.07394"}},"official":{"repos":["Monoxide-Chen/uncertainty_retrieval"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/altclip-altering-the-language-encoder-in-clip","slug":"altclip-altering-the-language-encoder-in-clip","title":"AltCLIP: Altering the Language Encoder in CLIP for Extended Language Capabilities","date":"2022-11-12","arxiv_id":"2211.06679","repositories_listed":2,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/altclip-altering-the-language-encoder-in-clip#ran","syntology_url":"https://syntology.ai/paper/2211.06679","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.06679"}},"official":{"repos":["flagai-open/flagai"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"808afbbc46b7a03f58bd56c5ef2e76c5e8011e8dcd365c072526e31a10581437","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}