{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/image-retrieval/papers/4","list_of":"/task/image-retrieval","task":"Image Retrieval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":23,"rows_per_page":100,"rows":[301,400],"of":2239,"counts":{"archive_papers_tagged":2239,"with_a_code_link":835,"where_syntology_ran_a_sample":218,"not_listed_spam_title":0,"listed":2239,"listed_where_code_ran":218,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":185,"every_run_a_failure_of_syntologys_instrument":33,"listed_with_a_run_with_no_instrument_failure":185,"listed_every_run_a_failure_of_syntologys_instrument":33,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/image-retrieval","prev":"/task/image-retrieval/papers/3","next":"/task/image-retrieval/papers/5","papers":[{"url":"/paper/cbvs-a-large-scale-chinese-image-text","slug":"cbvs-a-large-scale-chinese-image-text","title":"CBVS: A Large-Scale Chinese Image-Text Benchmark for Real-World Short Video Search Scenarios","date":"2024-01-19","arxiv_id":"2401.10475","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-masked-autoencoders-for-sensor","slug":"exploring-masked-autoencoders-for-sensor","title":"Exploring Masked Autoencoders for Sensor-Agnostic Image Retrieval in Remote Sensing","date":"2024-01-15","arxiv_id":"2401.07782","repositories_listed":1,"syntology":null},{"url":"/paper/hihpq-hierarchical-hyperbolic-product","slug":"hihpq-hierarchical-hyperbolic-product","title":"HiHPQ: Hierarchical Hyperbolic Product Quantization for Unsupervised Image Retrieval","date":"2024-01-14","arxiv_id":"2401.07212","repositories_listed":1,"syntology":null},{"url":"/paper/modality-aware-representation-learning-for","slug":"modality-aware-representation-learning-for","title":"Modality-Aware Representation Learning for Zero-shot Sketch-based Image Retrieval","date":"2024-01-10","arxiv_id":"2401.04860","repositories_listed":1,"syntology":null},{"url":"/paper/analysis-and-validation-of-image-search","slug":"analysis-and-validation-of-image-search","title":"Analysis and Validation of Image Search Engines in Histopathology","date":"2024-01-06","arxiv_id":"2401.03271","repositories_listed":1,"syntology":null},{"url":"/paper/concon-chi-concept-context-chimera-benchmark","slug":"concon-chi-concept-context-chimera-benchmark","title":"ConCon-Chi: Concept-Context Chimera Benchmark for Personalized Vision-Language Tasks","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/d3still-decoupled-differential-distillation","slug":"d3still-decoupled-differential-distillation","title":"D3still: Decoupled Differential Distillation for Asymmetric Image Retrieval","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/language-only-training-of-zero-shot-composed","slug":"language-only-training-of-zero-shot-composed","title":"Language-only Training of Zero-shot Composed Image Retrieval","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/linguistic-aware-patch-slimming-framework-for","slug":"linguistic-aware-patch-slimming-framework-for","title":"Linguistic-Aware Patch Slimming Framework for Fine-grained Cross-Modal Alignment","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-granularity-representation-learning-for","slug":"multi-granularity-representation-learning-for","title":"Multi-Granularity Representation Learning for Sketch-based Dynamic Face Image Retrieval","date":"2023-12-31","arxiv_id":"2401.00371","repositories_listed":1,"syntology":null},{"url":"/paper/bev-cv-birds-eye-view-transform-for-cross","slug":"bev-cv-birds-eye-view-transform-for-cross","title":"BEV-CV: Birds-Eye-View Transform for Cross-View Geo-Localisation","date":"2023-12-23","arxiv_id":"2312.15363","repositories_listed":1,"syntology":null},{"url":"/paper/gemini-a-family-of-highly-capable-multimodal-1","slug":"gemini-a-family-of-highly-capable-multimodal-1","title":"Gemini: A Family of Highly Capable Multimodal Models","date":"2023-12-19","arxiv_id":"2312.11805","repositories_listed":1,"syntology":null},{"url":"/paper/vqa4cir-boosting-composed-image-retrieval","slug":"vqa4cir-boosting-composed-image-retrieval","title":"VQA4CIR: Boosting Composed Image Retrieval with Visual Question Answering","date":"2023-12-19","arxiv_id":"2312.12273","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-image-retrieval-with-few-shot","slug":"advancing-image-retrieval-with-few-shot","title":"Advancing Image Retrieval with Few-Shot Learning and Relevance Feedback","date":"2023-12-18","arxiv_id":"2312.11078","repositories_listed":1,"syntology":null},{"url":"/paper/symmetrical-bidirectional-knowledge-alignment","slug":"symmetrical-bidirectional-knowledge-alignment","title":"Symmetrical Bidirectional Knowledge Alignment for Zero-Shot Sketch-Based Image Retrieval","date":"2023-12-16","arxiv_id":"2312.10320","repositories_listed":1,"syntology":null},{"url":"/paper/let-all-be-whitened-multi-teacher","slug":"let-all-be-whitened-multi-teacher","title":"Let All be Whitened: Multi-teacher Distillation for Efficient Visual Retrieval","date":"2023-12-15","arxiv_id":"2312.09716","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/let-all-be-whitened-multi-teacher#ran","syntology_url":"https://syntology.ai/paper/2312.09716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.09716"}},"official":{"repos":["maryeon/whiten_mtd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/collapse-oriented-adversarial-training-with","slug":"collapse-oriented-adversarial-training-with","title":"Collapse-Aware Triplet Decoupling for Adversarially Robust Image Retrieval","date":"2023-12-12","arxiv_id":"2312.07364","repositories_listed":1,"syntology":null},{"url":"/paper/contextually-affinitive-neighborhood-refinery-1","slug":"contextually-affinitive-neighborhood-refinery-1","title":"Contextually Affinitive Neighborhood Refinery for Deep Clustering","date":"2023-12-12","arxiv_id":"2312.07806","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contextually-affinitive-neighborhood-refinery-1#ran","syntology_url":"https://syntology.ai/paper/2312.07806","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07806"}},"official":{"repos":["cly234/deepclustering-connr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-weighted-combiner-for-mixed-modal","slug":"dynamic-weighted-combiner-for-mixed-modal","title":"Dynamic Weighted Combiner for Mixed-Modal Image Retrieval","date":"2023-12-11","arxiv_id":"2312.06179","repositories_listed":1,"syntology":null},{"url":"/paper/lite-mind-towards-efficient-and-versatile","slug":"lite-mind-towards-efficient-and-versatile","title":"Lite-Mind: Towards Efficient and Robust Brain Representation Network","date":"2023-12-06","arxiv_id":"2312.03781","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lite-mind-towards-efficient-and-versatile#ran","syntology_url":"https://syntology.ai/paper/2312.03781","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03781"}},"official":{"repos":["gongzix/lite-mind"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/freestyleret-retrieving-images-from-style","slug":"freestyleret-retrieving-images-from-style","title":"FreestyleRet: Retrieving Images from Style-Diversified Queries","date":"2023-12-05","arxiv_id":"2312.02428","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/freestyleret-retrieving-images-from-style#ran","syntology_url":"https://syntology.ai/paper/2312.02428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02428"}},"official":{"repos":["curisejia/freestyleret"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/language-only-efficient-training-of-zero-shot","slug":"language-only-efficient-training-of-zero-shot","title":"Language-only Efficient Training of Zero-shot Composed Image Retrieval","date":"2023-12-04","arxiv_id":"2312.01998","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/language-only-efficient-training-of-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2312.01998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01998"}},"official":{"repos":["navervision/lincir"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/grounding-everything-emerging-localization","slug":"grounding-everything-emerging-localization","title":"Grounding Everything: Emerging Localization Properties in Vision-Language Transformers","date":"2023-12-01","arxiv_id":"2312.00878","repositories_listed":1,"syntology":null},{"url":"/paper/hkust-at-semeval-2023-task-1-visual-word","slug":"hkust-at-semeval-2023-task-1-visual-word","title":"HKUST at SemEval-2023 Task 1: Visual Word Sense Disambiguation with Context Augmentation and Visual Assistance","date":"2023-11-30","arxiv_id":"2311.18273","repositories_listed":1,"syntology":null},{"url":"/paper/label-efficient-training-of-small-task","slug":"label-efficient-training-of-small-task","title":"Knowledge Transfer from Vision Foundation Models for Efficient Training of Small Task-specific Models","date":"2023-11-30","arxiv_id":"2311.18237","repositories_listed":1,"syntology":null},{"url":"/paper/synthesize-diagnose-and-optimize-towards-fine","slug":"synthesize-diagnose-and-optimize-towards-fine","title":"Synthesize, Diagnose, and Optimize: Towards Fine-Grained Vision-Language Understanding","date":"2023-11-30","arxiv_id":"2312.00081","repositories_listed":1,"syntology":{"n":11,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":11,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/synthesize-diagnose-and-optimize-towards-fine#ran","syntology_url":"https://syntology.ai/paper/2312.00081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.00081"}},"official":{"repos":["wjpoom/spec"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/removing-nsfw-concepts-from-vision-and","slug":"removing-nsfw-concepts-from-vision-and","title":"Safe-CLIP: Removing NSFW Concepts from Vision-and-Language Models","date":"2023-11-27","arxiv_id":"2311.16254","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/removing-nsfw-concepts-from-vision-and#ran","syntology_url":"https://syntology.ai/paper/2311.16254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.16254"}},"official":{"repos":["aimagelab/safe-clip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/3d-mir-a-benchmark-and-empirical-study-on-3d","slug":"3d-mir-a-benchmark-and-empirical-study-on-3d","title":"3D-MIR: A Benchmark and Empirical Study on 3D Medical Image Retrieval in Radiology","date":"2023-11-23","arxiv_id":"2311.13752","repositories_listed":1,"syntology":null},{"url":"/paper/ai-generated-images-introduce-invisible","slug":"ai-generated-images-introduce-invisible","title":"Invisible Relevance Bias: Text-Image Retrieval Models Prefer AI-Generated Images","date":"2023-11-23","arxiv_id":"2311.14084","repositories_listed":1,"syntology":null},{"url":"/paper/some-like-it-small-czech-semantic-embedding","slug":"some-like-it-small-czech-semantic-embedding","title":"Some Like It Small: Czech Semantic Embedding Models for Industry Applications","date":"2023-11-23","arxiv_id":"2311.13921","repositories_listed":1,"syntology":null},{"url":"/paper/attribute-aware-deep-hashing-with-self","slug":"attribute-aware-deep-hashing-with-self","title":"Attribute-Aware Deep Hashing with Self-Consistency for Large-Scale Fine-Grained Image Retrieval","date":"2023-11-21","arxiv_id":"2311.12894","repositories_listed":1,"syntology":null},{"url":"/paper/pretrain-like-you-inference-masked-tuning","slug":"pretrain-like-you-inference-masked-tuning","title":"Pretrain like Your Inference: Masked Tuning Improves Zero-Shot Composed Image Retrieval","date":"2023-11-13","arxiv_id":"2311.07622","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pretrain-like-you-inference-masked-tuning#ran","syntology_url":"https://syntology.ai/paper/2311.07622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07622"}},"official":{"repos":["Chen-Junyang-cn/PLI"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/training-clip-models-on-data-from-scientific","slug":"training-clip-models-on-data-from-scientific","title":"Training CLIP models on Data from Scientific Papers","date":"2023-11-08","arxiv_id":"2311.04711","repositories_listed":1,"syntology":null},{"url":"/paper/deeppatent2-a-large-scale-benchmarking-corpus","slug":"deeppatent2-a-large-scale-benchmarking-corpus","title":"DeepPatent2: A Large-Scale Benchmarking Corpus for Technical Drawing Understanding","date":"2023-11-07","arxiv_id":"2311.04098","repositories_listed":1,"syntology":null},{"url":"/paper/local-global-self-supervised-visual","slug":"local-global-self-supervised-visual","title":"Patch-Wise Self-Supervised Visual Representation Learning: A Fine-Grained Approach","date":"2023-10-28","arxiv_id":"2310.18651","repositories_listed":1,"syntology":null},{"url":"/paper/lipsim-a-provably-robust-perceptual","slug":"lipsim-a-provably-robust-perceptual","title":"LipSim: A Provably Robust Perceptual Similarity Metric","date":"2023-10-27","arxiv_id":"2310.18274","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lipsim-a-provably-robust-perceptual#ran","syntology_url":"https://syntology.ai/paper/2310.18274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18274"}},"official":{"repos":["saraghazanfari/lipsim"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/semantic-aware-adversarial-training-for-1","slug":"semantic-aware-adversarial-training-for-1","title":"Semantic-Aware Adversarial Training for Reliable Deep Hashing Retrieval","date":"2023-10-23","arxiv_id":"2310.14637","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-and-multimodal","slug":"large-language-models-and-multimodal","title":"Large Language Models and Multimodal Retrieval for Visual Word Sense Disambiguation","date":"2023-10-21","arxiv_id":"2310.14025","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-and-multimodal#ran","syntology_url":"https://syntology.ai/paper/2310.14025","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14025"}},"official":{"repos":["anastasiakrith/multimodal-retrieval-for-vwsd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/representation-learning-via-consistent-2","slug":"representation-learning-via-consistent-2","title":"Representation Learning via Consistent Assignment of Views over Random Partitions","date":"2023-10-19","arxiv_id":"2310.12692","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/representation-learning-via-consistent-2#ran","syntology_url":"https://syntology.ai/paper/2310.12692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12692"}},"official":{"repos":["sthalles/carp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-the-fairness-of-discriminative","slug":"evaluating-the-fairness-of-discriminative","title":"Evaluating the Fairness of Discriminative Foundation Models in Computer Vision","date":"2023-10-18","arxiv_id":"2310.11867","repositories_listed":1,"syntology":null},{"url":"/paper/vision-by-language-for-training-free","slug":"vision-by-language-for-training-free","title":"Vision-by-Language for Training-Free Compositional Image Retrieval","date":"2023-10-13","arxiv_id":"2310.09291","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vision-by-language-for-training-free#ran","syntology_url":"https://syntology.ai/paper/2310.09291","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09291"}},"official":{"repos":["explainableml/vision_by_language"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/sentence-level-prompts-benefit-composed-image","slug":"sentence-level-prompts-benefit-composed-image","title":"Sentence-level Prompts Benefit Composed Image Retrieval","date":"2023-10-09","arxiv_id":"2310.05473","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sentence-level-prompts-benefit-composed-image#ran","syntology_url":"https://syntology.ai/paper/2310.05473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05473"}},"official":{"repos":["chunmeifeng/sprc"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sub-token-vit-embedding-via-stochastic","slug":"sub-token-vit-embedding-via-stochastic","title":"Sub-token ViT Embedding via Stochastic Resonance Transformers","date":"2023-10-06","arxiv_id":"2310.03967","repositories_listed":1,"syntology":null},{"url":"/paper/context-i2w-mapping-images-to-context","slug":"context-i2w-mapping-images-to-context","title":"Context-I2W: Mapping Images to Context-dependent Words for Accurate Zero-Shot Composed Image Retrieval","date":"2023-09-28","arxiv_id":"2309.16137","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/context-i2w-mapping-images-to-context#ran","syntology_url":"https://syntology.ai/paper/2309.16137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16137"}},"official":{"repos":["pter61/context-i2w"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dark-side-augmentation-generating-diverse-1","slug":"dark-side-augmentation-generating-diverse-1","title":"Dark Side Augmentation: Generating Diverse Night Examples for Metric Learning","date":"2023-09-28","arxiv_id":"2309.16351","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dark-side-augmentation-generating-diverse-1#ran","syntology_url":"https://syntology.ai/paper/2309.16351","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16351"}},"official":{"repos":["mohwald/gandtr"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/forb-a-flat-object-retrieval-benchmark-for-1","slug":"forb-a-flat-object-retrieval-benchmark-for-1","title":"FORB: A Flat Object Retrieval Benchmark for Universal Image Embedding","date":"2023-09-28","arxiv_id":"2309.16249","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/forb-a-flat-object-retrieval-benchmark-for-1#ran","syntology_url":"https://syntology.ai/paper/2309.16249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16249"}},"official":{"repos":["pxiangwu/forb"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-novel-geo-localization-method-for-uav-and","slug":"a-novel-geo-localization-method-for-uav-and","title":"A Novel Geo-Localization Method for UAV and Satellite Images Using Cross-View Consistent Attention","date":"2023-09-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/resolving-references-in-visually-grounded","slug":"resolving-references-in-visually-grounded","title":"Resolving References in Visually-Grounded Dialogue via Text Generation","date":"2023-09-23","arxiv_id":"2309.13430","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-differentiable-outlier-detection","slug":"implicit-differentiable-outlier-detection","title":"Implicit Differentiable Outlier Detection Enable Robust Deep Multimodal Analysis","date":"2023-09-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/on-the-fly-sfm-what-you-capture-is-what-you","slug":"on-the-fly-sfm-what-you-capture-is-what-you","title":"On-the-Fly SfM: What you capture is What you get","date":"2023-09-21","arxiv_id":"2309.11883","repositories_listed":1,"syntology":null},{"url":"/paper/keep-it-simpool-who-said-supervised","slug":"keep-it-simpool-who-said-supervised","title":"Keep It SimPool: Who Said Supervised Transformers Suffer from Attention Deficit?","date":"2023-09-13","arxiv_id":"2309.06891","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/keep-it-simpool-who-said-supervised#ran","syntology_url":"https://syntology.ai/paper/2309.06891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06891"}},"official":{"repos":["billpsomas/simpool"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-content-based-pixel-retrieval-in","slug":"towards-content-based-pixel-retrieval-in","title":"Towards Content-based Pixel Retrieval in Revisited Oxford and Paris","date":"2023-09-11","arxiv_id":"2309.05438","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/towards-content-based-pixel-retrieval-in#ran","syntology_url":"https://syntology.ai/paper/2309.05438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05438"}},"official":{"repos":["anguoyuan/pixel_retrieval-segmented_instance_retrieval"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/collecting-visually-grounded-dialogue-with-a-1","slug":"collecting-visually-grounded-dialogue-with-a-1","title":"Collecting Visually-Grounded Dialogue with A Game Of Sorts","date":"2023-09-10","arxiv_id":"2309.05162","repositories_listed":1,"syntology":null},{"url":"/paper/patent-image-retrieval-using-transformer","slug":"patent-image-retrieval-using-transformer","title":"Patent image retrieval using transformer-based deep metric learning","date":"2023-09-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/covr-learning-composed-video-retrieval-from","slug":"covr-learning-composed-video-retrieval-from","title":"CoVR-2: Automatic Data Construction for Composed Video Retrieval","date":"2023-08-28","arxiv_id":"2308.14746","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/covr-learning-composed-video-retrieval-from#ran","syntology_url":"https://syntology.ai/paper/2308.14746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14746"}},"official":{"repos":["lucas-ventura/CoVR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-food-image-retrieval-via","slug":"towards-food-image-retrieval-via","title":"Towards Food Image Retrieval via Generalization-oriented Sampling and Loss Function Design","date":"2023-08-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fashionlogo-prompting-multimodal-large","slug":"fashionlogo-prompting-multimodal-large","title":"FashionLOGO: Prompting Multimodal Large Language Models for Fashion Logo Embeddings","date":"2023-08-17","arxiv_id":"2308.09012","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-visual-and-semantic-similarity","slug":"integrating-visual-and-semantic-similarity","title":"Integrating Visual and Semantic Similarity Using Hierarchies for Image Retrieval","date":"2023-08-16","arxiv_id":"2308.08431","repositories_listed":1,"syntology":null},{"url":"/paper/mixbct-towards-self-adapting-backward","slug":"mixbct-towards-self-adapting-backward","title":"MixBCT: Towards Self-Adapting Backward-Compatible Training","date":"2023-08-14","arxiv_id":"2308.06948","repositories_listed":1,"syntology":null},{"url":"/paper/aspectmmkg-a-multi-modal-knowledge-graph-with","slug":"aspectmmkg-a-multi-modal-knowledge-graph-with","title":"AspectMMKG: A Multi-modal Knowledge Graph with Aspect-aware Entities","date":"2023-08-09","arxiv_id":"2308.04992","repositories_listed":1,"syntology":null},{"url":"/paper/coarse-to-fine-learning-compact","slug":"coarse-to-fine-learning-compact","title":"Coarse-to-Fine: Learning Compact Discriminative Representation for Single-Stage Image Retrieval","date":"2023-08-08","arxiv_id":"2308.04008","repositories_listed":1,"syntology":{"n":21,"n_ran":18,"n_constructed":0,"n_ran_checked":17,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":11,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/coarse-to-fine-learning-compact#ran","syntology_url":"https://syntology.ai/paper/2308.04008","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04008"}},"official":{"repos":["bassyess/cfcd"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/unifying-two-stream-encoders-with","slug":"unifying-two-stream-encoders-with","title":"Unifying Two-Stream Encoders with Transformers for Cross-Modal Retrieval","date":"2023-08-08","arxiv_id":"2308.04343","repositories_listed":1,"syntology":null},{"url":"/paper/anyloc-towards-universal-visual-place","slug":"anyloc-towards-universal-visual-place","title":"AnyLoc: Towards Universal Visual Place Recognition","date":"2023-08-01","arxiv_id":"2308.00688","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/anyloc-towards-universal-visual-place#ran","syntology_url":"https://syntology.ai/paper/2308.00688","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.00688"}},"official":{"repos":["AnyLoc/AnyLoc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-the-gap-multi-level-cross-modality","slug":"bridging-the-gap-multi-level-cross-modality","title":"Bridging the Gap: Multi-Level Cross-Modality Joint Alignment for Visible-Infrared Person Re-Identification","date":"2023-07-17","arxiv_id":"2307.08316","repositories_listed":1,"syntology":null},{"url":"/paper/divide-classify-fine-grained-classification","slug":"divide-classify-fine-grained-classification","title":"Divide&Classify: Fine-Grained Classification for City-Wide Visual Place Recognition","date":"2023-07-17","arxiv_id":"2307.08417","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-3-dof-ground-to-satellite-camera","slug":"boosting-3-dof-ground-to-satellite-camera","title":"Boosting 3-DoF Ground-to-Satellite Camera Localization Accuracy via Geometry-Guided Cross-View Transformer","date":"2023-07-16","arxiv_id":"2307.08015","repositories_listed":1,"syntology":null},{"url":"/paper/risk-controlled-image-retrieval","slug":"risk-controlled-image-retrieval","title":"Risk Controlled Image Retrieval","date":"2023-07-14","arxiv_id":"2307.07336","repositories_listed":1,"syntology":null},{"url":"/paper/clipmasterprints-fooling-contrastive-language","slug":"clipmasterprints-fooling-contrastive-language","title":"Fooling Contrastive Language-Image Pre-trained Models with CLIPMasterPrints","date":"2023-07-07","arxiv_id":"2307.03798","repositories_listed":1,"syntology":null},{"url":"/paper/what-makes-imagenet-look-unlike-laion","slug":"what-makes-imagenet-look-unlike-laion","title":"What Makes ImageNet Look Unlike LAION","date":"2023-06-27","arxiv_id":"2306.15769","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-matching-and-reasoning-for-multi","slug":"hierarchical-matching-and-reasoning-for-multi","title":"Hierarchical Matching and Reasoning for Multi-Query Image Retrieval","date":"2023-06-26","arxiv_id":"2306.14460","repositories_listed":1,"syntology":null},{"url":"/paper/rs5m-a-large-scale-vision-language-dataset","slug":"rs5m-a-large-scale-vision-language-dataset","title":"RS5M and GeoRSCLIP: A Large Scale Vision-Language Dataset and A Large Vision-Language Model for Remote Sensing","date":"2023-06-20","arxiv_id":"2306.11300","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rs5m-a-large-scale-vision-language-dataset#ran","syntology_url":"https://syntology.ai/paper/2306.11300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11300"}},"official":{"repos":["om-ai-lab/rs5m"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-attribute-insertions-for","slug":"cross-modal-attribute-insertions-for","title":"Cross-Modal Attribute Insertions for Assessing the Robustness of Vision-and-Language Learning","date":"2023-06-19","arxiv_id":"2306.11065","repositories_listed":1,"syntology":null},{"url":"/paper/graph-convolution-based-efficient-re-ranking","slug":"graph-convolution-based-efficient-re-ranking","title":"Graph Convolution Based Efficient Re-Ranking for Visual Retrieval","date":"2023-06-15","arxiv_id":"2306.08792","repositories_listed":1,"syntology":null},{"url":"/paper/yes-we-cann-constrained-approximate-nearest","slug":"yes-we-cann-constrained-approximate-nearest","title":"Yes, we CANN: Constrained Approximate Nearest Neighbors for local feature-based visual localization","date":"2023-06-15","arxiv_id":"2306.09012","repositories_listed":1,"syntology":null},{"url":"/paper/mofi-learning-image-representations-from","slug":"mofi-learning-image-representations-from","title":"MOFI: Learning Image Representations from Noisy Entity Annotated Images","date":"2023-06-13","arxiv_id":"2306.07952","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mofi-learning-image-representations-from#ran","syntology_url":"https://syntology.ai/paper/2306.07952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07952"}},"official":{"repos":["apple/ml-mofi"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-composed-text-image-retrieval","slug":"zero-shot-composed-text-image-retrieval","title":"Zero-shot Composed Text-Image Retrieval","date":"2023-06-12","arxiv_id":"2306.07272","repositories_listed":1,"syntology":null},{"url":"/paper/self-enhancement-improves-text-image","slug":"self-enhancement-improves-text-image","title":"Self-Enhancement Improves Text-Image Retrieval in Foundation Visual-Language Models","date":"2023-06-11","arxiv_id":"2306.06691","repositories_listed":1,"syntology":null},{"url":"/paper/chatting-makes-perfect-chat-based-image-1","slug":"chatting-makes-perfect-chat-based-image-1","title":"Chatting Makes Perfect: Chat-based Image Retrieval","date":"2023-05-31","arxiv_id":"2305.20062","repositories_listed":1,"syntology":null},{"url":"/paper/crossget-cross-guided-ensemble-of-tokens-for","slug":"crossget-cross-guided-ensemble-of-tokens-for","title":"CrossGET: Cross-Guided Ensemble of Tokens for Accelerating Vision-Language Transformers","date":"2023-05-27","arxiv_id":"2305.17455","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crossget-cross-guided-ensemble-of-tokens-for#ran","syntology_url":"https://syntology.ai/paper/2305.17455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17455"}},"official":{"repos":["sdc17/crossget"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/factual-a-benchmark-for-faithful-and","slug":"factual-a-benchmark-for-faithful-and","title":"FACTUAL: A Benchmark for Faithful and Consistent Textual Scene Graph Parsing","date":"2023-05-27","arxiv_id":"2305.17497","repositories_listed":1,"syntology":null},{"url":"/paper/generating-images-with-multimodal-language","slug":"generating-images-with-multimodal-language","title":"Generating Images with Multimodal Language Models","date":"2023-05-26","arxiv_id":"2305.17216","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/generating-images-with-multimodal-language#ran","syntology_url":"https://syntology.ai/paper/2305.17216","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17216"}},"official":{"repos":["kohjingyu/gill"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/pace-unified-multi-modal-dialogue-pre","slug":"pace-unified-multi-modal-dialogue-pre","title":"PaCE: Unified Multi-modal Dialogue Pre-training with Progressive and Compositional Experts","date":"2023-05-24","arxiv_id":"2305.14839","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/pace-unified-multi-modal-dialogue-pre#ran","syntology_url":"https://syntology.ai/paper/2305.14839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14839"}},"official":{"repos":["AlibabaResearch/DAMO-ConvAI"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/edis-entity-driven-image-search-over","slug":"edis-entity-driven-image-search-over","title":"EDIS: Entity-Driven Image Search over Multimodal Web Content","date":"2023-05-23","arxiv_id":"2305.13631","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/edis-entity-driven-image-search-over#ran","syntology_url":"https://syntology.ai/paper/2305.13631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13631"}},"official":{"repos":["emerisly/edis"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-test-time-bias-for-fair-image-1","slug":"mitigating-test-time-bias-for-fair-image-1","title":"Mitigating Test-Time Bias for Fair Image Retrieval","date":"2023-05-23","arxiv_id":"2305.19329","repositories_listed":1,"syntology":null},{"url":"/paper/imaginator-pre-trained-image-text-joint","slug":"imaginator-pre-trained-image-text-joint","title":"IMAGINATOR: Pre-Trained Image+Text Joint Embeddings using Word-Level Grounding of Images","date":"2023-05-12","arxiv_id":"2305.10438","repositories_listed":1,"syntology":null},{"url":"/paper/searching-mobile-app-screens-via-text-doodle","slug":"searching-mobile-app-screens-via-text-doodle","title":"Searching Mobile App Screens via Text + Doodle","date":"2023-05-08","arxiv_id":"2305.06165","repositories_listed":1,"syntology":null},{"url":"/paper/fairness-in-image-search-a-study-of","slug":"fairness-in-image-search-a-study-of","title":"Fairness in Image Search: A Study of Occupational Stereotyping in Image Retrieval and its Debiasing","date":"2023-05-06","arxiv_id":"2305.03881","repositories_listed":1,"syntology":null},{"url":"/paper/cola-a-benchmark-for-compositional-text-to-1","slug":"cola-a-benchmark-for-compositional-text-to-1","title":"COLA: A Benchmark for Compositional Text-to-image Retrieval","date":"2023-05-05","arxiv_id":"2305.03689","repositories_listed":1,"syntology":null},{"url":"/paper/boundary-aware-backward-compatible","slug":"boundary-aware-backward-compatible","title":"Boundary-aware Backward-Compatible Representation via Adversarial Learning in Image Retrieval","date":"2023-05-04","arxiv_id":"2305.02610","repositories_listed":1,"syntology":null},{"url":"/paper/a-neural-divide-and-conquer-reasoning","slug":"a-neural-divide-and-conquer-reasoning","title":"A Neural Divide-and-Conquer Reasoning Framework for Image Retrieval from Linguistically Complex Text","date":"2023-05-03","arxiv_id":"2305.02265","repositories_listed":1,"syntology":null},{"url":"/paper/zerosearch-local-image-search-from-text-with","slug":"zerosearch-local-image-search-from-text-with","title":"ZeroSearch: Local Image Search from Text with Zero Shot Learning","date":"2023-05-01","arxiv_id":"2305.00715","repositories_listed":1,"syntology":null},{"url":"/paper/stir-siamese-transformer-for-image-retrieval","slug":"stir-siamese-transformer-for-image-retrieval","title":"STIR: Siamese Transformer for Image Retrieval Postprocessing","date":"2023-04-26","arxiv_id":"2304.13393","repositories_listed":1,"syntology":null},{"url":"/paper/rank-flow-embedding-for-unsupervised-and-semi-1","slug":"rank-flow-embedding-for-unsupervised-and-semi-1","title":"Rank Flow Embedding for Unsupervised and Semi-Supervised Manifold Learning","date":"2023-04-24","arxiv_id":"2304.12448","repositories_listed":1,"syntology":null},{"url":"/paper/toward-real-time-image-annotation-using","slug":"toward-real-time-image-annotation-using","title":"Toward Real-Time Image Annotation Using Marginalized Coupled Dictionary Learning","date":"2023-04-14","arxiv_id":"2304.06907","repositories_listed":1,"syntology":null},{"url":"/paper/open-transmind-a-new-baseline-and-benchmark","slug":"open-transmind-a-new-baseline-and-benchmark","title":"Open-TransMind: A New Baseline and Benchmark for 1st Foundation Model Challenge of Intelligent Transportation","date":"2023-04-12","arxiv_id":"2304.06051","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-ocr-for-building-a-diverse-digital","slug":"efficient-ocr-for-building-a-diverse-digital","title":"Efficient OCR for Building a Diverse Digital History","date":"2023-04-05","arxiv_id":"2304.02737","repositories_listed":1,"syntology":null},{"url":"/paper/logonet-a-fine-grained-network-for-instance","slug":"logonet-a-fine-grained-network-for-instance","title":"LogoNet: a fine-grained network for instance-level logo sketch retrieval","date":"2023-04-05","arxiv_id":"2304.02214","repositories_listed":1,"syntology":null},{"url":"/paper/learning-similarity-between-scene-graphs-and","slug":"learning-similarity-between-scene-graphs-and","title":"SPAN: Learning Similarity between Scene Graphs and Images with Transformers","date":"2023-04-02","arxiv_id":"2304.00590","repositories_listed":1,"syntology":null},{"url":"/paper/bi-directional-training-for-composed-image","slug":"bi-directional-training-for-composed-image","title":"Bi-directional Training for Composed Image Retrieval via Text Prompt Learning","date":"2023-03-29","arxiv_id":"2303.16604","repositories_listed":1,"syntology":null},{"url":"/paper/mammut-a-simple-architecture-for-joint","slug":"mammut-a-simple-architecture-for-joint","title":"MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks","date":"2023-03-29","arxiv_id":"2303.16839","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mammut-a-simple-architecture-for-joint#ran","syntology_url":"https://syntology.ai/paper/2303.16839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16839"}},"official":null}}],"record_sha256":"b3d6d7f484dac0191fde2bbdb84f175915d24cc7c27921a799c3306f8eddf09a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}