{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/image-retrieval/papers/9","list_of":"/task/image-retrieval","task":"Image Retrieval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":23,"rows_per_page":100,"rows":[801,900],"of":2239,"counts":{"archive_papers_tagged":2239,"with_a_code_link":835,"where_syntology_ran_a_sample":218,"not_listed_spam_title":0,"listed":2239,"listed_where_code_ran":218,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":185,"every_run_a_failure_of_syntologys_instrument":33,"listed_with_a_run_with_no_instrument_failure":185,"listed_every_run_a_failure_of_syntologys_instrument":33,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/image-retrieval","prev":"/task/image-retrieval/papers/8","next":"/task/image-retrieval/papers/10","papers":[{"url":"/paper/how-a-general-purpose-commonsense-ontology","slug":"how-a-general-purpose-commonsense-ontology","title":"How a General-Purpose Commonsense Ontology can Improve Performance of Learning-Based Image Retrieval","date":"2017-05-24","arxiv_id":"1705.08844","repositories_listed":1,"syntology":null},{"url":"/paper/hashing-as-tie-aware-learning-to-rank","slug":"hashing-as-tie-aware-learning-to-rank","title":"Hashing as Tie-Aware Learning to Rank","date":"2017-05-23","arxiv_id":"1705.08562","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-part-based-weighting-aggregation","slug":"unsupervised-part-based-weighting-aggregation","title":"Unsupervised Part-based Weighting Aggregation of Deep Convolutional Features for Image Retrieval","date":"2017-05-03","arxiv_id":"1705.01247","repositories_listed":1,"syntology":null},{"url":"/paper/convex-formulation-of-multiple-instance","slug":"convex-formulation-of-multiple-instance","title":"Convex Formulation of Multiple Instance Learning from Positive and Unlabeled Bags","date":"2017-04-22","arxiv_id":"1704.06767","repositories_listed":1,"syntology":null},{"url":"/paper/mihash-online-hashing-with-mutual-information","slug":"mihash-online-hashing-with-mutual-information","title":"MIHash: Online Hashing with Mutual Information","date":"2017-03-27","arxiv_id":"1703.08919","repositories_listed":1,"syntology":null},{"url":"/paper/medical-image-retrieval-using-deep","slug":"medical-image-retrieval-using-deep","title":"Medical Image Retrieval using Deep Convolutional Neural Network","date":"2017-03-24","arxiv_id":"1703.08472","repositories_listed":1,"syntology":null},{"url":"/paper/deep-sketch-hashing-fast-free-hand-sketch","slug":"deep-sketch-hashing-fast-free-hand-sketch","title":"Deep Sketch Hashing: Fast Free-hand Sketch-Based Image Retrieval","date":"2017-03-16","arxiv_id":"1703.05605","repositories_listed":1,"syntology":null},{"url":"/paper/context-aware-query-image-representation-for","slug":"context-aware-query-image-representation-for","title":"Context Aware Query Image Representation for Particular Object Retrieval","date":"2017-03-03","arxiv_id":"1703.01226","repositories_listed":1,"syntology":null},{"url":"/paper/deep-supervised-hashing-with-triplet-labels","slug":"deep-supervised-hashing-with-triplet-labels","title":"Deep Supervised Hashing with Triplet Labels","date":"2016-12-12","arxiv_id":"1612.03900","repositories_listed":1,"syntology":null},{"url":"/paper/fast-supervised-discrete-hashing-and-its","slug":"fast-supervised-discrete-hashing-and-its","title":"Fast Supervised Discrete Hashing and its Analysis","date":"2016-11-30","arxiv_id":"1611.10017","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-play-guess-who-and-inventing-a","slug":"learning-to-play-guess-who-and-inventing-a","title":"Learning to Play Guess Who? and Inventing a Grounded Language as a Consequence","date":"2016-11-10","arxiv_id":"1611.03218","repositories_listed":1,"syntology":null},{"url":"/paper/what-is-the-best-practice-for-cnns-applied-to","slug":"what-is-the-best-practice-for-cnns-applied-to","title":"What Is the Best Practice for CNNs Applied to Visual Instance Retrieval?","date":"2016-11-05","arxiv_id":"1611.01640","repositories_listed":1,"syntology":null},{"url":"/paper/deepdiary-automatic-caption-generation-for","slug":"deepdiary-automatic-caption-generation-for","title":"DeepDiary: Automatic Caption Generation for Lifelogging Image Streams","date":"2016-08-12","arxiv_id":"1608.03819","repositories_listed":1,"syntology":null},{"url":"/paper/sift-meets-cnn-a-decade-survey-of-instance","slug":"sift-meets-cnn-a-decade-survey-of-instance","title":"SIFT Meets CNN: A Decade Survey of Instance Retrieval","date":"2016-08-05","arxiv_id":"1608.01807","repositories_listed":1,"syntology":null},{"url":"/paper/deep-supervised-hashing-for-fast-image","slug":"deep-supervised-hashing-for-fast-image","title":"Deep Supervised Hashing for Fast Image Retrieval","date":"2016-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-location-recognition-and-the","slug":"large-scale-location-recognition-and-the","title":"Large-Scale Location Recognition and the Geometric Burstiness Problem","date":"2016-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-compact-binary-descriptors-with-1","slug":"learning-compact-binary-descriptors-with-1","title":"Learning compact binary descriptors with unsupervised deep neural networks","date":"2016-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/single-image-3d-interpreter-network","slug":"single-image-3d-interpreter-network","title":"Single Image 3D Interpreter Network","date":"2016-04-29","arxiv_id":"1604.08685","repositories_listed":1,"syntology":null},{"url":"/paper/distinctive-interest-point-selection-for","slug":"distinctive-interest-point-selection-for","title":"Distinctive Interest Point Selection for Efficient Near-duplicate Image Retrieval","date":"2016-04-28","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/selective-convolutional-descriptor","slug":"selective-convolutional-descriptor","title":"Selective Convolutional Descriptor Aggregation for Fine-Grained Image Retrieval","date":"2016-04-18","arxiv_id":"1604.04994","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-visual-sense-disambiguation-for","slug":"unsupervised-visual-sense-disambiguation-for","title":"Unsupervised Visual Sense Disambiguation for Verbs using Multimodal Embeddings","date":"2016-03-30","arxiv_id":"1603.09188","repositories_listed":1,"syntology":null},{"url":"/paper/planet-photo-geolocation-with-convolutional","slug":"planet-photo-geolocation-with-convolutional","title":"PlaNet - Photo Geolocation with Convolutional Neural Networks","date":"2016-02-17","arxiv_id":"1602.05314","repositories_listed":1,"syntology":null},{"url":"/paper/cross-dimensional-weighting-for-aggregated","slug":"cross-dimensional-weighting-for-aggregated","title":"Cross-dimensional Weighting for Aggregated Deep Convolutional Features","date":"2015-12-13","arxiv_id":"1512.04065","repositories_listed":1,"syntology":null},{"url":"/paper/visual-word2vec-vis-w2v-learning-visually","slug":"visual-word2vec-vis-w2v-learning-visually","title":"Visual Word2Vec (vis-w2v): Learning Visually Grounded Word Embeddings Using Abstract Scenes","date":"2015-11-22","arxiv_id":"1511.07067","repositories_listed":1,"syntology":null},{"url":"/paper/natural-language-object-retrieval","slug":"natural-language-object-retrieval","title":"Natural Language Object Retrieval","date":"2015-11-13","arxiv_id":"1511.04164","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/natural-language-object-retrieval#ran","syntology_url":"https://syntology.ai/paper/1511.04164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1511.04164"}},"official":null}},{"url":"/paper/feature-learning-based-deep-supervised","slug":"feature-learning-based-deep-supervised","title":"Feature Learning based Deep Supervised Hashing with Pairwise Labels","date":"2015-11-12","arxiv_id":"1511.03855","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-summarization-of-egocentric-photo","slug":"semantic-summarization-of-egocentric-photo","title":"Semantic Summarization of Egocentric Photo Stream Events","date":"2015-11-02","arxiv_id":"1511.00438","repositories_listed":1,"syntology":null},{"url":"/paper/hash-function-learning-via-codewords","slug":"hash-function-learning-via-codewords","title":"Hash Function Learning via Codewords","date":"2015-08-13","arxiv_id":"1508.03285","repositories_listed":1,"syntology":null},{"url":"/paper/supervised-learning-of-semantics-preserving","slug":"supervised-learning-of-semantics-preserving","title":"Supervised Learning of Semantics-Preserving Hash via Deep Convolutional Neural Networks","date":"2015-07-01","arxiv_id":"1507.00101","repositories_listed":1,"syntology":null},{"url":"/paper/barcode-annotations-for-medical-image","slug":"barcode-annotations-for-medical-image","title":"Barcode Annotations for Medical Image Retrieval: A Preliminary Investigation","date":"2015-05-19","arxiv_id":"1505.05212","repositories_listed":1,"syntology":null},{"url":"/paper/socializing-the-semantic-gap-a-comparative","slug":"socializing-the-semantic-gap-a-comparative","title":"Socializing the Semantic Gap: A Comparative Survey on Image Tag Assignment, Refinement and Retrieval","date":"2015-03-28","arxiv_id":"1503.08248","repositories_listed":1,"syntology":null},{"url":"/paper/supervised-hashing-using-graph-cuts-and","slug":"supervised-hashing-using-graph-cuts-and","title":"Supervised Hashing Using Graph Cuts and Boosted Decision Trees","date":"2014-08-24","arxiv_id":"1408.5574","repositories_listed":1,"syntology":null},{"url":"/paper/neural-codes-for-image-retrieval","slug":"neural-codes-for-image-retrieval","title":"Neural Codes for Image Retrieval","date":"2014-04-07","arxiv_id":"1404.1777","repositories_listed":1,"syntology":null},{"url":"/paper/recognizing-image-style","slug":"recognizing-image-style","title":"Recognizing Image Style","date":"2013-11-15","arxiv_id":"1311.3715","repositories_listed":1,"syntology":null},{"url":"/paper/gray-level-co-occurrence-matrices","slug":"gray-level-co-occurrence-matrices","title":"Gray Level Co-Occurrence Matrices: Generalisation and Some New Features","date":"2012-05-22","arxiv_id":"1205.4831","repositories_listed":1,"syntology":null},{"url":null,"slug":"far-net-multi-stage-fusion-network-with","title":"FAR-Net: Multi-Stage Fusion Network with Enhanced Semantic Alignment and Adaptive Reconciliation for Composed Image Retrieval","date":"2025-07-17","arxiv_id":"2507.12823","repositories_listed":0,"syntology":null},{"url":null,"slug":"mcot-re-multi-faceted-chain-of-thought-and-re","title":"MCoT-RE: Multi-Faceted Chain-of-Thought and Re-Ranking for Training-Free Zero-Shot Composed Image Retrieval","date":"2025-07-17","arxiv_id":"2507.12819","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-synthesis-of-high-quality-triplet","title":"Automatic Synthesis of High-Quality Triplet Data for Composed Image Retrieval","date":"2025-07-08","arxiv_id":"2507.05970","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-of-vision-language-models-for","title":"An analysis of vision-language models for fabric retrieval","date":"2025-07-07","arxiv_id":"2507.04735","repositories_listed":0,"syntology":null},{"url":null,"slug":"llama-nemoretriever-colembed-top-performing","title":"Llama Nemoretriever Colembed: Top-Performing Text-Image Retrieval Model","date":"2025-07-07","arxiv_id":"2507.05513","repositories_listed":0,"syntology":null},{"url":null,"slug":"mask-aware-text-to-image-retrieval-referring","title":"Mask-aware Text-to-Image Retrieval: Referring Expression Segmentation Meets Cross-modal Retrieval","date":"2025-06-28","arxiv_id":"2506.22864","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-burstiness-of-faces-in-set","title":"On the Burstiness of Faces in Set","date":"2025-06-25","arxiv_id":"2506.20312","repositories_listed":0,"syntology":null},{"url":null,"slug":"referring-expression-instance-retrieval-and-a","title":"Referring Expression Instance Retrieval and A Strong End-to-End Baseline","date":"2025-06-23","arxiv_id":"2506.18246","repositories_listed":0,"syntology":null},{"url":null,"slug":"class-agnostic-instance-level-descriptor-for","title":"Class Agnostic Instance-level Descriptor for Visual Instance Search","date":"2025-06-20","arxiv_id":"2506.16745","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-image-retrieval-via-dual-vision","title":"Fine-grained Image Retrieval via Dual-Vision Adaptation","date":"2025-06-19","arxiv_id":"2506.16273","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-semantically-aware-relevance-measure-for","title":"A Semantically-Aware Relevance Measure for Content-Based Medical Image Retrieval Evaluation","date":"2025-06-16","arxiv_id":"2506.13509","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-multi-positive-contrastive","title":"Hierarchical Multi-Positive Contrastive Learning for Patent Image Retrieval","date":"2025-06-16","arxiv_id":"2506.13496","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-image-matching-for-uav-absolute","title":"Hierarchical Image Matching for UAV Absolute Visual Localization via Semantic and Structural Constraints","date":"2025-06-11","arxiv_id":"2506.09748","repositories_listed":0,"syntology":null},{"url":null,"slug":"hidden-bias-in-the-machine-stereotypes-in","title":"Hidden Bias in the Machine: Stereotypes in Text-to-Image Models","date":"2025-06-09","arxiv_id":"2506.13780","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantization-based-bounds-on-the-wasserstein","title":"Quantization-based Bounds on the Wasserstein Metric","date":"2025-06-01","arxiv_id":"2506.00976","repositories_listed":0,"syntology":null},{"url":null,"slug":"sketch-down-the-flops-towards-efficient","title":"Sketch Down the FLOPs: Towards Efficient Networks for Human Sketch","date":"2025-05-29","arxiv_id":"2505.23763","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-feature-matching-of-uav-images-via","title":"Fast Feature Matching of UAV Images via Matrix Band Reduction-based GPU Data Schedule","date":"2025-05-28","arxiv_id":"2505.22089","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-visual-encoder-learn-to-see-arrows","title":"Can Visual Encoder Learn to See Arrows?","date":"2025-05-26","arxiv_id":"2505.19944","repositories_listed":0,"syntology":null},{"url":null,"slug":"mllm-guided-vlm-fine-tuning-with-joint","title":"MLLM-Guided VLM Fine-Tuning with Joint Inference for Zero-Shot Composed Image Retrieval","date":"2025-05-26","arxiv_id":"2505.19707","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-reasoning-agent-for-zero-shot","title":"Multimodal Reasoning Agent for Zero-Shot Composed Image Retrieval","date":"2025-05-26","arxiv_id":"2505.19952","repositories_listed":0,"syntology":null},{"url":null,"slug":"tng-clip-training-time-negation-data","title":"TNG-CLIP:Training-Time Negation Data Generation for Negation Awareness of CLIP","date":"2025-05-24","arxiv_id":"2505.18434","repositories_listed":0,"syntology":null},{"url":null,"slug":"dart-3-leveraging-distance-for-test-time","title":"DART$^3$: Leveraging Distance for Test Time Adaptation in Person Re-Identification","date":"2025-05-23","arxiv_id":"2505.18337","repositories_listed":0,"syntology":null},{"url":null,"slug":"detailfusion-a-dual-branch-framework-with","title":"DetailFusion: A Dual-branch Framework with Detail Enhancement for Composed Image Retrieval","date":"2025-05-23","arxiv_id":"2505.17796","repositories_listed":0,"syntology":null},{"url":"/paper/highlighting-what-matters-promptable","slug":"highlighting-what-matters-promptable","title":"Highlighting What Matters: Promptable Embeddings for Attribute-Focused Image Retrieval","date":"2025-05-21","arxiv_id":"2505.15877","repositories_listed":0,"syntology":null},{"url":null,"slug":"ia-t2i-internet-augmented-text-to-image","title":"IA-T2I: Internet-Augmented Text-to-Image Generation","date":"2025-05-21","arxiv_id":"2505.15779","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-rag-driven-anomaly-detection-and","title":"Multimodal RAG-driven Anomaly Detection and Classification in Laser Powder Bed Fusion using Large Language Models","date":"2025-05-20","arxiv_id":"2505.13828","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-planar-object-detection-and","title":"Non-planar Object Detection and Identification by Features Matching and Triangulation Growth","date":"2025-05-19","arxiv_id":"2506.13769","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11121","title":"Redundancy-Aware Pretraining of Vision-Language Foundation Models in Remote Sensing","date":"2025-05-16","arxiv_id":"2505.11121","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-bag-of-words-image-retrieval-with","title":"Improved Bag-of-Words Image Retrieval with Geometric Constraints for Ground Texture Localization","date":"2025-05-16","arxiv_id":"2505.11620","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-mapping-to-composing-a-two-stage","title":"From Mapping to Composing: A Two-Stage Framework for Zero-shot Composed Image Retrieval","date":"2025-04-25","arxiv_id":"2504.17990","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multimodal-recaptioning-framework-to","title":"A Multimodal Recaptioning Framework to Account for Perceptual Diversity in Multilingual Vision-Language Modeling","date":"2025-04-19","arxiv_id":"2504.14359","repositories_listed":0,"syntology":null},{"url":null,"slug":"semcore-a-semantic-enhanced-generative-cross","title":"SemCORE: A Semantic-Enhanced Generative Cross-Modal Retrieval Framework with MLLMs","date":"2025-04-17","arxiv_id":"2504.13172","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-visual-relation-detection-with","title":"Generalized Visual Relation Detection with Diffusion Models","date":"2025-04-16","arxiv_id":"2504.12100","repositories_listed":0,"syntology":null},{"url":"/paper/tmcir-token-merge-benefits-composed-image","slug":"tmcir-token-merge-benefits-composed-image","title":"TMCIR: Token Merge Benefits Composed Image Retrieval","date":"2025-04-15","arxiv_id":"2504.10995","repositories_listed":0,"syntology":null},{"url":null,"slug":"focallens-instruction-tuning-enables-zero","title":"FocalLens: Instruction Tuning Enables Zero-Shot Conditional Image Representations","date":"2025-04-11","arxiv_id":"2504.08368","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypergraph-vision-transformers-images-are","title":"Hypergraph Vision Transformers: Images are More than Nodes, More than Edges","date":"2025-04-11","arxiv_id":"2504.08710","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-reference-learning-for-fine","title":"Multi-modal Reference Learning for Fine-grained Text-to-Image Retrieval","date":"2025-04-10","arxiv_id":"2504.07718","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-oriented-image-retrieval-system-horse-a","title":"Human-Oriented Image Retrieval System (HORSE): A Neuro-Symbolic Approach to Optimizing Retrieval of Previewed Images","date":"2025-04-09","arxiv_id":"2504.10502","repositories_listed":0,"syntology":null},{"url":null,"slug":"rejepa-a-novel-joint-embedding-predictive","title":"REJEPA: A Novel Joint-Embedding Predictive Architecture for Efficient Remote Sensing Image Retrieval","date":"2025-04-04","arxiv_id":"2504.03169","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-guided-attention-head-selection-for","title":"Prompt-Guided Attention Head Selection for Focus-Oriented Image Retrieval","date":"2025-04-02","arxiv_id":"2504.01348","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-prompt-instructed-zero-shot-composed","title":"Scaling Prompt Instructed Zero Shot Composed Image Retrieval with Image-Only Data","date":"2025-04-01","arxiv_id":"2504.00812","repositories_listed":0,"syntology":null},{"url":null,"slug":"cibr-cross-modal-information-bottleneck","title":"CIBR: Cross-modal Information Bottleneck Regularization for Robust CLIP Generalization","date":"2025-03-31","arxiv_id":"2503.24182","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiview-image-based-localization","title":"Multiview Image-Based Localization","date":"2025-03-30","arxiv_id":"2503.23577","repositories_listed":0,"syntology":null},{"url":null,"slug":"clean-image-may-be-dangerous-data-poisoning","title":"Clean Image May be Dangerous: Data Poisoning Attacks Against Deep Hashing","date":"2025-03-27","arxiv_id":"2503.21236","repositories_listed":0,"syntology":null},{"url":null,"slug":"fwd2bot-lvlm-visual-token-compression-with","title":"Fwd2Bot: LVLM Visual Token Compression with Double Forward Bottleneck","date":"2025-03-27","arxiv_id":"2503.21757","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-sparse-to-dense-camera-relocalization","title":"From Sparse to Dense: Camera Relocalization with Scene-Specific Detector from Feature Gaussian Splatting","date":"2025-03-25","arxiv_id":"2503.19358","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-agnostic-pose-regression-for-visual","title":"Scene-agnostic Pose Regression for Visual Localization","date":"2025-03-25","arxiv_id":"2503.19543","repositories_listed":0,"syntology":null},{"url":null,"slug":"locdiffusion-identifying-locations-on-earth","title":"LocDiffusion: Identifying Locations on Earth by Diffusing in the Hilbert Space","date":"2025-03-23","arxiv_id":"2503.18142","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-time-tells-us-an-explorative-study-of","title":"What Time Tells Us? An Explorative Study of Time Awareness Learned from Static Images","date":"2025-03-23","arxiv_id":"2503.17899","repositories_listed":0,"syntology":null},{"url":null,"slug":"good4cir-generating-detailed-synthetic","title":"good4cir: Generating Detailed Synthetic Captions for Composed Image Retrieval","date":"2025-03-22","arxiv_id":"2503.17871","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-product-search-interfaces-with","title":"Enhancing Product Search Interfaces with Sketch-Guided Diffusion and Language Agents","date":"2025-03-21","arxiv_id":"2504.08739","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompthash-affinity-prompted-collaborative","title":"PromptHash: Affinity-Prompted Collaborative Cross-Modal Learning for Adaptive Hashing Retrieval","date":"2025-03-20","arxiv_id":"2503.16064","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-medical-image-retrieval-via","title":"Revisiting Medical Image Retrieval via Knowledge Consolidation","date":"2025-03-12","arxiv_id":"2503.09370","repositories_listed":0,"syntology":null},{"url":null,"slug":"mari-material-retrieval-integration-across","title":"MaRI: Material Retrieval Integration across Domains","date":"2025-03-11","arxiv_id":"2503.08111","repositories_listed":0,"syntology":null},{"url":null,"slug":"find-your-needle-small-object-image-retrieval","title":"Find your Needle: Small Object Image Retrieval via Multi-Object Attention Optimization","date":"2025-03-10","arxiv_id":"2503.07038","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-hashing-based-on-reconstruction","title":"Zero-Shot Hashing Based on Reconstruction With Part Alignment","date":"2025-03-10","arxiv_id":"2503.07037","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-generalization-for-zero-shot","title":"Data-Efficient Generalization for Zero-shot Composed Image Retrieval","date":"2025-03-07","arxiv_id":"2503.05204","repositories_listed":0,"syntology":null},{"url":null,"slug":"radir-a-scalable-framework-for-multi-grained","title":"RadIR: A Scalable Framework for Multi-Grained Medical Image Retrieval via Radiology Report Mining","date":"2025-03-06","arxiv_id":"2503.04653","repositories_listed":0,"syntology":null},{"url":null,"slug":"class-independent-increment-an-efficient","title":"Class-Independent Increment: An Efficient Approach for Multi-label Class-Incremental Learning","date":"2025-03-01","arxiv_id":"2503.00515","repositories_listed":0,"syntology":null},{"url":null,"slug":"2502-20826","title":"CoTMR: Chain-of-Thought Multi-Scale Reasoning for Training-Free Zero-Shot Composed Image Retrieval","date":"2025-02-28","arxiv_id":"2502.20826","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-importance-of-text-preprocessing-for","title":"On the Importance of Text Preprocessing for Multimodal Representation Learning and Pathology Report Generation","date":"2025-02-26","arxiv_id":"2502.19285","repositories_listed":0,"syntology":null},{"url":null,"slug":"elip-enhanced-visual-language-foundation","title":"ELIP: Enhanced Visual-Language Foundation Models for Image Retrieval","date":"2025-02-21","arxiv_id":"2502.15682","repositories_listed":0,"syntology":null},{"url":null,"slug":"scale-up-composed-image-retrieval-learning","title":"Scale Up Composed Image Retrieval Learning via Modification Text Generation","date":"2025-02-21","arxiv_id":"2504.05316","repositories_listed":0,"syntology":null},{"url":null,"slug":"descriminative-generative-custom-tokens-for","title":"Descriminative-Generative Custom Tokens for Vision-Language Models","date":"2025-02-17","arxiv_id":"2502.12095","repositories_listed":0,"syntology":null},{"url":null,"slug":"precise-gps-denied-uav-self-positioning-via","title":"Precise GPS-Denied UAV Self-Positioning via Context-Enhanced Cross-View Geo-Localization","date":"2025-02-17","arxiv_id":"2502.11408","repositories_listed":0,"syntology":null}],"record_sha256":"c63ea515ac1a91736e787923716bda650b41bb64b791043dec092b247aae4224","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}