{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/cross-modal-retrieval/papers/4","list_of":"/task/cross-modal-retrieval","task":"Cross-Modal Retrieval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":6,"rows_per_page":100,"rows":[301,400],"of":522,"counts":{"archive_papers_tagged":522,"with_a_code_link":244,"where_syntology_ran_a_sample":70,"not_listed_spam_title":0,"listed":522,"listed_where_code_ran":70,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":56,"every_run_a_failure_of_syntologys_instrument":14,"listed_with_a_run_with_no_instrument_failure":56,"listed_every_run_a_failure_of_syntologys_instrument":14,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/cross-modal-retrieval","prev":"/task/cross-modal-retrieval/papers/3","next":"/task/cross-modal-retrieval/papers/5","papers":[{"url":null,"slug":"efficient-and-versatile-robust-fine-tuning-of","title":"Efficient and Versatile Robust Fine-Tuning of Zero-shot Models","date":"2024-08-11","arxiv_id":"2408.05749","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-noisy-correspondence-learning","title":"Disentangled Noisy Correspondence Learning","date":"2024-08-10","arxiv_id":"2408.05503","repositories_listed":0,"syntology":null},{"url":null,"slug":"start-from-video-music-retrieval-an-inter","title":"Start from Video-Music Retrieval: An Inter-Intra Modal Loss for Cross Modal Retrieval","date":"2024-07-28","arxiv_id":"2407.19415","repositories_listed":0,"syntology":null},{"url":null,"slug":"revolutionizing-text-to-image-retrieval-as","title":"Revolutionizing Text-to-Image Retrieval as Autoregressive Token-to-Voken Generation","date":"2024-07-24","arxiv_id":"2407.17274","repositories_listed":0,"syntology":null},{"url":null,"slug":"second-place-solution-of-wsdm2023-toloka","title":"Second Place Solution of WSDM2023 Toloka Visual Question Answering Challenge","date":"2024-07-05","arxiv_id":"2407.04255","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-compositions-enhance-vision-language","title":"Semantic Compositions Enhance Vision-Language Contrastive Learning","date":"2024-07-01","arxiv_id":"2407.01408","repositories_listed":0,"syntology":null},{"url":null,"slug":"mate-meet-at-the-embedding-connecting-images","title":"MATE: Meet At The Embedding -- Connecting Images with Long Texts","date":"2024-06-26","arxiv_id":"2407.09541","repositories_listed":0,"syntology":null},{"url":null,"slug":"ace-a-generative-cross-modal-retrieval","title":"ACE: A Generative Cross-Modal Retrieval Framework with Coarse-To-Fine Semantic Modeling","date":"2024-06-25","arxiv_id":"2406.17507","repositories_listed":0,"syntology":null},{"url":"/paper/what-if-we-recaption-billions-of-web-images","slug":"what-if-we-recaption-billions-of-web-images","title":"What If We Recaption Billions of Web Images with LLaMA-3?","date":"2024-06-12","arxiv_id":"2406.08478","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-clip-help-clip-in-learning-3d","title":"No Captions, No Problem: Captionless 3D-CLIP Alignment with Hard Negatives via CLIP Knowledge and LLMs","date":"2024-06-04","arxiv_id":"2406.02202","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-generative-embedding-model","title":"Multi-Modal Generative Embedding Model","date":"2024-05-29","arxiv_id":"2405.19333","repositories_listed":0,"syntology":null},{"url":null,"slug":"rreh-reconstruction-relations-embedded","title":"RREH: Reconstruction Relations Embedded Hashing for Semi-Paired Cross-Modal Retrieval","date":"2024-05-28","arxiv_id":"2405.17777","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-vision-language-pretraining-for","title":"Distilling Vision-Language Pretraining for Efficient Cross-Modal Retrieval","date":"2024-05-23","arxiv_id":"2405.14726","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-cross-modal-backward-compatible","title":"Towards Cross-modal Backward-compatible Representation Learning for Vision-Language Models","date":"2024-05-23","arxiv_id":"2405.14715","repositories_listed":0,"syntology":null},{"url":null,"slug":"mvbind-self-supervised-music-recommendation","title":"MVBIND: Self-Supervised Music Recommendation For Videos Via Embedding Space Binding","date":"2024-05-15","arxiv_id":"2405.09286","repositories_listed":0,"syntology":null},{"url":"/paper/global-local-information-soft-alignment-for","slug":"global-local-information-soft-alignment-for","title":"Global–Local Information Soft-Alignment for Cross-Modal Remote-Sensing Image–Text Retrieval","date":"2024-05-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"all-in-one-framework-for-multimodal-re","title":"All in One Framework for Multimodal Re-identification in the Wild","date":"2024-05-08","arxiv_id":"2405.04741","repositories_listed":0,"syntology":null},{"url":null,"slug":"com3d-leveraging-cross-view-correspondence","title":"COM3D: Leveraging Cross-View Correspondence and Cross-Modal Mining for 3D Retrieval","date":"2024-05-07","arxiv_id":"2405.04103","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-retrieval-augmented-task","title":"Understanding Retrieval-Augmented Task Adaptation for Vision-Language Models","date":"2024-05-02","arxiv_id":"2405.01468","repositories_listed":0,"syntology":null},{"url":null,"slug":"anchor-aware-deep-metric-learning-for-audio","title":"Anchor-aware Deep Metric Learning for Audio-visual Retrieval","date":"2024-04-21","arxiv_id":"2404.13789","repositories_listed":0,"syntology":null},{"url":null,"slug":"wills-aligner-a-robust-multi-subject-brain","title":"Wills Aligner: Multi-Subject Collaborative Brain Visual Decoding","date":"2024-04-20","arxiv_id":"2404.13282","repositories_listed":0,"syntology":null},{"url":"/paper/learning-with-noisy-correspondence","slug":"learning-with-noisy-correspondence","title":"Learning with Noisy Correspondence","date":"2024-04-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-optimal-transport-framework-for","title":"A Unified Optimal Transport Framework for Cross-Modal Retrieval with Noisy Labels","date":"2024-03-20","arxiv_id":"2403.13480","repositories_listed":0,"syntology":null},{"url":null,"slug":"tri-modal-motion-retrieval-by-learning-a","title":"Tri-Modal Motion Retrieval by Learning a Joint Embedding Space","date":"2024-03-01","arxiv_id":"2403.00691","repositories_listed":0,"syntology":null},{"url":"/paper/generative-cross-modal-retrieval-memorizing","slug":"generative-cross-modal-retrieval-memorizing","title":"Generative Cross-Modal Retrieval: Memorizing Images in Multimodal Language Models for Retrieval and Beyond","date":"2024-02-16","arxiv_id":"2402.10805","repositories_listed":0,"syntology":{"n":12,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":12,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/generative-cross-modal-retrieval-memorizing#ran","syntology_url":"https://syntology.ai/paper/2402.10805","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10805"}},"official":null}},{"url":null,"slug":"mind-the-modality-gap-towards-a-remote","title":"Mind the Modality Gap: Towards a Remote Sensing Vision-Language Model via Cross-modal Alignment","date":"2024-02-15","arxiv_id":"2402.09816","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-for-captioning-and","title":"Large Language Models for Captioning and Retrieving Remote Sensing Images","date":"2024-02-09","arxiv_id":"2402.06475","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-coordination-across-a-diverse-set","title":"Cross-Modal Coordination Across a Diverse Set of Input Modalities","date":"2024-01-29","arxiv_id":"2401.16347","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-medical-vision-language-contrastive","title":"Enhancing medical vision-language contrastive learning via inter-matching relation modelling","date":"2024-01-19","arxiv_id":"2401.10501","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-chatgpt-for-biology-and-medicine-a","title":"Developing ChatGPT for Biology and Medicine: A Complete Review of Biomedical Question Answering","date":"2024-01-15","arxiv_id":"2401.07510","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-prototypical-voting-with","title":"Fine-grained Prototypical Voting with Heterogeneous Mixup for Semi-supervised 2D-3D Cross-modal Retrieval","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/noisy-correspondence-learning-with-self","slug":"noisy-correspondence-learning-with-self","title":"Noisy Correspondence Learning with Self-Reinforcing Errors Mitigation","date":"2023-12-27","arxiv_id":"2312.16478","repositories_listed":0,"syntology":null},{"url":null,"slug":"masked-contrastive-reconstruction-for-cross","title":"Masked Contrastive Reconstruction for Cross-modal Medical Image-Report Retrieval","date":"2023-12-26","arxiv_id":"2312.15840","repositories_listed":0,"syntology":null},{"url":null,"slug":"cl2cm-improving-cross-lingual-cross-modal","title":"CL2CM: Improving Cross-Lingual Cross-Modal Retrieval via Cross-Lingual Knowledge Transfer","date":"2023-12-14","arxiv_id":"2312.08984","repositories_listed":0,"syntology":null},{"url":null,"slug":"wikimute-a-web-sourced-dataset-of-semantic","title":"WikiMuTe: A web-sourced dataset of semantic descriptions for music audio","date":"2023-12-14","arxiv_id":"2312.09207","repositories_listed":0,"syntology":null},{"url":null,"slug":"uni3dl-unified-model-for-3d-and-language","title":"Uni3DL: Unified Model for 3D and Language Understanding","date":"2023-12-05","arxiv_id":"2312.03026","repositories_listed":0,"syntology":null},{"url":null,"slug":"t3d-towards-3d-medical-image-understanding","title":"T3D: Advancing 3D Medical Vision-Language Pre-training by Learning Multi-View Visual Consistency","date":"2023-12-03","arxiv_id":"2312.01529","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-stage-triplet-loss-training-with","title":"Two-Stage Triplet Loss Training with Curriculum Augmentation for Audio-Visual Retrieval","date":"2023-10-20","arxiv_id":"2310.13451","repositories_listed":0,"syntology":null},{"url":"/paper/direction-oriented-visual-semantic-embedding","slug":"direction-oriented-visual-semantic-embedding","title":"Direction-Oriented Visual-semantic Embedding Model for Remote Sensing Image-text Retrieval","date":"2023-10-12","arxiv_id":"2310.08276","repositories_listed":0,"syntology":null},{"url":null,"slug":"sound-source-localization-is-all-about-cross","title":"Sound Source Localization is All about Cross-Modal Alignment","date":"2023-09-19","arxiv_id":"2309.10724","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-view-curricular-optimal-transport-for","title":"Dual-view Curricular Optimal Transport for Cross-lingual Cross-modal Retrieval","date":"2023-09-11","arxiv_id":"2309.05451","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-retrieval-meets-inference","title":"Cross-Modal Retrieval Meets Inference:Improving Zero-Shot Classification with Cross-Modal Retrieval","date":"2023-08-29","arxiv_id":"2308.15273","repositories_listed":0,"syntology":null},{"url":null,"slug":"extending-cross-modal-retrieval-with","title":"Extending Cross-Modal Retrieval with Interactive Learning to Improve Image Retrieval Performance in Forensics","date":"2023-08-28","arxiv_id":"2308.14786","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-and-audio-are-images-a-cross-modal","title":"Video and Audio are Images: A Cross-Modal Mixer for Original Data on Video-Audio Retrieval","date":"2023-08-26","arxiv_id":"2308.13820","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-scoping-review-on-multimodal-deep-learning","title":"A scoping review on multimodal deep learning in biomedical images and texts","date":"2023-07-14","arxiv_id":"2307.07362","repositories_listed":0,"syntology":null},{"url":null,"slug":"pitl-cross-modal-retrieval-with-weakly","title":"PiTL: Cross-modal Retrieval with Weakly-supervised Vision-language Pre-training via Prompting","date":"2023-07-14","arxiv_id":"2307.07341","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-relation-extraction-with-cross","title":"Multimodal Relation Extraction with Cross-Modal Retrieval and Synthesis","date":"2023-05-25","arxiv_id":"2305.16166","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-vision-language-representaion","title":"Continual Vision-Language Representation Learning with Off-Diagonal Information","date":"2023-05-11","arxiv_id":"2305.07437","repositories_listed":0,"syntology":null},{"url":null,"slug":"instance-variant-loss-with-gaussian-rbf","title":"Instance-Variant Loss with Gaussian RBF Kernel for 3D Cross-modal Retriveal","date":"2023-05-07","arxiv_id":"2305.04239","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixer-image-to-multi-modal-retrieval-learning","title":"Category-Oriented Representation Learning for Image to Multi-Modal Retrieval","date":"2023-05-06","arxiv_id":"2305.03972","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-lifelong-cross-modal-hashing","title":"Deep Lifelong Cross-modal Hashing","date":"2023-04-26","arxiv_id":"2304.13357","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-specific-debiasing-for-better-image","title":"Sample-Specific Debiasing for Better Image-Text Models","date":"2023-04-25","arxiv_id":"2304.13181","repositories_listed":0,"syntology":null},{"url":null,"slug":"covlr-coordinating-cross-modal-consistency","title":"CoVLR: Coordinating Cross-Modal Consistency and Intra-Modal Structure for Vision-Language Retrieval","date":"2023-04-15","arxiv_id":"2304.07567","repositories_listed":0,"syntology":null},{"url":null,"slug":"exposing-and-mitigating-spurious-correlations","title":"Exposing and Mitigating Spurious Correlations for Cross-Modal Retrieval","date":"2023-04-06","arxiv_id":"2304.03391","repositories_listed":0,"syntology":null},{"url":null,"slug":"hindi-as-a-second-language-improving-visually","title":"Hindi as a Second Language: Improving Visually Grounded Speech with Semantically Similar Samples","date":"2023-03-30","arxiv_id":"2303.17517","repositories_listed":0,"syntology":null},{"url":null,"slug":"mxm-clr-a-unified-framework-for-contrastive","title":"MXM-CLR: A Unified Framework for Contrastive Learning of Multifold Cross-Modal Representations","date":"2023-03-20","arxiv_id":"2303.10839","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-retrieval-with-improved-graph","title":"Cross-modal Retrieval with Improved Graph Convolution","date":"2023-03-07","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"data-leakage-in-cross-modal-retrieval","title":"Data leakage in cross-modal retrieval training: A case study","date":"2023-02-23","arxiv_id":"2302.12258","repositories_listed":0,"syntology":null},{"url":null,"slug":"x-tra-improving-chest-x-ray-tasks-with-cross","title":"X-TRA: Improving Chest X-ray Tasks with Cross-Modal Retrieval Augmentation","date":"2023-02-22","arxiv_id":"2302.11352","repositories_listed":0,"syntology":null},{"url":null,"slug":"clip-rr-improved-clip-network-for-relation","title":"VITR: Augmenting Vision Transformers with Relation-Focused Learning for Cross-Modal Information Retrieval","date":"2023-02-13","arxiv_id":"2302.06350","repositories_listed":0,"syntology":null},{"url":null,"slug":"distribution-aligned-feature-clustering-for","title":"Distribution Aligned Feature Clustering for Zero-Shot Sketch-Based Image Retrieval","date":"2023-01-17","arxiv_id":"2301.06685","repositories_listed":0,"syntology":null},{"url":null,"slug":"pix2map-cross-modal-retrieval-for-inferring","title":"Pix2Map: Cross-modal Retrieval for Inferring Street Maps from Images","date":"2023-01-10","arxiv_id":"2301.04224","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-as-a-foreign-language-beit-pretraining-1","title":"Image as a Foreign Language: BEiT Pretraining for Vision and Vision-Language Tasks","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-concordant-attention-via-target","title":"Learning Concordant Attention via Target-aware Alignment for Visible-Infrared Person Re-identification","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bagformer-better-cross-modal-retrieval-via","title":"BagFormer: Better Cross-Modal Retrieval via bag-wise interaction","date":"2022-12-29","arxiv_id":"2212.14322","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-based-disentanglement-with-distant","title":"Retrieval-based Disentangled Representation Learning with Natural Language Supervision","date":"2022-12-15","arxiv_id":"2212.07699","repositories_listed":0,"syntology":null},{"url":null,"slug":"scale-semantic-joint-decoupling-network-for","title":"Scale-Semantic Joint Decoupling Network for Image-text Retrieval in Remote Sensing","date":"2022-12-12","arxiv_id":"2212.05752","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-multiple-instance-learning-to-build","title":"Using Multiple Instance Learning to Build Multimodal Representations","date":"2022-12-11","arxiv_id":"2212.05561","repositories_listed":0,"syntology":null},{"url":null,"slug":"timbreclip-connecting-timbre-to-text-and","title":"TimbreCLIP: Connecting Timbre to Text and Images","date":"2022-11-21","arxiv_id":"2211.11225","repositories_listed":0,"syntology":null},{"url":null,"slug":"complete-cross-triplet-loss-in-label-space","title":"Complete Cross-triplet Loss in Label Space for Audio-visual Cross-modal Retrieval","date":"2022-11-07","arxiv_id":"2211.03434","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-shape-knowledge-graph-for-cross-domain-and","title":"3D Shape Knowledge Graph for Cross-domain 3D Shape Retrieval","date":"2022-10-27","arxiv_id":"2210.15136","repositories_listed":0,"syntology":null},{"url":null,"slug":"fad-vlp-fashion-vision-and-language-pre","title":"FaD-VLP: Fashion Vision-and-Language Pre-training towards Unified Retrieval and Captioning","date":"2022-10-26","arxiv_id":"2210.15028","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-by-hallucinating-vision-language-pre","title":"Learning by Hallucinating: Vision-Language Pre-training with Weak Supervision","date":"2022-10-24","arxiv_id":"2210.13591","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-search-method-of-technology-video","title":"Cross-modal Search Method of Technology Video based on Adversarial Learning and Feature Fusion","date":"2022-10-11","arxiv_id":"2210.05243","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-adaptive-multiple-visual-prototype","title":"Text-Adaptive Multiple Visual Prototype Matching for Video-Text Retrieval","date":"2022-09-27","arxiv_id":"2209.13307","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-manifold-hashing-a-divide-and-conquer","title":"Deep Manifold Hashing: A Divide-and-Conquer Approach for Semi-Paired Unsupervised Cross-Modal Retrieval","date":"2022-09-26","arxiv_id":"2209.12599","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-theoretic-hashing-for-zero-shot","title":"Information-Theoretic Hashing for Zero-Shot Cross-Modal Retrieval","date":"2022-09-26","arxiv_id":"2209.12491","repositories_listed":0,"syntology":null},{"url":"/paper/omnivl-one-foundation-model-for-image","slug":"omnivl-one-foundation-model-for-image","title":"OmniVL:One Foundation Model for Image-Language and Video-Language Tasks","date":"2022-09-15","arxiv_id":"2209.07526","repositories_listed":0,"syntology":null},{"url":null,"slug":"see-what-you-see-self-supervised-cross-modal","title":"See What You See: Self-supervised Cross-modal Retrieval of Visual Stimuli from Brain Activity","date":"2022-08-07","arxiv_id":"2208.03666","repositories_listed":0,"syntology":null},{"url":null,"slug":"paired-cross-modal-data-augmentation-for-fine","title":"Paired Cross-Modal Data Augmentation for Fine-Grained Image-to-Text Retrieval","date":"2022-07-29","arxiv_id":"2207.14428","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-asymmetric-label-guided-hashing-for","title":"Adaptive Asymmetric Label-guided Hashing for Multimedia Search","date":"2022-07-26","arxiv_id":"2207.12625","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-cross-modal-knowledge-sharing-pre","title":"Contrastive Cross-Modal Knowledge Sharing Pre-training for Vision-Language Representation Learning and Retrieval","date":"2022-07-02","arxiv_id":"2207.00733","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-transformation-invariance-and","title":"Exploiting Transformation Invariance and Equivariance for Self-supervised Sound Localisation","date":"2022-06-26","arxiv_id":"2206.12772","repositories_listed":0,"syntology":null},{"url":null,"slug":"emphasizing-complementary-samples-for-non","title":"Emphasizing Complementary Samples for Non-literal Cross-modal Retrieval","date":"2022-06-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hivlp-hierarchical-vision-language-pre","title":"HiVLP: Hierarchical Vision-Language Pre-Training for Fast Image-Text Retrieval","date":"2022-05-24","arxiv_id":"2205.12105","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-supervised-information-bottleneck","title":"Deep Supervised Information Bottleneck Hashing for Cross-modal Retrieval based Computer-aided Diagnosis","date":"2022-05-06","arxiv_id":"2205.08365","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-based-cross-modal-retrieval-with","title":"Uncertainty-based Cross-Modal Retrieval with Probabilistic Representations","date":"2022-04-20","arxiv_id":"2204.09268","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-contrastive-hashing-for-cross","title":"Unsupervised Contrastive Hashing for Cross-Modal Retrieval in Remote Sensing","date":"2022-04-19","arxiv_id":"2204.08707","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-similarity-preserving-binary-codes","title":"Learning Similarity Preserving Binary Codes for Recommender Systems","date":"2022-04-18","arxiv_id":"2204.08569","repositories_listed":0,"syntology":null},{"url":"/paper/cots-collaborative-two-stream-vision-language","slug":"cots-collaborative-two-stream-vision-language","title":"COTS: Collaborative Two-Stream Vision-Language Pre-Training Model for Cross-Modal Retrieval","date":"2022-04-15","arxiv_id":"2204.07441","repositories_listed":0,"syntology":null},{"url":"/paper/vista-vision-and-scene-text-aggregation-for","slug":"vista-vision-and-scene-text-aggregation-for","title":"ViSTA: Vision and Scene Text Aggregation for Cross-Modal Retrieval","date":"2022-03-31","arxiv_id":"2203.16778","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-program-representations-for-food","title":"Learning Program Representations for Food Images and Cooking Recipes","date":"2022-03-30","arxiv_id":"2203.16071","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-media-scientific-research-achievements","title":"Cross-Media Scientific Research Achievements Retrieval Based on Deep Language Model","date":"2022-03-29","arxiv_id":"2203.15595","repositories_listed":0,"syntology":null},{"url":"/paper/lile-look-in-depth-before-looking-elsewhere-a","slug":"lile-look-in-depth-before-looking-elsewhere-a","title":"LILE: Look In-Depth before Looking Elsewhere -- A Dual Attention Network using Transformers for Cross-Modal Information Retrieval in Histopathology Archives","date":"2022-03-02","arxiv_id":"2203.01445","repositories_listed":0,"syntology":null},{"url":null,"slug":"discriminative-supervised-subspace-learning","title":"Discriminative Supervised Subspace Learning for Cross-modal Retrieval","date":"2022-01-26","arxiv_id":"2201.11843","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-unsupervised-contrastive-hashing-for","title":"Deep Unsupervised Contrastive Hashing for Large-Scale Cross-Modal Text-Image Retrieval in Remote Sensing","date":"2022-01-20","arxiv_id":"2201.08125","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-text-image-pair-is-not-enough-language","title":"A Text-Image Pair Is not Enough: Language-Vision Relation Inference with Auxiliary Modality Translation","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ei-clip-entity-aware-interventional","title":"EI-CLIP: Entity-Aware Interventional Contrastive Learning for E-Commerce Cross-Modal Retrieval","date":"2022-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"coco-bert-improving-video-language-pre","title":"CoCo-BERT: Improving Video-Language Pre-training with Contrastive Cross-modal Matching and Denoising","date":"2021-12-14","arxiv_id":"2112.07515","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-mutual-information-maximization-a","title":"Multi-Modal Mutual Information Maximization: A Novel Approach for Unsupervised Deep Cross-Modal Hashing","date":"2021-12-13","arxiv_id":"2112.06489","repositories_listed":0,"syntology":null}],"record_sha256":"d824f123757ce9d713143bf3ac0c9448c7909ed288de0ac01b7e743142266b70","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}