{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-retrieval/papers/5","list_of":"/task/text-retrieval","task":"Text Retrieval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":7,"rows_per_page":100,"rows":[401,500],"of":671,"counts":{"archive_papers_tagged":671,"with_a_code_link":335,"where_syntology_ran_a_sample":117,"not_listed_spam_title":0,"listed":671,"listed_where_code_ran":117,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":99,"every_run_a_failure_of_syntologys_instrument":18,"listed_with_a_run_with_no_instrument_failure":99,"listed_every_run_a_failure_of_syntologys_instrument":18,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-retrieval","prev":"/task/text-retrieval/papers/4","next":"/task/text-retrieval/papers/6","papers":[{"url":null,"slug":"robotic-state-recognition-with-image-to-text","title":"Robotic State Recognition with Image-to-Text Retrieval Task of Pre-Trained Vision-Language Model and Black-Box Optimization","date":"2024-10-30","arxiv_id":"2410.22707","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-audio-language-models-understand","title":"Do Audio-Language Models Understand Linguistic Variations?","date":"2024-10-21","arxiv_id":"2410.16505","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-general-text-embedding-model","title":"Improving General Text Embedding Model: Tackling Task Conflict and Data Imbalance through Model Merging","date":"2024-10-19","arxiv_id":"2410.15035","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-coarse-grained-matching-in-video-text","title":"Beyond Coarse-Grained Matching in Video-Text Retrieval","date":"2024-10-16","arxiv_id":"2410.12407","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctrlsynth-controllable-image-text-synthesis","title":"CtrlSynth: Controllable Image Text Synthesis for Data-Efficient Multimodal Learning","date":"2024-10-15","arxiv_id":"2410.11963","repositories_listed":0,"syntology":null},{"url":null,"slug":"lamp-language-motion-pretraining-for-motion","title":"LaMP: Language-Motion Pretraining for Motion Generation, Retrieval, and Captioning","date":"2024-10-09","arxiv_id":"2410.07093","repositories_listed":0,"syntology":null},{"url":null,"slug":"anyattack-towards-large-scale-self-supervised","title":"AnyAttack: Towards Large-scale Self-supervised Adversarial Attacks on Vision-language Models","date":"2024-10-07","arxiv_id":"2410.05346","repositories_listed":0,"syntology":null},{"url":null,"slug":"collap-contrastive-long-form-language-audio","title":"CoLLAP: Contrastive Long-form Language-Audio Pretraining with Musical Temporal Structure Augmentation","date":"2024-10-03","arxiv_id":"2410.02271","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-environmental-state-recognition-with","title":"Robotic Environmental State Recognition with Pre-Trained Vision-Language Models and Black-Box Optimization","date":"2024-09-26","arxiv_id":"2409.17519","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffatr-diffusion-based-generative-modeling","title":"DiffATR: Diffusion-based Generative Modeling for Audio-Text Retrieval","date":"2024-09-16","arxiv_id":"2409.10025","repositories_listed":0,"syntology":null},{"url":null,"slug":"nevlp-noise-robust-framework-for-efficient","title":"NEVLP: Noise-Robust Framework for Efficient Vision-Language Pre-training","date":"2024-09-15","arxiv_id":"2409.09582","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-q-a-text-retrieval-with-ranking","title":"Enhancing Q&A Text Retrieval with Ranking Models: Benchmarking, fine-tuning and deploying Rerankers for RAG","date":"2024-09-12","arxiv_id":"2409.07691","repositories_listed":0,"syntology":null},{"url":null,"slug":"pushing-the-limits-of-vision-language-models","title":"Pushing the Limits of Vision-Language Models in Remote Sensing without Human Annotations","date":"2024-09-11","arxiv_id":"2409.07048","repositories_listed":0,"syntology":null},{"url":null,"slug":"nllb-e5-a-scalable-multilingual-retrieval","title":"Benchmarking and Building Zero-Shot Hindi Retrieval Model with Hindi-BEIR and NLLB-E5","date":"2024-09-09","arxiv_id":"2409.05401","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-embedding-with-contrastive-fine","title":"Improving embedding with contrastive fine-tuning on small datasets with expert-augmented scores","date":"2024-08-19","arxiv_id":"2408.11868","repositories_listed":0,"syntology":null},{"url":null,"slug":"navero-unlocking-fine-grained-semantics-for","title":"NAVERO: Unlocking Fine-Grained Semantics for Video-Language Compositionality","date":"2024-08-18","arxiv_id":"2408.09511","repositories_listed":0,"syntology":null},{"url":null,"slug":"mamba-retriever-utilizing-mamba-for-effective","title":"Mamba Retriever: Utilizing Mamba for Effective and Efficient Dense Retrieval","date":"2024-08-15","arxiv_id":"2408.08066","repositories_listed":0,"syntology":null},{"url":null,"slug":"pairing-clustered-inverted-indexes-with-knn","title":"Pairing Clustered Inverted Indexes with kNN Graphs for Fast Approximate Retrieval over Learned Sparse Representations","date":"2024-08-08","arxiv_id":"2408.04443","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01363","title":"Toward Automatic Relevance Judgment using Vision--Language Models for Image--Text Retrieval Evaluation","date":"2024-08-02","arxiv_id":"2408.01363","repositories_listed":0,"syntology":null},{"url":"/paper/mgte-generalized-long-context-text","slug":"mgte-generalized-long-context-text","title":"mGTE: Generalized Long-Context Text Representation and Reranking Models for Multilingual Text Retrieval","date":"2024-07-29","arxiv_id":"2407.19669","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mgte-generalized-long-context-text#ran","syntology_url":"https://syntology.ai/paper/2407.19669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.19669"}},"official":null}},{"url":null,"slug":"assessing-brittleness-of-image-text-retrieval","title":"Assessing Brittleness of Image-Text Retrieval Benchmarks from Vision-Language Models Perspective","date":"2024-07-21","arxiv_id":"2407.15239","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-misinformation-detection-using","title":"Multimodal Misinformation Detection using Large Vision-Language Models","date":"2024-07-19","arxiv_id":"2407.14321","repositories_listed":0,"syntology":null},{"url":null,"slug":"cosmoclip-generalizing-large-vision-language","title":"CosmoCLIP: Generalizing Large Vision-Language Models for Astronomical Imaging","date":"2024-07-10","arxiv_id":"2407.07315","repositories_listed":0,"syntology":null},{"url":null,"slug":"ea-vtr-event-aware-video-text-retrieval","title":"EA-VTR: Event-Aware Video-Text Retrieval","date":"2024-07-10","arxiv_id":"2407.07478","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-make-cross-encoder-a-good-teacher-for-1","title":"How to Make Cross Encoder a Good Teacher for Efficient Image-Text Retrieval?","date":"2024-07-10","arxiv_id":"2407.07479","repositories_listed":0,"syntology":null},{"url":null,"slug":"ceia-clip-based-event-image-alignment-for","title":"CEIA: CLIP-Based Event-Image Alignment for Open-World Event-Based Understanding","date":"2024-07-09","arxiv_id":"2407.06611","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-memory-3-language-modeling-with-explicit","title":"$\\text{Memory}^3$: Language Modeling with Explicit Memory","date":"2024-07-01","arxiv_id":"2407.01178","repositories_listed":0,"syntology":null},{"url":null,"slug":"pathalign-a-vision-language-model-for-whole","title":"PathAlign: A vision-language model for whole slide images in histopathology","date":"2024-06-27","arxiv_id":"2406.19578","repositories_listed":0,"syntology":null},{"url":null,"slug":"ace-a-generative-cross-modal-retrieval","title":"ACE: A Generative Cross-Modal Retrieval Framework with Coarse-To-Fine Semantic Modeling","date":"2024-06-25","arxiv_id":"2406.17507","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-d-merit-of-partial-annotation-on","title":"Evaluating D-MERIT of Partial-annotation on Information Retrieval","date":"2024-06-23","arxiv_id":"2406.16048","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-scale-temporal-difference-transformer","title":"Multi-Scale Temporal Difference Transformer for Video-Text Retrieval","date":"2024-06-23","arxiv_id":"2406.16111","repositories_listed":0,"syntology":null},{"url":null,"slug":"re-adaptir-improving-information-retrieval","title":"RE-AdaptIR: Improving Information Retrieval through Reverse Engineered Adaptation","date":"2024-06-20","arxiv_id":"2406.14764","repositories_listed":0,"syntology":null},{"url":null,"slug":"unifying-multimodal-retrieval-via-document","title":"Unifying Multimodal Retrieval via Document Screenshot Embedding","date":"2024-06-17","arxiv_id":"2406.11251","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-knowledge-retrieval-with-in-context","title":"Enhancing Knowledge Retrieval with In-Context Learning and Semantic Search through Generative AI","date":"2024-06-13","arxiv_id":"2406.09621","repositories_listed":0,"syntology":null},{"url":null,"slug":"beat-bi-directional-one-to-many-embedding","title":"Beat: Bi-directional One-to-Many Embedding Alignment for Text-based Person Retrieval","date":"2024-06-09","arxiv_id":"2406.05620","repositories_listed":0,"syntology":null},{"url":null,"slug":"henasy-learning-to-assemble-scene-entities","title":"HENASY: Learning to Assemble Scene-Entities for Egocentric Video-Language Model","date":"2024-06-01","arxiv_id":"2406.00307","repositories_listed":0,"syntology":null},{"url":null,"slug":"jina-clip-your-clip-model-is-also-your-text","title":"Jina CLIP: Your CLIP Model Is Also Your Text Retriever","date":"2024-05-30","arxiv_id":"2405.20204","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-grounded-adaptation-strategy-for","title":"Knowledge-grounded Adaptation Strategy for Vision-language Models: Building Unique Case-set for Screening Mammograms for Residents Training","date":"2024-05-30","arxiv_id":"2405.19675","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-sign-language-video","title":"Uncertainty-aware sign language video retrieval with probability distribution modeling","date":"2024-05-30","arxiv_id":"2405.19689","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-many-to-many-relationships-for","title":"Multimodal Adversarial Defense for Vision-Language Models by Leveraging One-To-Many Relationships","date":"2024-05-29","arxiv_id":"2405.18770","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-diversity-improves-vision","title":"Multilingual Diversity Improves Vision-Language Representations","date":"2024-05-27","arxiv_id":"2405.16915","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-effect-of-using","title":"Understanding the Effect of using Semantically Meaningful Tokens for Visual Representation Learning","date":"2024-05-26","arxiv_id":"2405.16401","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-learning-for-finely-categorized-image","title":"Active Learning for Finely-Categorized Image-Text Retrieval by Selecting Hard Negative Unpaired Samples","date":"2024-05-25","arxiv_id":"2405.16301","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-study-of-excitation-and","title":"An Empirical Study of Excitation and Aggregation Design Adaptions in CLIP4Clip for Video-Text Retrieval","date":"2024-05-25","arxiv_id":"2406.01604","repositories_listed":0,"syntology":null},{"url":"/paper/global-local-information-soft-alignment-for","slug":"global-local-information-soft-alignment-for","title":"Global–Local Information Soft-Alignment for Cross-Modal Remote-Sensing Image–Text Retrieval","date":"2024-05-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-enhanced-zero-shot-video-captioning","title":"RETTA: Retrieval-Enhanced Test-Time Adaptation for Zero-Shot Video Captioning","date":"2024-05-11","arxiv_id":"2405.07046","repositories_listed":0,"syntology":null},{"url":null,"slug":"distance-sampling-based-paraphraser","title":"Distance Sampling-based Paraphraser Leveraging ChatGPT for Text Data Manipulation","date":"2024-05-01","arxiv_id":"2405.00367","repositories_listed":0,"syntology":null},{"url":null,"slug":"visla-benchmark-evaluating-embedding","title":"VISLA Benchmark: Evaluating Embedding Sensitivity to Semantic and Lexical Alterations","date":"2024-04-25","arxiv_id":"2404.16365","repositories_listed":0,"syntology":null},{"url":null,"slug":"urbancross-enhancing-satellite-image-text","title":"UrbanCross: Enhancing Satellite Image-Text Retrieval with Cross-Domain Adaptation","date":"2024-04-22","arxiv_id":"2404.14241","repositories_listed":0,"syntology":null},{"url":null,"slug":"mindtuner-cross-subject-visual-decoding-with","title":"MindTuner: Cross-Subject Visual Decoding with Visual Fingerprint and Semantic Correction","date":"2024-04-19","arxiv_id":"2404.12630","repositories_listed":0,"syntology":null},{"url":null,"slug":"fectek-enhancing-term-weight-in-lexicon-based","title":"FecTek: Enhancing Term Weight in Lexicon-Based Retrieval with Feature Context and Term-level Knowledge","date":"2024-04-18","arxiv_id":"2404.12152","repositories_listed":0,"syntology":null},{"url":null,"slug":"text2taste-a-versatile-egocentric-vision","title":"TEXT2TASTE: A Versatile Egocentric Vision System for Intelligent Reading Assistance Using Large Language Model","date":"2024-04-14","arxiv_id":"2404.09254","repositories_listed":0,"syntology":null},{"url":"/paper/learning-with-noisy-correspondence","slug":"learning-with-noisy-correspondence","title":"Learning with Noisy Correspondence","date":"2024-04-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"havtr-improving-video-text-retrieval-through","title":"HaVTR: Improving Video-Text Retrieval Through Augmentation Using Large Foundation Models","date":"2024-04-07","arxiv_id":"2404.05083","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-training-large-language-models-for","title":"Self-Training Large Language Models for Improved Visual Program Synthesis With Visual Reinforcement","date":"2024-04-06","arxiv_id":"2404.04627","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-retrieval-for-rag-based-question","title":"Improving Retrieval for RAG based Question Answering Models on Financial Documents","date":"2024-03-23","arxiv_id":"2404.07221","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-adversarial-transferability-of","title":"Improving Adversarial Transferability of Vision-Language Pre-training Models through Collaborative Multimodal Interaction","date":"2024-03-16","arxiv_id":"2403.10883","repositories_listed":0,"syntology":null},{"url":null,"slug":"luojiahog-a-hierarchy-oriented-geo-aware","title":"LuoJiaHOG: A Hierarchy Oriented Geo-aware Image Caption Dataset for Remote Sensing Image-Text Retrival","date":"2024-03-16","arxiv_id":"2403.10887","repositories_listed":0,"syntology":null},{"url":null,"slug":"refining-knowledge-transfer-on-audio-image","title":"Refining Knowledge Transfer on Audio-Image Temporal Agreement for Audio-Text Cross Retrieval","date":"2024-03-16","arxiv_id":"2403.10756","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiscale-matching-driven-by-cross-modal","title":"Multiscale Matching Driven by Cross-Modal Similarity Consistency for Audio-Text Retrieval","date":"2024-03-15","arxiv_id":"2403.10146","repositories_listed":0,"syntology":null},{"url":null,"slug":"clip-the-bias-how-useful-is-balancing-data-in","title":"CLIP the Bias: How Useful is Balancing Data in Multimodal Learning?","date":"2024-03-07","arxiv_id":"2403.04547","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-contrastive-learning-for-8192","title":"Multi-Task Contrastive Learning for 8192-Token Bilingual Text Embeddings","date":"2024-02-26","arxiv_id":"2402.17016","repositories_listed":0,"syntology":null},{"url":null,"slug":"unifying-latent-and-lexicon-representations","title":"Unifying Latent and Lexicon Representations for Effective Video-Text Retrieval","date":"2024-02-26","arxiv_id":"2402.16769","repositories_listed":0,"syntology":null},{"url":null,"slug":"more-multi-modal-retrieval-augmented","title":"MORE: Multi-mOdal REtrieval Augmented Generative Commonsense Reasoning","date":"2024-02-21","arxiv_id":"2402.13625","repositories_listed":0,"syntology":null},{"url":null,"slug":"pirb-a-comprehensive-benchmark-of-polish","title":"PIRB: A Comprehensive Benchmark of Polish Dense and Hybrid Text Retrieval Methods","date":"2024-02-20","arxiv_id":"2402.13350","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-learned-sparse-retrieval-for-image","title":"Multimodal Learned Sparse Retrieval for Image Suggestion","date":"2024-02-12","arxiv_id":"2402.07736","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-editing-for-video-retrieval","title":"Video Editing for Video Retrieval","date":"2024-02-04","arxiv_id":"2402.02335","repositories_listed":0,"syntology":null},{"url":"/paper/sycoca-symmetrizing-contrastive-captioners","slug":"sycoca-symmetrizing-contrastive-captioners","title":"SyCoCa: Symmetrizing Contrastive Captioners with Attentive Masking for Multimodal Alignment","date":"2024-01-04","arxiv_id":"2401.02137","repositories_listed":0,"syntology":null},{"url":null,"slug":"bev-clip-multi-modal-bev-retrieval","title":"BEV-TSR: Text-Scene Retrieval in BEV Space for Autonomous Driving","date":"2024-01-02","arxiv_id":"2401.01065","repositories_listed":0,"syntology":null},{"url":null,"slug":"accept-the-modality-gap-an-exploration-in-the","title":"Accept the Modality Gap: An Exploration in the Hyperbolic Space","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"building-vision-language-models-on-solid","title":"Building Vision-Language Models on Solid Foundations with Masked Distillation","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ote-exploring-accurate-scene-text-recognition","title":"OTE: Exploring Accurate Scene Text Recognition Using One Token","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"compress-align-curating-image-text-data-with","title":"Filter & Align: Leveraging Human Knowledge to Curate Image-Text Data","date":"2023-12-11","arxiv_id":"2312.06726","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-generative-language-models-for","title":"Leveraging Generative Language Models for Weakly Supervised Sentence Component Analysis in Video-Language Joint Learning","date":"2023-12-10","arxiv_id":"2312.06699","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightclip-learning-multi-level-interaction","title":"LightCLIP: Learning Multi-Level Interaction for Lightweight Vision-Language Models","date":"2023-12-01","arxiv_id":"2312.00674","repositories_listed":0,"syntology":null},{"url":null,"slug":"ig-captioner-information-gain-captioners-are","title":"IG Captioner: Information Gain Captioners are Strong Zero-shot Classifiers","date":"2023-11-27","arxiv_id":"2311.17072","repositories_listed":0,"syntology":null},{"url":null,"slug":"noisy-pair-corrector-for-dense-retrieval","title":"Noisy Pair Corrector for Dense Retrieval","date":"2023-11-07","arxiv_id":"2311.03798","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-prominent-fragments-enhancement","title":"A New Fine-grained Alignment Method for Image-text Matching","date":"2023-11-03","arxiv_id":"2311.02183","repositories_listed":0,"syntology":null},{"url":null,"slug":"flap-fast-language-audio-pre-training","title":"FLAP: Fast Language-Audio Pre-training","date":"2023-11-02","arxiv_id":"2311.01615","repositories_listed":0,"syntology":null},{"url":null,"slug":"mcad-multi-teacher-cross-modal-alignment","title":"MCAD: Multi-teacher Cross-modal Alignment Distillation for efficient image-text retrieval","date":"2023-10-30","arxiv_id":"2310.19654","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-autoregressive-retrieval-via","title":"End-to-End Autoregressive Retrieval via Bootstrapping for Smart Reply Systems","date":"2023-10-29","arxiv_id":"2310.18956","repositories_listed":0,"syntology":null},{"url":"/paper/silc-improving-vision-language-pretraining","slug":"silc-improving-vision-language-pretraining","title":"SILC: Improving Vision Language Pretraining with Self-Distillation","date":"2023-10-20","arxiv_id":"2310.13355","repositories_listed":0,"syntology":null},{"url":"/paper/direction-oriented-visual-semantic-embedding","slug":"direction-oriented-visual-semantic-embedding","title":"Direction-Oriented Visual-semantic Embedding Model for Remote Sensing Image-text Retrieval","date":"2023-10-12","arxiv_id":"2310.08276","repositories_listed":0,"syntology":null},{"url":null,"slug":"ziya-vl-bilingual-large-vision-language-model","title":"Ziya-Visual: Bilingual Large Vision-Language Model via Multi-Task Instruction Tuning","date":"2023-10-12","arxiv_id":"2310.08166","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-training-of-language-models","title":"Policy-Gradient Training of Language Models for Ranking","date":"2023-10-06","arxiv_id":"2310.04407","repositories_listed":0,"syntology":null},{"url":null,"slug":"constructing-image-text-pair-dataset-from","title":"Constructing Image-Text Pair Dataset from Books","date":"2023-10-03","arxiv_id":"2310.01936","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-alignment-network-for-cross","title":"Uncertainty-Aware Alignment Network for Cross-Domain Video-Text Retrieval","date":"2023-09-21","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-alignment-network-for-cross-1","title":"Uncertainty-Aware Alignment Network for Cross-Domain Video-Text Retrieval","date":"2023-09-21","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-multi-view-visual-semantic","title":"Dynamic Visual Semantic Sub-Embeddings and Fast Re-Ranking","date":"2023-09-15","arxiv_id":"2309.08154","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-relation-alignment-for-composed-image","title":"Dual Relation Alignment for Composed Image Retrieval","date":"2023-09-05","arxiv_id":"2309.02169","repositories_listed":0,"syntology":null},{"url":"/paper/contrastive-feature-masking-open-vocabulary","slug":"contrastive-feature-masking-open-vocabulary","title":"Contrastive Feature Masking Open-Vocabulary Vision Transformer","date":"2023-09-02","arxiv_id":"2309.00775","repositories_listed":0,"syntology":null},{"url":null,"slug":"killing-two-birds-with-one-stone-can-an-audio","title":"Killing two birds with one stone: Can an audio captioning system also be used for audio-text retrieval?","date":"2023-08-29","arxiv_id":"2308.15090","repositories_listed":0,"syntology":null},{"url":null,"slug":"dlip-distilling-language-image-pre-training","title":"DLIP: Distilling Language-Image Pre-training","date":"2023-08-24","arxiv_id":"2308.12956","repositories_listed":0,"syntology":null},{"url":null,"slug":"eve-efficient-vision-language-pre-training","title":"EVE: Efficient Vision-Language Pre-training with Masked Prediction and Modality-Aware MoE","date":"2023-08-23","arxiv_id":"2308.11971","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-retrieval-and-multi-stage-text-ranking","title":"Hybrid Retrieval and Multi-stage Text Ranking Solution at TREC 2022 Deep Learning Track","date":"2023-08-23","arxiv_id":"2308.12039","repositories_listed":0,"syntology":null},{"url":null,"slug":"free-atm-exploring-unsupervised-learning-on","title":"Free-ATM: Exploring Unsupervised Learning on Diffusion-Generated Images with Free Attention Masks","date":"2023-08-13","arxiv_id":"2308.06739","repositories_listed":0,"syntology":null},{"url":null,"slug":"embedding-based-retrieval-with-llm-for","title":"Embedding-based Retrieval with LLM for Effective Agriculture Information Extracting from Unstructured Data","date":"2023-08-06","arxiv_id":"2308.03107","repositories_listed":0,"syntology":null},{"url":null,"slug":"defense-of-adversarial-ranking-attack-in-text","title":"Defense of Adversarial Ranking Attack in Text Retrieval: Benchmark and Baseline via Detection","date":"2023-07-31","arxiv_id":"2307.16816","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-visual-language-foundation-model","title":"Towards a Visual-Language Foundation Model for Computational Pathology","date":"2023-07-24","arxiv_id":"2307.12914","repositories_listed":0,"syntology":null},{"url":null,"slug":"extracting-molecular-properties-from-natural","title":"Extracting Molecular Properties from Natural Language with Multimodal Contrastive Learning","date":"2023-07-22","arxiv_id":"2307.12996","repositories_listed":0,"syntology":null}],"record_sha256":"13bbadc1e5a55153711e5ca4ad665d4fee23ba079742f3e2886745a2c63471fd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}