{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/cross-modal-retrieval/papers/3","list_of":"/task/cross-modal-retrieval","task":"Cross-Modal Retrieval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":6,"rows_per_page":100,"rows":[201,300],"of":522,"counts":{"archive_papers_tagged":522,"with_a_code_link":244,"where_syntology_ran_a_sample":70,"not_listed_spam_title":0,"listed":522,"listed_where_code_ran":70,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":56,"every_run_a_failure_of_syntologys_instrument":14,"listed_with_a_run_with_no_instrument_failure":56,"listed_every_run_a_failure_of_syntologys_instrument":14,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/cross-modal-retrieval","prev":"/task/cross-modal-retrieval/papers/2","next":"/task/cross-modal-retrieval/papers/4","papers":[{"url":"/paper/chef-cross-modal-hierarchical-embeddings-for","slug":"chef-cross-modal-hierarchical-embeddings-for","title":"CHEF: Cross-modal Hierarchical Embeddings for Food Domain Retrieval","date":"2021-02-04","arxiv_id":"2102.02547","repositories_listed":1,"syntology":null},{"url":"/paper/similarity-reasoning-and-filtration-for-image","slug":"similarity-reasoning-and-filtration-for-image","title":"Similarity Reasoning and Filtration for Image-Text Matching","date":"2021-01-05","arxiv_id":"2101.01368","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":6,"n_ran_checked":7,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":12,"phrase":"10 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/similarity-reasoning-and-filtration-for-image#ran","syntology_url":"https://syntology.ai/paper/2101.01368","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.01368"}},"official":{"repos":["Paranioar/SGRAF"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/cookie-contrastive-cross-modal-knowledge","slug":"cookie-contrastive-cross-modal-knowledge","title":"COOKIE: Contrastive Cross-Modal Knowledge Sharing Pre-Training for Vision-Language Representation","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/visualsparta-sparse-transformer-fragment","slug":"visualsparta-sparse-transformer-fragment","title":"VisualSparta: An Embarrassingly Simple Approach to Large-scale Text-to-Image Search with Weighted Bag-of-words","date":"2021-01-01","arxiv_id":"2101.00265","repositories_listed":1,"syntology":null},{"url":"/paper/stacmr-scene-text-aware-cross-modal-retrieval","slug":"stacmr-scene-text-aware-cross-modal-retrieval","title":"StacMR: Scene-Text Aware Cross-Modal Retrieval","date":"2020-12-08","arxiv_id":"2012.04329","repositories_listed":1,"syntology":null},{"url":"/paper/codecmr-cross-modal-retrieval-for-function","slug":"codecmr-cross-modal-retrieval-for-function","title":"CodeCMR: Cross-Modal Retrieval For Function-Level Binary Source Code Matching","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/coot-cooperative-hierarchical-transformer-for","slug":"coot-cooperative-hierarchical-transformer-for","title":"COOT: Cooperative Hierarchical Transformer for Video-Text Representation Learning","date":"2020-11-01","arxiv_id":"2011.00597","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coot-cooperative-hierarchical-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2011.00597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.00597"}},"official":{"repos":["gingsi/coot-videotext"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-metric-learning-for-tag-based","slug":"multimodal-metric-learning-for-tag-based","title":"Multimodal Metric Learning for Tag-based Music Retrieval","date":"2020-10-30","arxiv_id":"2010.16030","repositories_listed":1,"syntology":null},{"url":"/paper/learning-dual-semantic-relations-with-graph","slug":"learning-dual-semantic-relations-with-graph","title":"Learning Dual Semantic Relations with Graph Attention for Image-Text Matching","date":"2020-10-22","arxiv_id":"2010.11550","repositories_listed":1,"syntology":null},{"url":"/paper/dime-an-online-tool-for-the-visual-comparison","slug":"dime-an-online-tool-for-the-visual-comparison","title":"DIME: An Online Tool for the Visual Comparison of Cross-Modal Retrieval Models","date":"2020-10-19","arxiv_id":"2010.09641","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-visual-textual-alignment-for","slug":"fine-grained-visual-textual-alignment-for","title":"Fine-grained Visual Textual Alignment for Cross-Modal Retrieval using Transformer Encoders","date":"2020-08-12","arxiv_id":"2008.05231","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":12,"n_pointer_only":2,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fine-grained-visual-textual-alignment-for#ran","syntology_url":"https://syntology.ai/paper/2008.05231","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.05231"}},"official":{"repos":["mesnico/TERAN"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-acoustic-images-for-effective-self","slug":"leveraging-acoustic-images-for-effective-self","title":"Leveraging Acoustic Images for Effective Self-Supervised Audio Representation Learning","date":"2020-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/neural-methods-for-point-wise-dependency","slug":"neural-methods-for-point-wise-dependency","title":"Neural Methods for Point-wise Dependency Estimation","date":"2020-06-09","arxiv_id":"2006.05553","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":2,"n_no_contract":9,"n_pointer_only":3,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 2 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/neural-methods-for-point-wise-dependency#ran","syntology_url":"https://syntology.ai/paper/2006.05553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.05553"}},"official":{"repos":["yaohungt/Pointwise_Dependency_Neural_Estimation"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/sketch-less-for-more-on-the-fly-fine-grained-1","slug":"sketch-less-for-more-on-the-fly-fine-grained-1","title":"Sketch Less for More: On-the-Fly Fine-Grained Sketch-Based Image Retrieval","date":"2020-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/cobra-contrastive-bi-modal-representation","slug":"cobra-contrastive-bi-modal-representation","title":"COBRA: Contrastive Bi-Modal Representation Algorithm","date":"2020-05-07","arxiv_id":"2005.03687","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cobra-contrastive-bi-modal-representation#ran","syntology_url":"https://syntology.ai/paper/2005.03687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.03687"}},"official":{"repos":["ovshake/cobra"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/graph-structured-network-for-image-text","slug":"graph-structured-network-for-image-text","title":"Graph Structured Network for Image-Text Matching","date":"2020-04-01","arxiv_id":"2004.00277","repositories_listed":1,"syntology":null},{"url":"/paper/imram-iterative-matching-with-recurrent","slug":"imram-iterative-matching-with-recurrent","title":"IMRAM: Iterative Matching with Recurrent Attention Memory for Cross-Modal Image-Text Retrieval","date":"2020-03-08","arxiv_id":"2003.03772","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imram-iterative-matching-with-recurrent#ran","syntology_url":"https://syntology.ai/paper/2003.03772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.03772"}},"official":{"repos":["HuiChen24/IMRAM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sketch-less-for-more-on-the-fly-fine-grained","slug":"sketch-less-for-more-on-the-fly-fine-grained","title":"Sketch Less for More: On-the-Fly Fine-Grained Sketch Based Image Retrieval","date":"2020-02-24","arxiv_id":"2002.10310","repositories_listed":1,"syntology":null},{"url":"/paper/sketchformer-transformer-based-representation","slug":"sketchformer-transformer-based-representation","title":"Sketchformer: Transformer-based Representation for Sketched Structure","date":"2020-02-24","arxiv_id":"2002.10381","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sketchformer-transformer-based-representation#ran","syntology_url":"https://syntology.ai/paper/2002.10381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.10381"}},"official":null}},{"url":"/paper/target-oriented-deformation-of-visual","slug":"target-oriented-deformation-of-visual","title":"Target-Oriented Deformation of Visual-Semantic Embedding Space","date":"2019-10-15","arxiv_id":"1910.06514","repositories_listed":1,"syntology":null},{"url":"/paper/aligning-multilingual-word-embeddings-for","slug":"aligning-multilingual-word-embeddings-for","title":"Aligning Multilingual Word Embeddings for Cross-Modal Retrieval Task","date":"2019-10-08","arxiv_id":"1910.03291","repositories_listed":1,"syntology":null},{"url":"/paper/deep-joint-semantics-reconstructing-hashing","slug":"deep-joint-semantics-reconstructing-hashing","title":"Deep Joint-Semantics Reconstructing Hashing for Large-Scale Unsupervised Cross-Modal Retrieval","date":"2019-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/language-agnostic-visual-semantic-embeddings","slug":"language-agnostic-visual-semantic-embeddings","title":"Language-Agnostic Visual-Semantic Embeddings","date":"2019-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/camp-cross-modal-adaptive-message-passing-for","slug":"camp-cross-modal-adaptive-message-passing-for","title":"CAMP: Cross-Modal Adaptive Message Passing for Text-Image Retrieval","date":"2019-09-12","arxiv_id":"1909.05506","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/camp-cross-modal-adaptive-message-passing-for#ran","syntology_url":"https://syntology.ai/paper/1909.05506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.05506"}},"official":{"repos":["ZihaoWang-CV/CAMP_iccv19"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/harmonized-multimodal-learning-with-gaussian","slug":"harmonized-multimodal-learning-with-gaussian","title":"Harmonized Multimodal Learning with Gaussian Process Latent Variable Models","date":"2019-08-14","arxiv_id":"1908.04979","repositories_listed":1,"syntology":null},{"url":"/paper/learning-visual-actions-using-multiple-verb","slug":"learning-visual-actions-using-multiple-verb","title":"Learning Visual Actions Using Multiple Verb-Only Labels","date":"2019-07-25","arxiv_id":"1907.11117","repositories_listed":1,"syntology":null},{"url":"/paper/polysemous-visual-semantic-embedding-for-1","slug":"polysemous-visual-semantic-embedding-for-1","title":"Polysemous Visual-Semantic Embedding for Cross-Modal Retrieval","date":"2019-06-11","arxiv_id":"1906.04402","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/polysemous-visual-semantic-embedding-for-1#ran","syntology_url":"https://syntology.ai/paper/1906.04402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.04402"}},"official":null}},{"url":"/paper/deep-supervised-cross-modal-retrieval","slug":"deep-supervised-cross-modal-retrieval","title":"Deep Supervised Cross-Modal Retrieval","date":"2019-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unified-visual-semantic-embeddings-bridging-1","slug":"unified-visual-semantic-embeddings-bridging-1","title":"Unified Visual-Semantic Embeddings: Bridging Vision and Language With Structured Meaning Representations","date":"2019-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/effective-and-efficient-indexing-in-cross","slug":"effective-and-efficient-indexing-in-cross","title":"Effective and Efficient Indexing in Cross-Modal Hashing-Based Datasets","date":"2019-04-30","arxiv_id":"1904.13325","repositories_listed":1,"syntology":null},{"url":"/paper/unified-visual-semantic-embeddings-bridging","slug":"unified-visual-semantic-embeddings-bridging","title":"UniVSE: Robust Visual Semantic Embeddings via Structured Semantic Representations","date":"2019-04-11","arxiv_id":"1904.05521","repositories_listed":1,"syntology":null},{"url":"/paper/context-aware-embeddings-for-automatic-art","slug":"context-aware-embeddings-for-automatic-art","title":"Context-Aware Embeddings for Automatic Art Analysis","date":"2019-04-10","arxiv_id":"1904.04985","repositories_listed":1,"syntology":null},{"url":"/paper/cmir-net-a-deep-learning-based-model-for","slug":"cmir-net-a-deep-learning-based-model-for","title":"CMIR-NET : A Deep Learning Based Model For Cross-Modal Retrieval In Remote Sensing","date":"2019-04-09","arxiv_id":"1904.04794","repositories_listed":1,"syntology":null},{"url":"/paper/show-translate-and-tell","slug":"show-translate-and-tell","title":"Show, Translate and Tell","date":"2019-03-14","arxiv_id":"1903.06275","repositories_listed":1,"syntology":null},{"url":"/paper/deep-cross-modal-projection-learning-for","slug":"deep-cross-modal-projection-learning-for","title":"Deep Cross-Modal Projection Learning for Image-Text Matching","date":"2018-09-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mtfh-a-matrix-tri-factorization-hashing","slug":"mtfh-a-matrix-tri-factorization-hashing","title":"MTFH: A Matrix Tri-Factorization Hashing Framework for Efficient Cross-Modal Retrieval","date":"2018-05-04","arxiv_id":"1805.01963","repositories_listed":1,"syntology":null},{"url":"/paper/learnable-pins-cross-modal-embeddings-for","slug":"learnable-pins-cross-modal-embeddings-for","title":"Learnable PINs: Cross-Modal Embeddings for Person Identity","date":"2018-05-02","arxiv_id":"1805.00833","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learnable-pins-cross-modal-embeddings-for#ran","syntology_url":"https://syntology.ai/paper/1805.00833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.00833"}},"official":null}},{"url":"/paper/cross-modal-retrieval-in-the-cooking-context","slug":"cross-modal-retrieval-in-the-cooking-context","title":"Cross-Modal Retrieval in the Cooking Context: Learning Semantic Text-Image Embeddings","date":"2018-04-30","arxiv_id":"1804.11146","repositories_listed":1,"syntology":null},{"url":"/paper/finding-beans-in-burgers-deep-semantic-visual","slug":"finding-beans-in-burgers-deep-semantic-visual","title":"Finding beans in burgers: Deep semantic-visual embedding with localization","date":"2018-04-05","arxiv_id":"1804.01720","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-adversarial-hashing-networks","slug":"self-supervised-adversarial-hashing-networks","title":"Self-Supervised Adversarial Hashing Networks for Cross-Modal Retrieval","date":"2018-04-04","arxiv_id":"1804.01223","repositories_listed":1,"syntology":null},{"url":"/paper/deep-binary-reconstruction-for-cross-modal","slug":"deep-binary-reconstruction-for-cross-modal","title":"Deep Binary Reconstruction for Cross-modal Hashing","date":"2017-08-17","arxiv_id":"1708.05127","repositories_listed":1,"syntology":null},{"url":"/paper/modality-specific-cross-modal-similarity","slug":"modality-specific-cross-modal-similarity","title":"Modality-specific Cross-modal Similarity Measurement with Recurrent Attention Network","date":"2017-08-16","arxiv_id":"1708.04776","repositories_listed":1,"syntology":null},{"url":"/paper/see-hear-and-read-deep-aligned","slug":"see-hear-and-read-deep-aligned","title":"See, Hear, and Read: Deep Aligned Representations","date":"2017-06-03","arxiv_id":"1706.00932","repositories_listed":1,"syntology":null},{"url":"/paper/multi-label-cross-modal-retrieval","slug":"multi-label-cross-modal-retrieval","title":"Multi-Label Cross-Modal Retrieval","date":"2015-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":null,"slug":"an-analysis-of-vision-language-models-for","title":"An analysis of vision-language models for fabric retrieval","date":"2025-07-07","arxiv_id":"2507.04735","repositories_listed":0,"syntology":null},{"url":null,"slug":"mask-aware-text-to-image-retrieval-referring","title":"Mask-aware Text-to-Image Retrieval: Referring Expression Segmentation Meets Cross-modal Retrieval","date":"2025-06-28","arxiv_id":"2506.22864","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximal-matching-matters-preventing","title":"Maximal Matching Matters: Preventing Representation Collapse for Robust Cross-Modal Retrieval","date":"2025-06-26","arxiv_id":"2506.21538","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-medical-image-binding-via-shared","title":"Multimodal Medical Image Binding via Shared Text Embeddings","date":"2025-06-22","arxiv_id":"2506.18072","repositories_listed":0,"syntology":null},{"url":null,"slug":"fednano-toward-lightweight-federated-tuning","title":"FedNano: Toward Lightweight Federated Tuning for Pretrained Multimodal Large Language Models","date":"2025-06-12","arxiv_id":"2506.14824","repositories_listed":0,"syntology":null},{"url":null,"slug":"sa-person-text-based-person-retrieval-with","title":"SA-Person: Text-Based Person Retrieval with Scene-aware Re-ranking","date":"2025-05-30","arxiv_id":"2505.24466","repositories_listed":0,"syntology":null},{"url":null,"slug":"emotionrankclap-bridging-natural-language","title":"EmotionRankCLAP: Bridging Natural Language Speaking Styles and Ordinal Speech Emotion via Rank-N-Contrast","date":"2025-05-29","arxiv_id":"2505.23732","repositories_listed":0,"syntology":null},{"url":null,"slug":"foliage-towards-physical-intelligence-world","title":"FOLIAGE: Towards Physical Intelligence World Models Via Unbounded Surface Evolution","date":"2025-05-29","arxiv_id":"2506.03173","repositories_listed":0,"syntology":null},{"url":null,"slug":"docmmir-a-framework-for-document-multi-modal","title":"DocMMIR: A Framework for Document Multi-modal Information Retrieval","date":"2025-05-25","arxiv_id":"2505.19312","repositories_listed":0,"syntology":null},{"url":null,"slug":"gmm-based-comprehensive-feature-extraction","title":"GMM-Based Comprehensive Feature Extraction and Relative Distance Preservation For Few-Shot Cross-Modal Retrieval","date":"2025-05-19","arxiv_id":"2505.13306","repositories_listed":0,"syntology":null},{"url":null,"slug":"sat2sound-a-unified-framework-for-zero-shot","title":"Sat2Sound: A Unified Framework for Zero-Shot Soundscape Mapping","date":"2025-05-19","arxiv_id":"2505.13777","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10921","title":"Towards Cross-modal Retrieval in Chinese Cultural Heritage Documents: Dataset and Solution","date":"2025-05-16","arxiv_id":"2505.10921","repositories_listed":0,"syntology":null},{"url":null,"slug":"cellclip-learning-perturbation-effects-in","title":"CellCLIP -- Learning Perturbation Effects in Cell Painting via Text-Guided Contrastive Learning","date":"2025-05-16","arxiv_id":"2506.06290","repositories_listed":0,"syntology":null},{"url":"/paper/omgm-orchestrate-multiple-granularities-and","slug":"omgm-orchestrate-multiple-granularities-and","title":"OMGM: Orchestrate Multiple Granularities and Modalities for Efficient Multimodal Retrieval","date":"2025-05-10","arxiv_id":"2505.07879","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/omgm-orchestrate-multiple-granularities-and#ran","syntology_url":"https://syntology.ai/paper/2505.07879","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07879"}},"official":null}},{"url":null,"slug":"improving-sound-source-localization-with","title":"Improving Sound Source Localization with Joint Slot Attention on Image and Audio","date":"2025-04-21","arxiv_id":"2504.15118","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-1st-erel-mir-workshop-on-efficient","title":"The 1st EReL@MIR Workshop on Efficient Representation Learning for Multimodal Information Retrieval","date":"2025-04-21","arxiv_id":"2504.14788","repositories_listed":0,"syntology":null},{"url":null,"slug":"semcore-a-semantic-enhanced-generative-cross","title":"SemCORE: A Semantic-Enhanced Generative Cross-Modal Retrieval Framework with MLLMs","date":"2025-04-17","arxiv_id":"2504.13172","repositories_listed":0,"syntology":null},{"url":null,"slug":"patfinger-prompt-adapted-transferable","title":"PATFinger: Prompt-Adapted Transferable Fingerprinting against Unauthorized Multimodal Dataset Usage","date":"2025-04-15","arxiv_id":"2504.11509","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-sparse-disentangled-representations","title":"Learning Sparse Disentangled Representations for Multimodal Exclusion Retrieval","date":"2025-04-04","arxiv_id":"2504.03184","repositories_listed":0,"syntology":null},{"url":null,"slug":"finelip-extending-clip-s-reach-via-fine","title":"FineLIP: Extending CLIP's Reach via Fine-Grained Alignment with Longer Text Inputs","date":"2025-04-02","arxiv_id":"2504.01916","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-speech-and-sound-distinguishing-and","title":"Seeing Speech and Sound: Distinguishing and Locating Audios in Visual Scenes","date":"2025-03-24","arxiv_id":"2503.18880","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompthash-affinity-prompted-collaborative","title":"PromptHash: Affinity-Prompted Collaborative Cross-Modal Learning for Adaptive Hashing Retrieval","date":"2025-03-20","arxiv_id":"2503.16064","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-inner-speech-text-alignment-for-llm","title":"Adaptive Inner Speech-Text Alignment for LLM-based Speech Translation","date":"2025-03-13","arxiv_id":"2503.10211","repositories_listed":0,"syntology":null},{"url":null,"slug":"astrea-a-moe-based-visual-understanding-model","title":"Astrea: A MOE-based Visual Understanding Model with Progressive Alignment","date":"2025-03-12","arxiv_id":"2503.09445","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-recipe-for-improving-remote-sensing-vlm","title":"A Recipe for Improving Remote Sensing VLM Zero Shot Generalization","date":"2025-03-10","arxiv_id":"2503.08722","repositories_listed":0,"syntology":null},{"url":null,"slug":"x2ct-clip-enable-multi-abnormality-detection","title":"X2CT-CLIP: Enable Multi-Abnormality Detection in Computed Tomography from Chest Radiography via Tri-Modal Contrastive Learning","date":"2025-03-04","arxiv_id":"2503.02162","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-contrastive-distilled-hashing-for","title":"Lightweight Contrastive Distilled Hashing for Online Cross-modal Retrieval","date":"2025-02-27","arxiv_id":"2502.19751","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-importance-of-text-preprocessing-for","title":"On the Importance of Text Preprocessing for Multimodal Representation Learning and Pathology Report Generation","date":"2025-02-26","arxiv_id":"2502.19285","repositories_listed":0,"syntology":null},{"url":null,"slug":"pathology-report-generation-and-multimodal","title":"Pathology Report Generation and Multimodal Representation Learning for Cutaneous Melanocytic Lesions","date":"2025-02-26","arxiv_id":"2502.19293","repositories_listed":0,"syntology":null},{"url":"/paper/class-enhancing-cross-modal-text-molecule","slug":"class-enhancing-cross-modal-text-molecule","title":"CLASS: Enhancing Cross-Modal Text-Molecule Retrieval Performance and Training Efficiency","date":"2025-02-17","arxiv_id":"2502.11633","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-interactive-text-to-image-retrieval","title":"Zero-Shot Interactive Text-to-Image Retrieval via Diffusion-Augmented Representations","date":"2025-01-26","arxiv_id":"2501.15379","repositories_listed":0,"syntology":null},{"url":null,"slug":"tsvc-tripartite-learning-with-semantic","title":"TSVC:Tripartite Learning with Semantic Variation Consistency for Robust Image-Text Retrieval","date":"2025-01-19","arxiv_id":"2501.10935","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-3d-representation-with-multi-view","title":"Cross-Modal 3D Representation with Multi-View Images and Point Clouds","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-dense-knowledge-alignment-into","title":"Incorporating Dense Knowledge Alignment into Unified Multimodal Representation Models","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-speech-and-sound-distinguishing-and-1","title":"Seeing Speech and Sound: Distinguishing and Locating Audio Sources in Visual Scenes","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"maybe-you-are-looking-for-croqs-cross-modal","title":"Maybe you are looking for CroQS: Cross-modal Query Suggestion for Text-to-Image Retrieval","date":"2024-12-18","arxiv_id":"2412.13834","repositories_listed":0,"syntology":null},{"url":null,"slug":"rebalanced-vision-language-retrieval","title":"Rebalanced Vision-Language Retrieval Considering Structure-Aware Distillation","date":"2024-12-14","arxiv_id":"2412.10761","repositories_listed":0,"syntology":null},{"url":null,"slug":"clip-ping-boosting-lightweight-vision","title":"CLIP-PING: Boosting Lightweight Vision-Language Models with Proximus Intrinsic Neighbors Guidance","date":"2024-12-05","arxiv_id":"2412.03871","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-and-interpretable-multimodal","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","date":"2024-12-03","arxiv_id":"2412.02104","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusing-physics-driven-strategies-and-cross","title":"Fusing Physics-Driven Strategies and Cross-Modal Adversarial Learning: Toward Multi-Domain Applications","date":"2024-11-30","arxiv_id":"2412.00341","repositories_listed":0,"syntology":null},{"url":null,"slug":"flex-clip-feature-level-generation-network","title":"FLEX-CLIP: Feature-Level GEneration Network Enhanced CLIP for X-shot Cross-modal Retrieval","date":"2024-11-26","arxiv_id":"2411.17454","repositories_listed":0,"syntology":null},{"url":null,"slug":"clips-an-enhanced-clip-framework-for-learning","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","date":"2024-11-25","arxiv_id":"2411.16828","repositories_listed":0,"syntology":null},{"url":null,"slug":"finecaption-compositional-image-captioning","title":"FINECAPTION: Compositional Image Captioning Focusing on Wherever You Want at Any Granularity","date":"2024-11-23","arxiv_id":"2411.15411","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-factuality-of-3d-brain-mri-report","title":"Improving Factuality of 3D Brain MRI Report Generation with Paired Image-domain Retrieval and Text-domain Augmentation","date":"2024-11-23","arxiv_id":"2411.15490","repositories_listed":0,"syntology":null},{"url":null,"slug":"everything-is-a-video-unifying-modalities","title":"Everything is a Video: Unifying Modalities through Next-Frame Prediction","date":"2024-11-15","arxiv_id":"2411.10503","repositories_listed":0,"syntology":null},{"url":"/paper/exploring-optimal-transport-based-multi","slug":"exploring-optimal-transport-based-multi","title":"Exploring Optimal Transport-Based Multi-Grained Alignments for Text-Molecule Retrieval","date":"2024-11-04","arxiv_id":"2411.11875","repositories_listed":0,"syntology":null},{"url":null,"slug":"mm-embed-universal-multimodal-retrieval-with","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","date":"2024-11-04","arxiv_id":"2411.02571","repositories_listed":0,"syntology":null},{"url":null,"slug":"test-time-adaptation-for-cross-modal","title":"Test-time Adaptation for Cross-modal Retrieval with Query Shift","date":"2024-10-21","arxiv_id":"2410.15624","repositories_listed":0,"syntology":null},{"url":null,"slug":"gleanvec-accelerating-vector-search-with","title":"GleanVec: Accelerating vector search with minimalist nonlinear dimensionality reduction","date":"2024-10-14","arxiv_id":"2410.22347","repositories_listed":0,"syntology":null},{"url":"/paper/mmcomposition-revisiting-the-compositionality","slug":"mmcomposition-revisiting-the-compositionality","title":"MMCOMPOSITION: Revisiting the Compositionality of Pre-trained Vision-Language Models","date":"2024-10-13","arxiv_id":"2410.09733","repositories_listed":0,"syntology":null},{"url":null,"slug":"csa-data-efficient-mapping-of-unimodal","title":"CSA: Data-efficient Mapping of Unimodal Features to Multimodal Features","date":"2024-10-10","arxiv_id":"2410.07610","repositories_listed":0,"syntology":null},{"url":null,"slug":"snap-and-diagnose-an-advanced-multimodal","title":"Snap and Diagnose: An Advanced Multimodal Retrieval System for Identifying Plant Diseases in the Wild","date":"2024-08-27","arxiv_id":"2408.14723","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-chemistry-foundation-models-to","title":"Leveraging Chemistry Foundation Models to Facilitate Structure Focused Retrieval Augmented Generation in Multi-Agent Workflows for Catalyst and Materials Design","date":"2024-08-21","arxiv_id":"2408.11793","repositories_listed":0,"syntology":null},{"url":null,"slug":"limitations-in-employing-natural-language","title":"Limitations in Employing Natural Language Supervision for Sensor-Based Human Activity Recognition -- And Ways to Overcome Them","date":"2024-08-21","arxiv_id":"2408.12023","repositories_listed":0,"syntology":null},{"url":null,"slug":"gqe-generalized-query-expansion-for-enhanced","title":"Bridging Information Asymmetry in Text-video Retrieval: A Data-centric Approach","date":"2024-08-14","arxiv_id":"2408.07249","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-masked-auto-encoders-based-self","title":"Contrastive masked auto-encoders based self-supervised hashing for 2D image and 3D point cloud cross-modal retrieval","date":"2024-08-11","arxiv_id":"2408.05711","repositories_listed":0,"syntology":null}],"record_sha256":"57b04c76a0e59901874d3e739c59d9a293c7a45c1e934314ff8ae8de8e552190","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}