{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/cross-modal-retrieval/papers/2","list_of":"/task/cross-modal-retrieval","task":"Cross-Modal Retrieval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":6,"rows_per_page":100,"rows":[101,200],"of":522,"counts":{"archive_papers_tagged":522,"with_a_code_link":244,"where_syntology_ran_a_sample":70,"not_listed_spam_title":0,"listed":522,"listed_where_code_ran":70,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":56,"every_run_a_failure_of_syntologys_instrument":14,"listed_with_a_run_with_no_instrument_failure":56,"listed_every_run_a_failure_of_syntologys_instrument":14,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/cross-modal-retrieval","prev":"/task/cross-modal-retrieval","next":"/task/cross-modal-retrieval/papers/3","papers":[{"url":"/paper/tf-clip-learning-text-free-clip-for-video","slug":"tf-clip-learning-text-free-clip-for-video","title":"TF-CLIP: Learning Text-free CLIP for Video-based Person Re-Identification","date":"2023-12-15","arxiv_id":"2312.09627","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/tf-clip-learning-text-free-clip-for-video#ran","syntology_url":"https://syntology.ai/paper/2312.09627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.09627"}},"official":{"repos":["asuradayuci/tf-clip"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/car-consolidation-augmentation-and-regulation","slug":"car-consolidation-augmentation-and-regulation","title":"Enhancing Recipe Retrieval with Foundation Models: A Data Augmentation Perspective","date":"2023-12-08","arxiv_id":"2312.04763","repositories_listed":1,"syntology":null},{"url":"/paper/removing-nsfw-concepts-from-vision-and","slug":"removing-nsfw-concepts-from-vision-and","title":"Safe-CLIP: Removing NSFW Concepts from Vision-and-Language Models","date":"2023-11-27","arxiv_id":"2311.16254","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/removing-nsfw-concepts-from-vision-and#ran","syntology_url":"https://syntology.ai/paper/2311.16254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.16254"}},"official":{"repos":["aimagelab/safe-clip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ai-generated-images-introduce-invisible","slug":"ai-generated-images-introduce-invisible","title":"Invisible Relevance Bias: Text-Image Retrieval Models Prefer AI-Generated Images","date":"2023-11-23","arxiv_id":"2311.14084","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-transformer-learning-with","slug":"contrastive-transformer-learning-with","title":"Contrastive Transformer Learning with Proximity Data Generation for Text-Based Person Search","date":"2023-11-15","arxiv_id":"2311.09084","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contrastive-transformer-learning-with#ran","syntology_url":"https://syntology.ai/paper/2311.09084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09084"}},"official":{"repos":["hcplab-sysu/personsearch-ctlg"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/weakly-supervised-cross-model-learning-in","slug":"weakly-supervised-cross-model-learning-in","title":"Weakly supervised cross-modal learning in high-content screening","date":"2023-11-08","arxiv_id":"2311.04678","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/weakly-supervised-cross-model-learning-in#ran","syntology_url":"https://syntology.ai/paper/2311.04678","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04678"}},"official":{"repos":["gwatkinson/jump_download"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/birdsat-cross-view-contrastive-masked","slug":"birdsat-cross-view-contrastive-masked","title":"BirdSAT: Cross-View Contrastive Masked Autoencoders for Bird Species Classification and Mapping","date":"2023-10-29","arxiv_id":"2310.19168","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/birdsat-cross-view-contrastive-masked#ran","syntology_url":"https://syntology.ai/paper/2310.19168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19168"}},"official":{"repos":["mvrl/birdsat"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/a-prior-instruction-representation-framework","slug":"a-prior-instruction-representation-framework","title":"A Prior Instruction Representation Framework for Remote Sensing Image-text Retrieval","date":"2023-10-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/invgc-robust-cross-modal-retrieval-by-inverse","slug":"invgc-robust-cross-modal-retrieval-by-inverse","title":"InvGC: Robust Cross-Modal Retrieval by Inverse Graph Convolution","date":"2023-10-20","arxiv_id":"2310.13276","repositories_listed":1,"syntology":null},{"url":"/paper/balance-act-mitigating-hubness-in-cross-modal","slug":"balance-act-mitigating-hubness-in-cross-modal","title":"Balance Act: Mitigating Hubness in Cross-Modal Retrieval with Query and Gallery Banks","date":"2023-10-17","arxiv_id":"2310.11612","repositories_listed":1,"syntology":null},{"url":"/paper/pali-3-vision-language-models-smaller-faster","slug":"pali-3-vision-language-models-smaller-faster","title":"PaLI-3 Vision Language Models: Smaller, Faster, Stronger","date":"2023-10-13","arxiv_id":"2310.09199","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pali-3-vision-language-models-smaller-faster#ran","syntology_url":"https://syntology.ai/paper/2310.09199","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09199"}},"official":null}},{"url":"/paper/biobridge-bridging-biomedical-foundation","slug":"biobridge-bridging-biomedical-foundation","title":"BioBridge: Bridging Biomedical Foundation Models via Knowledge Graphs","date":"2023-10-05","arxiv_id":"2310.03320","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/biobridge-bridging-biomedical-foundation#ran","syntology_url":"https://syntology.ai/paper/2310.03320","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03320"}},"official":{"repos":["ryanwangzf/biobridge"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/prototype-based-aleatoric-uncertainty-1","slug":"prototype-based-aleatoric-uncertainty-1","title":"Prototype-based Aleatoric Uncertainty Quantification for Cross-modal Retrieval","date":"2023-09-29","arxiv_id":"2309.17093","repositories_listed":1,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":10,"n_instrument":7,"n_unverified":2,"n_honours":2,"n_violates":1,"n_no_contract":7,"n_pointer_only":8,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 1 violated, 7 with no contract checked; 7 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/prototype-based-aleatoric-uncertainty-1#ran","syntology_url":"https://syntology.ai/paper/2309.17093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17093"}},"official":{"repos":["leolee99/pau"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/align-before-search-aligning-ads-image-to","slug":"align-before-search-aligning-ads-image-to","title":"Align before Search: Aligning Ads Image to Text for Accurate Cross-Modal Sponsored Search","date":"2023-09-28","arxiv_id":"2309.16141","repositories_listed":1,"syntology":null},{"url":"/paper/elip-efficient-language-image-pre-training","slug":"elip-efficient-language-image-pre-training","title":"ELIP: Efficient Language-Image Pre-training with Fewer Vision Tokens","date":"2023-09-28","arxiv_id":"2309.16738","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-differentiable-outlier-detection","slug":"implicit-differentiable-outlier-detection","title":"Implicit Differentiable Outlier Detection Enable Robust Deep Multimodal Analysis","date":"2023-09-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-tri-modal-embeddings-for-zero-shot","slug":"learning-tri-modal-embeddings-for-zero-shot","title":"Learning Tri-modal Embeddings for Zero-Shot Soundscape Mapping","date":"2023-09-19","arxiv_id":"2309.10667","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/learning-tri-modal-embeddings-for-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2309.10667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.10667"}},"official":{"repos":["mvrl/geoclap"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/a-survey-on-interpretable-cross-modal","slug":"a-survey-on-interpretable-cross-modal","title":"A Survey on Interpretable Cross-modal Reasoning","date":"2023-09-05","arxiv_id":"2309.01955","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-foundation-models-for","slug":"multimodal-foundation-models-for","title":"Multimodal Foundation Models For Echocardiogram Interpretation","date":"2023-08-29","arxiv_id":"2308.15670","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-retrieval-a-systematic-review-of","slug":"cross-modal-retrieval-a-systematic-review-of","title":"Cross-Modal Retrieval: A Systematic Review of Methods and Future Directions","date":"2023-08-28","arxiv_id":"2308.14263","repositories_listed":1,"syntology":null},{"url":"/paper/parameter-efficient-transfer-learning-for-1","slug":"parameter-efficient-transfer-learning-for-1","title":"Parameter-Efficient Transfer Learning for Remote Sensing Image-Text Retrieval","date":"2023-08-24","arxiv_id":"2308.12509","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/parameter-efficient-transfer-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2308.12509","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12509"}},"official":{"repos":["ZhanYang-nwpu/PE-RSITR"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/an-empirical-study-of-clip-for-text-based","slug":"an-empirical-study-of-clip-for-text-based","title":"An Empirical Study of CLIP for Text-based Person Search","date":"2023-08-19","arxiv_id":"2308.10045","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-two-stream-encoders-with","slug":"unifying-two-stream-encoders-with","title":"Unifying Two-Stream Encoders with Transformers for Cross-Modal Retrieval","date":"2023-08-08","arxiv_id":"2308.04343","repositories_listed":1,"syntology":null},{"url":"/paper/clip-kd-an-empirical-study-of-distilling-clip","slug":"clip-kd-an-empirical-study-of-distilling-clip","title":"CLIP-KD: An Empirical Study of CLIP Model Distillation","date":"2023-07-24","arxiv_id":"2307.12732","repositories_listed":1,"syntology":{"n":18,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":9,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":18,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/clip-kd-an-empirical-study-of-distilling-clip#ran","syntology_url":"https://syntology.ai/paper/2307.12732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12732"}},"official":{"repos":["winycg/clip-kd"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/mclip-multilingual-clip-via-cross-lingual","slug":"mclip-multilingual-clip-via-cross-lingual","title":"mCLIP: Multilingual CLIP via Cross-lingual Transfer","date":"2023-07-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/alternative-telescopic-displacement-an","slug":"alternative-telescopic-displacement-an","title":"Alternative Telescopic Displacement: An Efficient Multimodal Alignment Method","date":"2023-06-29","arxiv_id":"2306.16950","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-transformers-for-infrared-and","slug":"cross-modal-transformers-for-infrared-and","title":"Cross-modal transformers for infrared and visible image fusion","date":"2023-06-26","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rs5m-a-large-scale-vision-language-dataset","slug":"rs5m-a-large-scale-vision-language-dataset","title":"RS5M and GeoRSCLIP: A Large Scale Vision-Language Dataset and A Large Vision-Language Model for Remote Sensing","date":"2023-06-20","arxiv_id":"2306.11300","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rs5m-a-large-scale-vision-language-dataset#ran","syntology_url":"https://syntology.ai/paper/2306.11300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11300"}},"official":{"repos":["om-ai-lab/rs5m"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/remoteclip-a-vision-language-foundation-model","slug":"remoteclip-a-vision-language-foundation-model","title":"RemoteCLIP: A Vision Language Foundation Model for Remote Sensing","date":"2023-06-19","arxiv_id":"2306.11029","repositories_listed":1,"syntology":null},{"url":"/paper/youku-mplug-a-10-million-large-scale-chinese","slug":"youku-mplug-a-10-million-large-scale-chinese","title":"Youku-mPLUG: A 10 Million Large-scale Chinese Video-Language Dataset for Pre-training and Benchmarks","date":"2023-06-07","arxiv_id":"2306.04362","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/youku-mplug-a-10-million-large-scale-chinese#ran","syntology_url":"https://syntology.ai/paper/2306.04362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.04362"}},"official":{"repos":["x-plug/youku-mplug"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-knowledge-retrieval-with-multi","slug":"end-to-end-knowledge-retrieval-with-multi","title":"End-to-end Knowledge Retrieval with Multi-modal Queries","date":"2023-06-01","arxiv_id":"2306.00424","repositories_listed":1,"syntology":null},{"url":"/paper/dense-and-aligned-captions-dac-promote","slug":"dense-and-aligned-captions-dac-promote","title":"Dense and Aligned Captions (DAC) Promote Compositional Reasoning in VL Models","date":"2023-05-31","arxiv_id":"2305.19595","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-retrieval-for-motion-and-text-via","slug":"cross-modal-retrieval-for-motion-and-text-via","title":"Cross-Modal Retrieval for Motion and Text via DopTriple Loss","date":"2023-05-07","arxiv_id":"2305.04195","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-benchmarks-for-cross-modal-image","slug":"rethinking-benchmarks-for-cross-modal-image","title":"Rethinking Benchmarks for Cross-modal Image-text Retrieval","date":"2023-04-21","arxiv_id":"2304.10824","repositories_listed":1,"syntology":null},{"url":"/paper/rococo-robust-benchmark-ms-coco-to-stress","slug":"rococo-robust-benchmark-ms-coco-to-stress","title":"RoCOCO: Robustness Benchmark of MS-COCO to Stress-test Image-Text Matching Models","date":"2023-04-21","arxiv_id":"2304.10727","repositories_listed":1,"syntology":null},{"url":"/paper/image-text-retrieval-via-preserving-main","slug":"image-text-retrieval-via-preserving-main","title":"Image-text Retrieval via Preserving Main Semantics of Vision","date":"2023-04-20","arxiv_id":"2304.10254","repositories_listed":1,"syntology":null},{"url":"/paper/valor-vision-audio-language-omni-perception","slug":"valor-vision-audio-language-omni-perception","title":"VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and Dataset","date":"2023-04-17","arxiv_id":"2304.08345","repositories_listed":1,"syntology":null},{"url":"/paper/noisy-correspondence-learning-with-meta","slug":"noisy-correspondence-learning-with-meta","title":"Noisy Correspondence Learning with Meta Similarity Correction","date":"2023-04-13","arxiv_id":"2304.06275","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/noisy-correspondence-learning-with-meta#ran","syntology_url":"https://syntology.ai/paper/2304.06275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.06275"}},"official":{"repos":["hhc1997/mscn"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mammut-a-simple-architecture-for-joint","slug":"mammut-a-simple-architecture-for-joint","title":"MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks","date":"2023-03-29","arxiv_id":"2303.16839","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mammut-a-simple-architecture-for-joint#ran","syntology_url":"https://syntology.ai/paper/2303.16839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16839"}},"official":null}},{"url":"/paper/plug-and-play-regulators-for-image-text","slug":"plug-and-play-regulators-for-image-text","title":"Plug-and-Play Regulators for Image-Text Matching","date":"2023-03-23","arxiv_id":"2303.13371","repositories_listed":1,"syntology":null},{"url":"/paper/single-branch-network-for-multimodal-training","slug":"single-branch-network-for-multimodal-training","title":"Single-branch Network for Multimodal Training","date":"2023-03-10","arxiv_id":"2303.06129","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-modality-alignment-network-for","slug":"adversarial-modality-alignment-network-for","title":"Adversarial Modality Alignment Network for Cross-Modal Molecule Retrieval","date":"2023-03-08","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fame-vil-multi-tasking-vision-language-model","slug":"fame-vil-multi-tasking-vision-language-model","title":"FAME-ViL: Multi-Tasking Vision-Language Model for Heterogeneous Fashion Tasks","date":"2023-03-04","arxiv_id":"2303.02483","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-retrieval-with-partially","slug":"cross-modal-retrieval-with-partially","title":"Cross-Modal Retrieval with Partially Mismatched Pairs","date":"2023-02-22","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/scene-centric-vs-object-centric-image-text","slug":"scene-centric-vs-object-centric-image-text","title":"Scene-centric vs. Object-centric Image-Text Cross-modal Retrieval: A Reproducibility Study","date":"2023-01-12","arxiv_id":"2301.05174","repositories_listed":1,"syntology":null},{"url":"/paper/toward-building-general-foundation-models-for","slug":"toward-building-general-foundation-models-for","title":"Toward Building General Foundation Models for Language, Vision, and Vision-Language Understanding Tasks","date":"2023-01-12","arxiv_id":"2301.05065","repositories_listed":1,"syntology":null},{"url":"/paper/napreg-nouns-as-proxies-regularization-for","slug":"napreg-nouns-as-proxies-regularization-for","title":"NAPReg: Nouns As Proxies Regularization for Semantically Aware Cross-Modal Embeddings","date":"2023-01-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-semantic-relationship-among","slug":"learning-semantic-relationship-among","title":"Learning Semantic Relationship Among Instances for Image-Text Matching","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rono-robust-discriminative-learning-with","slug":"rono-robust-discriminative-learning-with","title":"RONO: Robust Discriminative Learning With Noisy Labels for 2D-3D Cross-Modal Retrieval","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/position-guided-text-prompt-for-vision","slug":"position-guided-text-prompt-for-vision","title":"Position-guided Text Prompt for Vision-Language Pre-training","date":"2022-12-19","arxiv_id":"2212.09737","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/position-guided-text-prompt-for-vision#ran","syntology_url":"https://syntology.ai/paper/2212.09737","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09737"}},"official":{"repos":["sail-sg/ptp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/structured-vision-language-pretraining-for","slug":"structured-vision-language-pretraining-for","title":"Vision and Structured-Language Pretraining for Cross-Modal Food Retrieval","date":"2022-12-08","arxiv_id":"2212.04267","repositories_listed":1,"syntology":null},{"url":"/paper/improving-cross-modal-retrieval-with-set-of","slug":"improving-cross-modal-retrieval-with-set-of","title":"Improving Cross-Modal Retrieval with Set of Diverse Embeddings","date":"2022-11-30","arxiv_id":"2211.16761","repositories_listed":1,"syntology":null},{"url":"/paper/normalized-contrastive-learning-for-text","slug":"normalized-contrastive-learning-for-text","title":"Normalized Contrastive Learning for Text-Video Retrieval","date":"2022-11-30","arxiv_id":"2212.11790","repositories_listed":1,"syntology":null},{"url":"/paper/vop-text-video-co-operative-prompt-tuning-for","slug":"vop-text-video-co-operative-prompt-tuning-for","title":"VoP: Text-Video Co-operative Prompt Tuning for Cross-Modal Retrieval","date":"2022-11-23","arxiv_id":"2211.12764","repositories_listed":1,"syntology":null},{"url":"/paper/perceiver-vl-efficient-vision-and-language","slug":"perceiver-vl-efficient-vision-and-language","title":"Perceiver-VL: Efficient Vision-and-Language Modeling with Iterative Latent Attention","date":"2022-11-21","arxiv_id":"2211.11701","repositories_listed":1,"syntology":null},{"url":"/paper/posescript-3d-human-poses-from-natural","slug":"posescript-3d-human-poses-from-natural","title":"PoseScript: Linking 3D Human Poses and Natural Language","date":"2022-10-21","arxiv_id":"2210.11795","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-fusion-distillation-for-fine","slug":"cross-modal-fusion-distillation-for-fine","title":"Cross-Modal Fusion Distillation for Fine-Grained Sketch-Based Image Retrieval","date":"2022-10-19","arxiv_id":"2210.10486","repositories_listed":1,"syntology":null},{"url":"/paper/deep-evidential-learning-with-noisy","slug":"deep-evidential-learning-with-noisy","title":"Deep Evidential Learning with Noisy Correspondence for Cross-Modal Retrieval","date":"2022-10-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ernie-vil-2-0-multi-view-contrastive-learning","slug":"ernie-vil-2-0-multi-view-contrastive-learning","title":"ERNIE-ViL 2.0: Multi-view Contrastive Learning for Image-Text Pre-training","date":"2022-09-30","arxiv_id":"2209.15270","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-evaluate-performance-of-multi","slug":"learning-to-evaluate-performance-of-multi","title":"Learning to Evaluate Performance of Multi-modal Semantic Localization","date":"2022-09-14","arxiv_id":"2209.06515","repositories_listed":1,"syntology":null},{"url":"/paper/cross-lingual-cross-modal-retrieval-with","slug":"cross-lingual-cross-modal-retrieval-with","title":"Cross-Lingual Cross-Modal Retrieval with Noise-Robust Learning","date":"2022-08-26","arxiv_id":"2208.12526","repositories_listed":1,"syntology":null},{"url":"/paper/mulan-a-joint-embedding-of-music-audio-and","slug":"mulan-a-joint-embedding-of-music-audio-and","title":"MuLan: A Joint Embedding of Music Audio and Natural Language","date":"2022-08-26","arxiv_id":"2208.12415","repositories_listed":1,"syntology":null},{"url":"/paper/learning-modal-invariant-and-temporal-memory-1","slug":"learning-modal-invariant-and-temporal-memory-1","title":"Learning Modal-Invariant and Temporal-Memory for Video-based Visible-Infrared Person Re-Identification","date":"2022-08-04","arxiv_id":"2208.02450","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/learning-modal-invariant-and-temporal-memory-1#ran","syntology_url":"https://syntology.ai/paper/2208.02450","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.02450"}},"official":{"repos":["vcm-project233/mitml"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"url":"/paper/aladin-distilling-fine-grained-alignment","slug":"aladin-distilling-fine-grained-alignment","title":"ALADIN: Distilling Fine-grained Alignment Scores for Efficient Image-Text Matching and Retrieval","date":"2022-07-29","arxiv_id":"2207.14757","repositories_listed":1,"syntology":null},{"url":"/paper/intra-modal-constraint-loss-for-image-text","slug":"intra-modal-constraint-loss-for-image-text","title":"Intra-Modal Constraint Loss For Image-Text Retrieval","date":"2022-07-11","arxiv_id":"2207.05024","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-multi-label-contrastive-learning","slug":"integrating-multi-label-contrastive-learning","title":"Integrating multi-label contrastive learning with dual adversarial graph neural networks for cross-modal retrieval","date":"2022-07-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/comprehending-and-ordering-semantics-for-1","slug":"comprehending-and-ordering-semantics-for-1","title":"Comprehending and Ordering Semantics for Image Captioning","date":"2022-06-14","arxiv_id":"2206.06930","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-a-fine-grained-multiscale-method","slug":"exploring-a-fine-grained-multiscale-method","title":"Exploring a Fine-Grained Multiscale Method for Cross-Modal Remote Sensing Image Retrieval","date":"2022-04-21","arxiv_id":"2204.09868","repositories_listed":1,"syntology":null},{"url":"/paper/remote-sensing-cross-modal-text-image","slug":"remote-sensing-cross-modal-text-image","title":"Remote Sensing Cross-Modal Text-Image Retrieval Based on Global and Local Information","date":"2022-04-21","arxiv_id":"2204.09860","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/remote-sensing-cross-modal-text-image#ran","syntology_url":"https://syntology.ai/paper/2204.09860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.09860"}},"official":{"repos":["xiaoyuan1996/galr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/transformer-decoders-with-multimodal","slug":"transformer-decoders-with-multimodal","title":"Transformer Decoders with MultiModal Regularization for Cross-Modal Food Retrieval","date":"2022-04-20","arxiv_id":"2204.09730","repositories_listed":1,"syntology":null},{"url":"/paper/on-metric-learning-for-audio-text-cross-modal","slug":"on-metric-learning-for-audio-text-cross-modal","title":"On Metric Learning for Audio-Text Cross-Modal Retrieval","date":"2022-03-29","arxiv_id":"2203.15537","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-pre-training-with-triple","slug":"vision-language-pre-training-with-triple","title":"Vision-Language Pre-Training with Triple Contrastive Learning","date":"2022-02-21","arxiv_id":"2202.10401","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vision-language-pre-training-with-triple#ran","syntology_url":"https://syntology.ai/paper/2202.10401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10401"}},"official":{"repos":["uta-smile/TCL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-cross-modal-retrieval-via-deep","slug":"efficient-cross-modal-retrieval-via-deep","title":"Efficient Cross-Modal Retrieval via Deep Binary Hashing and Quantization","date":"2022-02-15","arxiv_id":"2202.10232","repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-empirical-study-of-vision","slug":"a-comprehensive-empirical-study-of-vision","title":"A Comprehensive Empirical Study of Vision-Language Pre-trained Model for Supervised Cross-Modal Retrieval","date":"2022-01-08","arxiv_id":"2201.02772","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-retrieval-with-querybank","slug":"cross-modal-retrieval-with-querybank","title":"Cross Modal Retrieval with Querybank Normalisation","date":"2021-12-23","arxiv_id":"2112.12777","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cross-modal-retrieval-with-querybank#ran","syntology_url":"https://syntology.ai/paper/2112.12777","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.12777"}},"official":{"repos":["ioanacroi/qb-norm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-with-noisy-correspondence-for-cross","slug":"learning-with-noisy-correspondence-for-cross","title":"Learning with Noisy Correspondence for Cross-modal Matching","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/emotion-embedding-spaces-for-matching-music","slug":"emotion-embedding-spaces-for-matching-music","title":"Emotion Embedding Spaces for Matching Music to Stories","date":"2021-11-26","arxiv_id":"2111.13468","repositories_listed":1,"syntology":null},{"url":"/paper/multi-grained-vision-language-pre-training","slug":"multi-grained-vision-language-pre-training","title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","date":"2021-11-16","arxiv_id":"2111.08276","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-grained-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2111.08276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.08276"}},"official":{"repos":["zengyan-97/x-vlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-curious-layperson-fine-grained-image","slug":"the-curious-layperson-fine-grained-image","title":"The Curious Layperson: Fine-Grained Image Recognition without Expert Labels","date":"2021-11-05","arxiv_id":"2111.03651","repositories_listed":1,"syntology":null},{"url":"/paper/text2mol-cross-modal-molecule-retrieval-with","slug":"text2mol-cross-modal-molecule-retrieval-with","title":"Text2Mol: Cross-Modal Molecule Retrieval with Natural Language Queries","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-text-image-joint-embedding-for","slug":"learning-text-image-joint-embedding-for","title":"Learning Text-Image Joint Embedding for Efficient Cross-Modal Retrieval with Deep Feature Engineering","date":"2021-10-22","arxiv_id":"2110.11592","repositories_listed":1,"syntology":null},{"url":"/paper/wav2clip-learning-robust-audio","slug":"wav2clip-learning-robust-audio","title":"Wav2CLIP: Learning Robust Audio Representations From CLIP","date":"2021-10-21","arxiv_id":"2110.11499","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wav2clip-learning-robust-audio#ran","syntology_url":"https://syntology.ai/paper/2110.11499","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.11499"}},"official":{"repos":["descriptinc/lyrebird-wav2clip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/text-based-person-search-with-limited-data","slug":"text-based-person-search-with-limited-data","title":"Text-Based Person Search with Limited Data","date":"2021-10-20","arxiv_id":"2110.10807","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-label-aware-graph-convolutional","slug":"adaptive-label-aware-graph-convolutional","title":"Adaptive label-aware graph convolutional networks for cross-modal retrieval","date":"2021-08-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-tfidf-enhanced-joint-embedding-for","slug":"learning-tfidf-enhanced-joint-embedding-for","title":"Learning TFIDF Enhanced Joint Embedding for Recipe-Image Cross-Modal Retrieval Service","date":"2021-08-02","arxiv_id":"2108.00724","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-audiovisual-representation","slug":"self-supervised-audiovisual-representation","title":"Self-supervised Audiovisual Representation Learning for Remote Sensing Data","date":"2021-08-02","arxiv_id":"2108.00688","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-modality-interaction-modeling-for","slug":"dynamic-modality-interaction-modeling-for","title":"Dynamic Modality Interaction Modeling for Image-Text Retrieval","date":"2021-07-11","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/evaluation-of-audio-visual-alignments-in","slug":"evaluation-of-audio-visual-alignments-in","title":"Evaluation of Audio-Visual Alignments in Visually Grounded Speech Models","date":"2021-07-05","arxiv_id":"2108.02562","repositories_listed":1,"syntology":null},{"url":"/paper/fedcmr-federated-cross-modal-retrieval","slug":"fedcmr-federated-cross-modal-retrieval","title":"FedCMR: Federated Cross-Modal Retrieval","date":"2021-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/domain-smoothing-network-for-zero-shot-sketch","slug":"domain-smoothing-network-for-zero-shot-sketch","title":"Domain-Smoothing Network for Zero-Shot Sketch-Based Image Retrieval","date":"2021-06-22","arxiv_id":"2106.11841","repositories_listed":1,"syntology":null},{"url":"/paper/learning-cross-modal-retrieval-with-noisy","slug":"learning-cross-modal-retrieval-with-noisy","title":"Learning Cross-Modal Retrieval With Noisy Labels","date":"2021-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/exploring-modality-agnostic-representations","slug":"exploring-modality-agnostic-representations","title":"Exploring modality-agnostic representations for music classification","date":"2021-06-02","arxiv_id":"2106.01149","repositories_listed":1,"syntology":null},{"url":"/paper/learning-relation-alignment-for-calibrated","slug":"learning-relation-alignment-for-calibrated","title":"Learning Relation Alignment for Calibrated Cross-modal Retrieval","date":"2021-05-28","arxiv_id":"2105.13868","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-relation-alignment-for-calibrated#ran","syntology_url":"https://syntology.ai/paper/2105.13868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.13868"}},"official":{"repos":["lancopku/IAIS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-adversarial-graph-neural-networks-for","slug":"dual-adversarial-graph-neural-networks-for","title":"Dual adversarial graph neural networks for multi-label cross-modal retrieval","date":"2021-05-18","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fddh-fast-discriminative-discrete-hashing-for","slug":"fddh-fast-discriminative-discrete-hashing-for","title":"FDDH: Fast Discriminative Discrete Hashing for Large-Scale Cross-Modal Retrieval","date":"2021-05-15","arxiv_id":"2105.07128","repositories_listed":1,"syntology":null},{"url":"/paper/muscaps-generating-captions-for-music-audio","slug":"muscaps-generating-captions-for-music-audio","title":"MusCaps: Generating Captions for Music Audio","date":"2021-04-24","arxiv_id":"2104.11984","repositories_listed":1,"syntology":null},{"url":"/paper/more-photos-are-all-you-need-semi-supervised","slug":"more-photos-are-all-you-need-semi-supervised","title":"More Photos are All You Need: Semi-Supervised Learning for Fine-Grained Sketch Based Image Retrieval","date":"2021-03-25","arxiv_id":"2103.13990","repositories_listed":1,"syntology":null},{"url":"/paper/revamping-cross-modal-recipe-retrieval-with","slug":"revamping-cross-modal-recipe-retrieval-with","title":"Revamping Cross-Modal Recipe Retrieval with Hierarchical Transformers and Self-supervised Learning","date":"2021-03-24","arxiv_id":"2103.13061","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/revamping-cross-modal-recipe-retrieval-with#ran","syntology_url":"https://syntology.ai/paper/2103.13061","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.13061"}},"official":{"repos":["amzn/image-to-recipe-transformers"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/retrieve-fast-rerank-smart-cooperative-and","slug":"retrieve-fast-rerank-smart-cooperative-and","title":"Retrieve Fast, Rerank Smart: Cooperative and Joint Approaches for Improved Cross-Modal Retrieval","date":"2021-03-22","arxiv_id":"2103.11920","repositories_listed":1,"syntology":null},{"url":"/paper/part2whole-iteratively-enrich-detail-for","slug":"part2whole-iteratively-enrich-detail-for","title":"Ask&Confirm: Active Detail Enriching for Cross-Modal Retrieval with Partial Query","date":"2021-03-02","arxiv_id":"2103.01654","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/part2whole-iteratively-enrich-detail-for#ran","syntology_url":"https://syntology.ai/paper/2103.01654","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.01654"}},"official":{"repos":["cuthbertcai/ask-confirm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"da99434571acb77fc5b6a7d1fe517d44bfc8f942a37516dd4500b205c168e367","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}