{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-retrieval/papers/3","list_of":"/task/text-retrieval","task":"Text Retrieval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":7,"rows_per_page":100,"rows":[201,300],"of":671,"counts":{"archive_papers_tagged":671,"with_a_code_link":335,"where_syntology_ran_a_sample":117,"not_listed_spam_title":0,"listed":671,"listed_where_code_ran":117,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":99,"every_run_a_failure_of_syntologys_instrument":18,"listed_with_a_run_with_no_instrument_failure":99,"listed_every_run_a_failure_of_syntologys_instrument":18,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-retrieval","prev":"/task/text-retrieval/papers/2","next":"/task/text-retrieval/papers/4","papers":[{"url":"/paper/rs5m-a-large-scale-vision-language-dataset","slug":"rs5m-a-large-scale-vision-language-dataset","title":"RS5M and GeoRSCLIP: A Large Scale Vision-Language Dataset and A Large Vision-Language Model for Remote Sensing","date":"2023-06-20","arxiv_id":"2306.11300","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rs5m-a-large-scale-vision-language-dataset#ran","syntology_url":"https://syntology.ai/paper/2306.11300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11300"}},"official":{"repos":["om-ai-lab/rs5m"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/remoteclip-a-vision-language-foundation-model","slug":"remoteclip-a-vision-language-foundation-model","title":"RemoteCLIP: A Vision Language Foundation Model for Remote Sensing","date":"2023-06-19","arxiv_id":"2306.11029","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-token-guided-image-text-retrieval","slug":"efficient-token-guided-image-text-retrieval","title":"Efficient Token-Guided Image-Text Retrieval with Consistent Multimodal Contrastive Training","date":"2023-06-15","arxiv_id":"2306.08789","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/efficient-token-guided-image-text-retrieval#ran","syntology_url":"https://syntology.ai/paper/2306.08789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08789"}},"official":null}},{"url":"/paper/babel-imagenet-massively-multilingual","slug":"babel-imagenet-massively-multilingual","title":"Babel-ImageNet: Massively Multilingual Evaluation of Vision-and-Language Representations","date":"2023-06-14","arxiv_id":"2306.08658","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/babel-imagenet-massively-multilingual#ran","syntology_url":"https://syntology.ai/paper/2306.08658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08658"}},"official":{"repos":["gregor-ge/babel-imagenet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/global-and-local-semantic-completion-learning","slug":"global-and-local-semantic-completion-learning","title":"Global and Local Semantic Completion Learning for Vision-Language Pre-training","date":"2023-06-12","arxiv_id":"2306.07096","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-pre-training-for-medical-vision","slug":"multi-modal-pre-training-for-medical-vision","title":"Multi-modal Pre-training for Medical Vision-language Understanding and Generation: An Empirical Study with A New Benchmark","date":"2023-06-10","arxiv_id":"2306.06494","repositories_listed":1,"syntology":null},{"url":"/paper/visualgptscore-visio-linguistic-reasoning","slug":"visualgptscore-visio-linguistic-reasoning","title":"Revisiting the Role of Language Priors in Vision-Language Models","date":"2023-06-02","arxiv_id":"2306.01879","repositories_listed":1,"syntology":null},{"url":"/paper/test-time-adaptation-with-clip-reward-for","slug":"test-time-adaptation-with-clip-reward-for","title":"Test-Time Adaptation with CLIP Reward for Zero-Shot Generalization in Vision-Language Models","date":"2023-05-29","arxiv_id":"2305.18010","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/test-time-adaptation-with-clip-reward-for#ran","syntology_url":"https://syntology.ai/paper/2305.18010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18010"}},"official":{"repos":["mzhaoshuai/rlcf"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/fusecap-leveraging-large-language-models-to","slug":"fusecap-leveraging-large-language-models-to","title":"FuseCap: Leveraging Large Language Models for Enriched Fused Image Captions","date":"2023-05-28","arxiv_id":"2305.17718","repositories_listed":1,"syntology":null},{"url":"/paper/crossget-cross-guided-ensemble-of-tokens-for","slug":"crossget-cross-guided-ensemble-of-tokens-for","title":"CrossGET: Cross-Guided Ensemble of Tokens for Accelerating Vision-Language Transformers","date":"2023-05-27","arxiv_id":"2305.17455","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crossget-cross-guided-ensemble-of-tokens-for#ran","syntology_url":"https://syntology.ai/paper/2305.17455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17455"}},"official":{"repos":["sdc17/crossget"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/integrating-listwise-ranking-into-pairwise","slug":"integrating-listwise-ranking-into-pairwise","title":"Integrating Listwise Ranking into Pairwise-based Image-Text Retrieval","date":"2023-05-26","arxiv_id":"2305.16566","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-the-ranking-context-of-dense","slug":"enhancing-the-ranking-context-of-dense","title":"Enhancing the Ranking Context of Dense Retrieval Methods through Reciprocal Nearest Neighbors","date":"2023-05-25","arxiv_id":"2305.15720","repositories_listed":1,"syntology":null},{"url":"/paper/pace-unified-multi-modal-dialogue-pre","slug":"pace-unified-multi-modal-dialogue-pre","title":"PaCE: Unified Multi-modal Dialogue Pre-training with Progressive and Compositional Experts","date":"2023-05-24","arxiv_id":"2305.14839","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/pace-unified-multi-modal-dialogue-pre#ran","syntology_url":"https://syntology.ai/paper/2305.14839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14839"}},"official":{"repos":["AlibabaResearch/DAMO-ConvAI"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/s-clip-semi-supervised-vision-language-1","slug":"s-clip-semi-supervised-vision-language-1","title":"S-CLIP: Semi-supervised Vision-Language Learning using Few Specialist Captions","date":"2023-05-23","arxiv_id":"2305.14095","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/s-clip-semi-supervised-vision-language-1#ran","syntology_url":"https://syntology.ai/paper/2305.14095","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14095"}},"official":{"repos":["alinlab/s-clip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-the-music-stops-tip-of-the-tongue","slug":"when-the-music-stops-tip-of-the-tongue","title":"When the Music Stops: Tip-of-the-Tongue Retrieval for Music","date":"2023-05-23","arxiv_id":"2305.14072","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-vision-language-pre-training-with","slug":"enhancing-vision-language-pre-training-with","title":"Enhancing Vision-Language Pre-Training with Jointly Learned Questioner and Dense Captioner","date":"2023-05-19","arxiv_id":"2305.11769","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-retrieval-for-motion-and-text-via","slug":"cross-modal-retrieval-for-motion-and-text-via","title":"Cross-Modal Retrieval for Motion and Text via DopTriple Loss","date":"2023-05-07","arxiv_id":"2305.04195","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-differential-search-index-for","slug":"understanding-differential-search-index-for","title":"Understanding Differential Search Index for Text Retrieval","date":"2023-05-03","arxiv_id":"2305.02073","repositories_listed":1,"syntology":null},{"url":"/paper/from-association-to-generation-text-only","slug":"from-association-to-generation-text-only","title":"From Association to Generation: Text-only Captioning by Unsupervised Cross-modal Mapping","date":"2023-04-26","arxiv_id":"2304.13273","repositories_listed":1,"syntology":null},{"url":"/paper/learnable-pillar-based-re-ranking-for-image","slug":"learnable-pillar-based-re-ranking-for-image","title":"Learnable Pillar-based Re-ranking for Image-Text Retrieval","date":"2023-04-25","arxiv_id":"2304.12570","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-benchmarks-for-cross-modal-image","slug":"rethinking-benchmarks-for-cross-modal-image","title":"Rethinking Benchmarks for Cross-modal Image-text Retrieval","date":"2023-04-21","arxiv_id":"2304.10824","repositories_listed":1,"syntology":null},{"url":"/paper/image-text-retrieval-via-preserving-main","slug":"image-text-retrieval-via-preserving-main","title":"Image-text Retrieval via Preserving Main Semantics of Vision","date":"2023-04-20","arxiv_id":"2304.10254","repositories_listed":1,"syntology":null},{"url":"/paper/svitt-temporal-learning-of-sparse-video-text","slug":"svitt-temporal-learning-of-sparse-video-text","title":"SViTT: Temporal Learning of Sparse Video-Text Transformers","date":"2023-04-18","arxiv_id":"2304.08809","repositories_listed":1,"syntology":null},{"url":"/paper/equivariant-similarity-for-vision-language","slug":"equivariant-similarity-for-vision-language","title":"Equivariant Similarity for Vision-Language Foundation Models","date":"2023-03-25","arxiv_id":"2303.14465","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/equivariant-similarity-for-vision-language#ran","syntology_url":"https://syntology.ai/paper/2303.14465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.14465"}},"official":{"repos":["wangt-cn/eqben"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cico-domain-aware-sign-language-retrieval-via","slug":"cico-domain-aware-sign-language-retrieval-via","title":"CiCo: Domain-Aware Sign Language Retrieval via Cross-Lingual Contrastive Learning","date":"2023-03-22","arxiv_id":"2303.12793","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-preserving-augmentation-for-robust","slug":"semantic-preserving-augmentation-for-robust","title":"Semantic-Preserving Augmentation for Robust Image-Text Retrieval","date":"2023-03-10","arxiv_id":"2303.05692","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-retrieval-with-partially","slug":"cross-modal-retrieval-with-partially","title":"Cross-Modal Retrieval with Partially Mismatched Pairs","date":"2023-02-22","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/video-text-retrieval-by-supervised-multi","slug":"video-text-retrieval-by-supervised-multi","title":"Video-Text Retrieval by Supervised Sparse Multi-Grained Learning","date":"2023-02-19","arxiv_id":"2302.09473","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-federated-learning-via-contrastive","slug":"multimodal-federated-learning-via-contrastive","title":"Multimodal Federated Learning via Contrastive Representation Ensemble","date":"2023-02-17","arxiv_id":"2302.08888","repositories_listed":1,"syntology":null},{"url":"/paper/differentiable-outlier-detection-enable","slug":"differentiable-outlier-detection-enable","title":"Differentiable Outlier Detection Enable Robust Deep Multimodal Analysis","date":"2023-02-11","arxiv_id":"2302.05608","repositories_listed":1,"syntology":null},{"url":"/paper/lexlip-lexicon-bottlenecked-language-image","slug":"lexlip-lexicon-bottlenecked-language-image","title":"LexLIP: Lexicon-Bottlenecked Language-Image Pre-Training for Large-Scale Image-Text Retrieval","date":"2023-02-06","arxiv_id":"2302.02908","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-temporal-modeling-for-clip-based","slug":"revisiting-temporal-modeling-for-clip-based","title":"Revisiting Temporal Modeling for CLIP-based Image-to-Video Knowledge Transferring","date":"2023-01-26","arxiv_id":"2301.11116","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-video-adapter-for-parameter","slug":"multimodal-video-adapter-for-parameter","title":"MV-Adapter: Multimodal Video Transfer Learning for Video Text Retrieval","date":"2023-01-19","arxiv_id":"2301.07868","repositories_listed":1,"syntology":null},{"url":"/paper/user-unified-semantic-enhancement-with","slug":"user-unified-semantic-enhancement-with","title":"USER: Unified Semantic Enhancement with Momentum Contrast for Image-Text Retrieval","date":"2023-01-17","arxiv_id":"2301.06844","repositories_listed":1,"syntology":null},{"url":"/paper/napreg-nouns-as-proxies-regularization-for","slug":"napreg-nouns-as-proxies-regularization-for","title":"NAPReg: Nouns As Proxies Regularization for Semantically Aware Cross-Modal Embeddings","date":"2023-01-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-image-text-matching-by-cross","slug":"fine-grained-image-text-matching-by-cross","title":"Fine-Grained Image-Text Matching by Cross-Modal Hard Aligning Network","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-semantic-relationship-among","slug":"learning-semantic-relationship-among","title":"Learning Semantic Relationship Among Instances for Image-Text Matching","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/lexlip-lexicon-bottlenecked-language-image-1","slug":"lexlip-lexicon-bottlenecked-language-image-1","title":"LexLIP: Lexicon-Bottlenecked Language-Image Pre-Training for Large-Scale Image-Text Sparse Retrieval","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-molecule-structure-text-model-for","slug":"multi-modal-molecule-structure-text-model-for","title":"Multi-modal Molecule Structure-text Model for Text-based Retrieval and Editing","date":"2022-12-21","arxiv_id":"2212.10789","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-modal-molecule-structure-text-model-for#ran","syntology_url":"https://syntology.ai/paper/2212.10789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.10789"}},"official":{"repos":["chao1224/moleculestm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/unsupervised-dense-retrieval-deserves-better","slug":"unsupervised-dense-retrieval-deserves-better","title":"AugTriever: Unsupervised Dense Retrieval and Domain Adaptation by Scalable Data Augmentation","date":"2022-12-17","arxiv_id":"2212.08841","repositories_listed":1,"syntology":null},{"url":"/paper/are-multimodal-models-robust-to-image-and","slug":"are-multimodal-models-robust-to-image-and","title":"Benchmarking Robustness of Multimodal Image-Text Models under Distribution Shift","date":"2022-12-15","arxiv_id":"2212.08044","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-multimodal-models-robust-to-image-and#ran","syntology_url":"https://syntology.ai/paper/2212.08044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.08044"}},"official":null}},{"url":"/paper/dialogcc-large-scale-multi-modal-dialogue","slug":"dialogcc-large-scale-multi-modal-dialogue","title":"DialogCC: An Automated Pipeline for Creating High-Quality Multi-Modal Dialogue Dataset","date":"2022-12-08","arxiv_id":"2212.04119","repositories_listed":1,"syntology":null},{"url":"/paper/named-entity-and-relation-extraction-with","slug":"named-entity-and-relation-extraction-with","title":"Named Entity and Relation Extraction with Multi-Modal Retrieval","date":"2022-12-03","arxiv_id":"2212.01612","repositories_listed":1,"syntology":null},{"url":"/paper/comclip-training-free-compositional-image-and","slug":"comclip-training-free-compositional-image-and","title":"ComCLIP: Training-Free Compositional Image and Text Matching","date":"2022-11-25","arxiv_id":"2211.13854","repositories_listed":1,"syntology":null},{"url":"/paper/seeing-what-you-miss-vision-language-pre","slug":"seeing-what-you-miss-vision-language-pre","title":"Seeing What You Miss: Vision-Language Pre-training with Semantic Completion Learning","date":"2022-11-24","arxiv_id":"2211.13437","repositories_listed":1,"syntology":null},{"url":"/paper/coco-dr-combating-distribution-shifts-in-zero","slug":"coco-dr-combating-distribution-shifts-in-zero","title":"COCO-DR: Combating Distribution Shifts in Zero-Shot Dense Retrieval with Contrastive and Distributionally Robust Learning","date":"2022-10-27","arxiv_id":"2210.15212","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/coco-dr-combating-distribution-shifts-in-zero#ran","syntology_url":"https://syntology.ai/paper/2210.15212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.15212"}},"official":{"repos":["openmatch/coco-dr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rsvg-exploring-data-and-models-for-visual","slug":"rsvg-exploring-data-and-models-for-visual","title":"RSVG: Exploring Data and Models for Visual Grounding on Remote Sensing Data","date":"2022-10-23","arxiv_id":"2210.12634","repositories_listed":1,"syntology":null},{"url":"/paper/simans-simple-ambiguous-negatives-sampling","slug":"simans-simple-ambiguous-negatives-sampling","title":"SimANS: Simple Ambiguous Negatives Sampling for Dense Text Retrieval","date":"2022-10-21","arxiv_id":"2210.11773","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simans-simple-ambiguous-negatives-sampling#ran","syntology_url":"https://syntology.ai/paper/2210.11773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.11773"}},"official":{"repos":["microsoft/simxns"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vtc-improving-video-text-retrieval-with-user","slug":"vtc-improving-video-text-retrieval-with-user","title":"VTC: Improving Video-Text Retrieval with User Comments","date":"2022-10-19","arxiv_id":"2210.10820","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vtc-improving-video-text-retrieval-with-user#ran","syntology_url":"https://syntology.ai/paper/2210.10820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10820"}},"official":null}},{"url":"/paper/medclip-contrastive-learning-from-unpaired","slug":"medclip-contrastive-learning-from-unpaired","title":"MedCLIP: Contrastive Learning from Unpaired Medical Images and Text","date":"2022-10-18","arxiv_id":"2210.10163","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-pre-training-basics-recent","slug":"vision-language-pre-training-basics-recent","title":"Vision-Language Pre-training: Basics, Recent Advances, and Future Trends","date":"2022-10-17","arxiv_id":"2210.09263","repositories_listed":1,"syntology":null},{"url":"/paper/map-modality-agnostic-uncertainty-aware","slug":"map-modality-agnostic-uncertainty-aware","title":"MAP: Multimodal Uncertainty-Aware Vision-Language Pre-training Model","date":"2022-10-11","arxiv_id":"2210.05335","repositories_listed":1,"syntology":null},{"url":"/paper/mixed-modality-representation-learning-and","slug":"mixed-modality-representation-learning-and","title":"Mixed-modality Representation Learning and Pre-training for Joint Table-and-Text Retrieval in OpenQA","date":"2022-10-11","arxiv_id":"2210.05197","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mixed-modality-representation-learning-and#ran","syntology_url":"https://syntology.ai/paper/2210.05197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05197"}},"official":{"repos":["jun-jie-huang/otter"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/contextualized-generative-retrieval","slug":"contextualized-generative-retrieval","title":"Nonparametric Decoding for Generative Retrieval","date":"2022-10-05","arxiv_id":"2210.02068","repositories_listed":1,"syntology":null},{"url":"/paper/speechclip-integrating-speech-with-pre","slug":"speechclip-integrating-speech-with-pre","title":"SpeechCLIP: Integrating Speech with Pre-Trained Vision and Language Model","date":"2022-10-03","arxiv_id":"2210.00705","repositories_listed":1,"syntology":null},{"url":"/paper/decaf-joint-decoding-of-answers-and-logical","slug":"decaf-joint-decoding-of-answers-and-logical","title":"DecAF: Joint Decoding of Answers and Logical Forms for Question Answering over Knowledge Bases","date":"2022-09-30","arxiv_id":"2210.00063","repositories_listed":1,"syntology":null},{"url":"/paper/audio-retrieval-with-wavtext5k-and-clap","slug":"audio-retrieval-with-wavtext5k-and-clap","title":"Audio Retrieval with WavText5K and CLAP Training","date":"2022-09-28","arxiv_id":"2209.14275","repositories_listed":1,"syntology":null},{"url":"/paper/mr-right-multimodal-retrieval-on","slug":"mr-right-multimodal-retrieval-on","title":"Mr. Right: Multimodal Retrieval on Representation of ImaGe witH Text","date":"2022-09-28","arxiv_id":"2209.13764","repositories_listed":1,"syntology":null},{"url":"/paper/clip-vip-adapting-pre-trained-image-text","slug":"clip-vip-adapting-pre-trained-image-text","title":"CLIP-ViP: Adapting Pre-trained Image-Text Model to Video-Language Representation Alignment","date":"2022-09-14","arxiv_id":"2209.06430","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clip-vip-adapting-pre-trained-image-text#ran","syntology_url":"https://syntology.ai/paper/2209.06430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.06430"}},"official":{"repos":["microsoft/xpretrain"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-taboo-an-analysis-of-attribute-based-zero","slug":"vl-taboo-an-analysis-of-attribute-based-zero","title":"VL-Taboo: An Analysis of Attribute-based Zero-shot Capabilities of Vision-Language Models","date":"2022-09-12","arxiv_id":"2209.06103","repositories_listed":1,"syntology":null},{"url":"/paper/feta-towards-specializing-foundation-models","slug":"feta-towards-specializing-foundation-models","title":"FETA: Towards Specializing Foundation Models for Expert Task Applications","date":"2022-09-08","arxiv_id":"2209.03648","repositories_listed":1,"syntology":null},{"url":"/paper/design-of-the-topology-for-contrastive-visual","slug":"design-of-the-topology-for-contrastive-visual","title":"Design of the topology for contrastive visual-textual alignment","date":"2022-09-05","arxiv_id":"2209.02127","repositories_listed":1,"syntology":null},{"url":"/paper/universal-multi-modality-retrieval-with-one","slug":"universal-multi-modality-retrieval-with-one","title":"Universal Vision-Language Dense Retrieval: Learning A Unified Representation Space for Multi-Modal Retrieval","date":"2022-09-01","arxiv_id":"2209.00179","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/universal-multi-modality-retrieval-with-one#ran","syntology_url":"https://syntology.ai/paper/2209.00179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.00179"}},"official":{"repos":["openmatch/univl-dr"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-vision-language-pretraining-with","slug":"efficient-vision-language-pretraining-with","title":"Efficient Vision-Language Pretraining with Visual Concepts and Hierarchical Alignment","date":"2022-08-29","arxiv_id":"2208.13628","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-vision-language-pretraining-with#ran","syntology_url":"https://syntology.ai/paper/2208.13628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.13628"}},"official":{"repos":["mshukor/vicha"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contrastive-audio-language-learning-for-music","slug":"contrastive-audio-language-learning-for-music","title":"Contrastive Audio-Language Learning for Music","date":"2022-08-25","arxiv_id":"2208.12208","repositories_listed":1,"syntology":null},{"url":"/paper/intra-modal-constraint-loss-for-image-text","slug":"intra-modal-constraint-loss-for-image-text","title":"Intra-Modal Constraint Loss For Image-Text Retrieval","date":"2022-07-11","arxiv_id":"2207.05024","repositories_listed":1,"syntology":null},{"url":"/paper/a-dense-representation-framework-for-lexical","slug":"a-dense-representation-framework-for-lexical","title":"A Dense Representation Framework for Lexical and Semantic Matching","date":"2022-06-20","arxiv_id":"2206.09912","repositories_listed":1,"syntology":null},{"url":"/paper/mixgen-a-new-multi-modal-data-augmentation","slug":"mixgen-a-new-multi-modal-data-augmentation","title":"MixGen: A New Multi-Modal Data Augmentation","date":"2022-06-16","arxiv_id":"2206.08358","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/mixgen-a-new-multi-modal-data-augmentation#ran","syntology_url":"https://syntology.ai/paper/2206.08358","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08358"}},"official":{"repos":["amazon-research/mix-generation"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/coarse-to-fine-vision-language-pre-training","slug":"coarse-to-fine-vision-language-pre-training","title":"Coarse-to-Fine Vision-Language Pre-training with Fusion in the Backbone","date":"2022-06-15","arxiv_id":"2206.07643","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/coarse-to-fine-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2206.07643","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07643"}},"official":{"repos":["microsoft/fiber"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/uni-perceiver-moe-learning-sparse-generalist","slug":"uni-perceiver-moe-learning-sparse-generalist","title":"Uni-Perceiver-MoE: Learning Sparse Generalist Models with Conditional MoEs","date":"2022-06-09","arxiv_id":"2206.04674","repositories_listed":1,"syntology":null},{"url":"/paper/cross-lingual-and-multilingual-clip","slug":"cross-lingual-and-multilingual-clip","title":"Cross-lingual and Multilingual CLIP","date":"2022-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/cross-view-language-modeling-towards-unified","slug":"cross-view-language-modeling-towards-unified","title":"Cross-View Language Modeling: Towards Unified Cross-Lingual Cross-Modal Pre-training","date":"2022-06-01","arxiv_id":"2206.00621","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cross-view-language-modeling-towards-unified#ran","syntology_url":"https://syntology.ai/paper/2206.00621","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.00621"}},"official":{"repos":["zengyan-97/cclm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-and-light-weight-answer-text-retrieval","slug":"fast-and-light-weight-answer-text-retrieval","title":"Fast and Light-Weight Answer Text Retrieval in Dialogue Systems","date":"2022-05-27","arxiv_id":"2205.14226","repositories_listed":1,"syntology":null},{"url":"/paper/hlatr-enhance-multi-stage-text-retrieval-with","slug":"hlatr-enhance-multi-stage-text-retrieval-with","title":"HLATR: Enhance Multi-stage Text Retrieval with Hybrid List Aware Transformer Reranking","date":"2022-05-21","arxiv_id":"2205.10569","repositories_listed":1,"syntology":null},{"url":"/paper/zero-and-r2d2-a-large-scale-chinese-cross","slug":"zero-and-r2d2-a-large-scale-chinese-cross","title":"CCMB: A Large-scale Chinese Cross-modal Benchmark","date":"2022-05-08","arxiv_id":"2205.03860","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/zero-and-r2d2-a-large-scale-chinese-cross#ran","syntology_url":"https://syntology.ai/paper/2205.03860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.03860"}},"official":{"repos":["yuxie11/R2D2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-contrastive-learning-for-speech-1","slug":"cross-modal-contrastive-learning-for-speech-1","title":"Cross-modal Contrastive Learning for Speech Translation","date":"2022-05-05","arxiv_id":"2205.02444","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cross-modal-contrastive-learning-for-speech-1#ran","syntology_url":"https://syntology.ai/paper/2205.02444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.02444"}},"official":{"repos":["reneeye/const"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-retrieval-for-long-sequences","slug":"generative-retrieval-for-long-sequences","title":"Generative Multi-hop Retrieval","date":"2022-04-27","arxiv_id":"2204.13596","repositories_listed":1,"syntology":null},{"url":"/paper/miles-visual-bert-pre-training-with-injected","slug":"miles-visual-bert-pre-training-with-injected","title":"MILES: Visual BERT Pre-training with Injected Language Semantics for Video-text Retrieval","date":"2022-04-26","arxiv_id":"2204.12408","repositories_listed":1,"syntology":null},{"url":"/paper/socratic-models-composing-zero-shot","slug":"socratic-models-composing-zero-shot","title":"Socratic Models: Composing Zero-Shot Multimodal Reasoning with Language","date":"2022-04-01","arxiv_id":"2204.00598","repositories_listed":1,"syntology":null},{"url":"/paper/on-metric-learning-for-audio-text-cross-modal","slug":"on-metric-learning-for-audio-text-cross-modal","title":"On Metric Learning for Audio-Text Cross-Modal Retrieval","date":"2022-03-29","arxiv_id":"2203.15537","repositories_listed":1,"syntology":null},{"url":"/paper/single-stream-multi-level-alignment-for","slug":"single-stream-multi-level-alignment-for","title":"Single-Stream Multi-Level Alignment for Vision-Language Pretraining","date":"2022-03-27","arxiv_id":"2203.14395","repositories_listed":1,"syntology":null},{"url":"/paper/laprador-unsupervised-pretrained-dense","slug":"laprador-unsupervised-pretrained-dense","title":"LaPraDoR: Unsupervised Pretrained Dense Retriever for Zero-Shot Text Retrieval","date":"2022-03-11","arxiv_id":"2203.06169","repositories_listed":1,"syntology":null},{"url":"/paper/where-does-the-performance-improvement-come","slug":"where-does-the-performance-improvement-come","title":"Where Does the Performance Improvement Come From? -- A Reproducibility Concern about Image-Text Retrieval","date":"2022-03-08","arxiv_id":"2203.03853","repositories_listed":1,"syntology":null},{"url":"/paper/an-unsupervised-cross-modal-hashing-method","slug":"an-unsupervised-cross-modal-hashing-method","title":"An Unsupervised Cross-Modal Hashing Method Robust to Noisy Training Image-Text Correspondences in Remote Sensing","date":"2022-02-26","arxiv_id":"2202.13117","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-pre-training-with-triple","slug":"vision-language-pre-training-with-triple","title":"Vision-Language Pre-Training with Triple Contrastive Learning","date":"2022-02-21","arxiv_id":"2202.10401","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vision-language-pre-training-with-triple#ran","syntology_url":"https://syntology.ai/paper/2202.10401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10401"}},"official":{"repos":["uta-smile/TCL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/wukong-100-million-large-scale-chinese-cross","slug":"wukong-100-million-large-scale-chinese-cross","title":"Wukong: A 100 Million Large-scale Chinese Cross-modal Pre-training Benchmark","date":"2022-02-14","arxiv_id":"2202.06767","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wukong-100-million-large-scale-chinese-cross#ran","syntology_url":"https://syntology.ai/paper/2202.06767","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.06767"}},"official":null}},{"url":"/paper/audio-retrieval-with-natural-language-queries-1","slug":"audio-retrieval-with-natural-language-queries-1","title":"Audio Retrieval with Natural Language Queries: A Benchmark Study","date":"2021-12-17","arxiv_id":"2112.09418","repositories_listed":1,"syntology":null},{"url":"/paper/clip-lite-information-efficient-visual","slug":"clip-lite-information-efficient-visual","title":"CLIP-Lite: Information Efficient Visual Representation Learning with Language Supervision","date":"2021-12-14","arxiv_id":"2112.07133","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/clip-lite-information-efficient-visual#ran","syntology_url":"https://syntology.ai/paper/2112.07133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.07133"}},"official":{"repos":["4m4n5/CLIP-Lite"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/densifying-sparse-representations-for-passage","slug":"densifying-sparse-representations-for-passage","title":"Densifying Sparse Representations for Passage Retrieval by Representational Slicing","date":"2021-12-09","arxiv_id":"2112.04666","repositories_listed":1,"syntology":null},{"url":"/paper/video-text-pre-training-with-learned-regions","slug":"video-text-pre-training-with-learned-regions","title":"Video-Text Pre-training with Learned Regions","date":"2021-12-02","arxiv_id":"2112.01194","repositories_listed":1,"syntology":null},{"url":"/paper/filip-fine-grained-interactive-language-image-1","slug":"filip-fine-grained-interactive-language-image-1","title":"FILIP: Fine-grained Interactive Language-Image Pre-Training","date":"2021-11-09","arxiv_id":"2111.07783","repositories_listed":1,"syntology":null},{"url":"/paper/negative-sample-is-negative-in-its-own-way","slug":"negative-sample-is-negative-in-its-own-way","title":"Negative Sample is Negative in Its Own Way: Tailoring Negative Sentences for Image-Text Retrieval","date":"2021-11-05","arxiv_id":"2111.03349","repositories_listed":1,"syntology":null},{"url":"/paper/less-is-more-pretrain-a-strong-siamese","slug":"less-is-more-pretrain-a-strong-siamese","title":"Less is More: Pretrain a Strong Siamese Encoder for Dense Text Retrieval Using a Weak Decoder","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/dense-hierarchical-retrieval-for-open-domain","slug":"dense-hierarchical-retrieval-for-open-domain","title":"Dense Hierarchical Retrieval for Open-Domain Question Answering","date":"2021-10-28","arxiv_id":"2110.15439","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-retriever-ranker-for-dense-text","slug":"adversarial-retriever-ranker-for-dense-text","title":"Adversarial Retriever-Ranker for dense text retrieval","date":"2021-10-07","arxiv_id":"2110.03611","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/adversarial-retriever-ranker-for-dense-text#ran","syntology_url":"https://syntology.ai/paper/2110.03611","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.03611"}},"official":{"repos":["microsoft/ar2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hanet-hierarchical-alignment-networks-for","slug":"hanet-hierarchical-alignment-networks-for","title":"HANet: Hierarchical Alignment Networks for Video-Text Retrieval","date":"2021-07-26","arxiv_id":"2107.12059","repositories_listed":1,"syntology":null},{"url":"/paper/multi-stage-pre-training-over-simplified","slug":"multi-stage-pre-training-over-simplified","title":"Multi-stage Pre-training over Simplified Multimodal Pre-training Models","date":"2021-07-22","arxiv_id":"2107.14596","repositories_listed":1,"syntology":{"n":9,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":9,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/multi-stage-pre-training-over-simplified#ran","syntology_url":"https://syntology.ai/paper/2107.14596","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.14596"}},"official":{"repos":["lttsmn/LXMERT-S"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/wikigraphs-a-wikipedia-text-knowledge-graph","slug":"wikigraphs-a-wikipedia-text-knowledge-graph","title":"WikiGraphs: A Wikipedia Text - Knowledge Graph Paired Dataset","date":"2021-07-20","arxiv_id":"2107.09556","repositories_listed":1,"syntology":null},{"url":"/paper/more-robust-dense-retrieval-with-contrastive","slug":"more-robust-dense-retrieval-with-contrastive","title":"More Robust Dense Retrieval with Contrastive Dual Learning","date":"2021-07-16","arxiv_id":"2107.07773","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/more-robust-dense-retrieval-with-contrastive#ran","syntology_url":"https://syntology.ai/paper/2107.07773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.07773"}},"official":{"repos":["thunlp/DANCE"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-modality-interaction-modeling-for","slug":"dynamic-modality-interaction-modeling-for","title":"Dynamic Modality Interaction Modeling for Image-Text Retrieval","date":"2021-07-11","arxiv_id":null,"repositories_listed":1,"syntology":null}],"record_sha256":"bde99863f11bd3f94cba4648071c6965c0bc550da7178ce66d0854b5f7f0e947","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}