{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-matching/papers/2","list_of":"/task/text-matching","task":"Text Matching","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":364,"counts":{"archive_papers_tagged":364,"with_a_code_link":161,"where_syntology_ran_a_sample":48,"not_listed_spam_title":0,"listed":364,"listed_where_code_ran":48,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":40,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":40,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-matching","prev":"/task/text-matching","next":"/task/text-matching/papers/3","papers":[{"url":"/paper/a-dense-representation-framework-for-lexical","slug":"a-dense-representation-framework-for-lexical","title":"A Dense Representation Framework for Lexical and Semantic Matching","date":"2022-06-20","arxiv_id":"2206.09912","repositories_listed":1,"syntology":null},{"url":"/paper/what-is-where-by-looking-weakly-supervised","slug":"what-is-where-by-looking-weakly-supervised","title":"What is Where by Looking: Weakly-Supervised Open-World Phrase-Grounding without Text Inputs","date":"2022-06-19","arxiv_id":"2206.09358","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-the-openness-of-clip","slug":"rethinking-the-openness-of-clip","title":"Delving into the Openness of CLIP","date":"2022-06-04","arxiv_id":"2206.01986","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/rethinking-the-openness-of-clip#ran","syntology_url":"https://syntology.ai/paper/2206.01986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01986"}},"official":{"repos":["lancopku/clip-openness"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/gr-gan-gradual-refinement-text-to-image","slug":"gr-gan-gradual-refinement-text-to-image","title":"GR-GAN: Gradual Refinement Text-to-image Generation","date":"2022-05-23","arxiv_id":"2205.11273","repositories_listed":1,"syntology":null},{"url":"/paper/zero-and-r2d2-a-large-scale-chinese-cross","slug":"zero-and-r2d2-a-large-scale-chinese-cross","title":"CCMB: A Large-scale Chinese Cross-modal Benchmark","date":"2022-05-08","arxiv_id":"2205.03860","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/zero-and-r2d2-a-large-scale-chinese-cross#ran","syntology_url":"https://syntology.ai/paper/2205.03860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.03860"}},"official":{"repos":["yuxie11/R2D2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/declaration-based-prompt-tuning-for-visual","slug":"declaration-based-prompt-tuning-for-visual","title":"Declaration-based Prompt Tuning for Visual Question Answering","date":"2022-05-05","arxiv_id":"2205.02456","repositories_listed":1,"syntology":null},{"url":"/paper/language-models-can-see-plugging-visual","slug":"language-models-can-see-plugging-visual","title":"Language Models Can See: Plugging Visual Controls in Text Generation","date":"2022-05-05","arxiv_id":"2205.02655","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-models-can-see-plugging-visual#ran","syntology_url":"https://syntology.ai/paper/2205.02655","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.02655"}},"official":{"repos":["yxuansu/magic"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/no-token-left-behind-explainability-aided","slug":"no-token-left-behind-explainability-aided","title":"No Token Left Behind: Explainability-Aided Image Classification and Generation","date":"2022-04-11","arxiv_id":"2204.04908","repositories_listed":1,"syntology":null},{"url":"/paper/improving-multi-task-generalization-ability","slug":"improving-multi-task-generalization-ability","title":"Match-Prompt: Improving Multi-task Generalization Ability for Neural Text Matching via Prompt Learning","date":"2022-04-06","arxiv_id":"2204.02725","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-matching-from-different-perspectives-1","slug":"semantic-matching-from-different-perspectives-1","title":"Semantic Matching from Different Perspectives","date":"2022-02-14","arxiv_id":"2202.06517","repositories_listed":1,"syntology":null},{"url":"/paper/mvp-multi-stage-vision-language-pre-training","slug":"mvp-multi-stage-vision-language-pre-training","title":"MVPTR: Multi-Level Semantic Alignment for Vision-Language Pre-Training via Multi-Stage Learning","date":"2022-01-29","arxiv_id":"2201.12596","repositories_listed":1,"syntology":null},{"url":"/paper/negative-aware-attention-framework-for-image","slug":"negative-aware-attention-framework-for-image","title":"Negative-Aware Attention Framework for Image-Text Matching","date":"2022-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/denseclip-language-guided-dense-prediction","slug":"denseclip-language-guided-dense-prediction","title":"DenseCLIP: Language-Guided Dense Prediction with Context-Aware Prompting","date":"2021-12-02","arxiv_id":"2112.01518","repositories_listed":1,"syntology":null},{"url":"/paper/learning-with-noisy-correspondence-for-cross","slug":"learning-with-noisy-correspondence-for-cross","title":"Learning with Noisy Correspondence for Cross-modal Matching","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/object-aware-video-language-pre-training-for","slug":"object-aware-video-language-pre-training-for","title":"Object-aware Video-language Pre-training for Retrieval","date":"2021-12-01","arxiv_id":"2112.00656","repositories_listed":1,"syntology":null},{"url":"/paper/video-and-text-matching-with-conditioned","slug":"video-and-text-matching-with-conditioned","title":"Video and Text Matching with Conditioned Embeddings","date":"2021-10-21","arxiv_id":"2110.11298","repositories_listed":1,"syntology":null},{"url":"/paper/supervised-contrastive-learning-for","slug":"supervised-contrastive-learning-for","title":"Supervised Contrastive Learning for Interpretable Long-Form Document Matching","date":"2021-08-20","arxiv_id":"2108.09190","repositories_listed":1,"syntology":null},{"url":"/paper/hanet-hierarchical-alignment-networks-for","slug":"hanet-hierarchical-alignment-networks-for","title":"HANet: Hierarchical Alignment Networks for Video-Text Retrieval","date":"2021-07-26","arxiv_id":"2107.12059","repositories_listed":1,"syntology":null},{"url":"/paper/part2word-learning-joint-embedding-of-point","slug":"part2word-learning-joint-embedding-of-point","title":"Parts2Words: Learning Joint Embedding of Point Clouds and Texts by Bidirectional Matching between Parts and Words","date":"2021-07-05","arxiv_id":"2107.01872","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/part2word-learning-joint-embedding-of-point#ran","syntology_url":"https://syntology.ai/paper/2107.01872","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.01872"}},"official":{"repos":["jlutangchuan/parts2words"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/a-modern-perspective-on-query-likelihood-with","slug":"a-modern-perspective-on-query-likelihood-with","title":"A Modern Perspective on Query Likelihood with Deep Generative Retrieval Models","date":"2021-06-25","arxiv_id":"2106.13618","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-local-and-global-scene-graph-matching","slug":"a-deep-local-and-global-scene-graph-matching","title":"A Deep Local and Global Scene-Graph Matching for Image-Text Retrieval","date":"2021-06-04","arxiv_id":"2106.02400","repositories_listed":1,"syntology":null},{"url":"/paper/learning-fine-grained-fact-article","slug":"learning-fine-grained-fact-article","title":"Learning Fine-grained Fact-Article Correspondence in Legal Cases","date":"2021-04-21","arxiv_id":"2104.10726","repositories_listed":1,"syntology":null},{"url":"/paper/let-linguistic-knowledge-enhanced-graph","slug":"let-linguistic-knowledge-enhanced-graph","title":"LET: Linguistic Knowledge Enhanced Graph Transformer for Chinese Short Text Matching","date":"2021-02-25","arxiv_id":"2102.12671","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-similarity-learning-for-language","slug":"hierarchical-similarity-learning-for-language","title":"Hierarchical Similarity Learning for Language-based Product Image Retrieval","date":"2021-02-18","arxiv_id":"2102.09375","repositories_listed":1,"syntology":null},{"url":"/paper/match-ignition-plugging-pagerank-into","slug":"match-ignition-plugging-pagerank-into","title":"Match-Ignition: Plugging PageRank into Transformer for Long-form Text Matching","date":"2021-01-16","arxiv_id":"2101.06423","repositories_listed":1,"syntology":null},{"url":"/paper/similarity-reasoning-and-filtration-for-image","slug":"similarity-reasoning-and-filtration-for-image","title":"Similarity Reasoning and Filtration for Image-Text Matching","date":"2021-01-05","arxiv_id":"2101.01368","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":6,"n_ran_checked":7,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":12,"phrase":"10 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/similarity-reasoning-and-filtration-for-image#ran","syntology_url":"https://syntology.ai/paper/2101.01368","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.01368"}},"official":{"repos":["Paranioar/SGRAF"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/cookie-contrastive-cross-modal-knowledge","slug":"cookie-contrastive-cross-modal-knowledge","title":"COOKIE: Contrastive Cross-Modal Knowledge Sharing Pre-Training for Vision-Language Representation","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-dual-semantic-relations-with-graph","slug":"learning-dual-semantic-relations-with-graph","title":"Learning Dual Semantic Relations with Graph Attention for Image-Text Matching","date":"2020-10-22","arxiv_id":"2010.11550","repositories_listed":1,"syntology":null},{"url":"/paper/wasserstein-distance-regularized-sequence","slug":"wasserstein-distance-regularized-sequence","title":"Wasserstein Distance Regularized Sequence Representation for Text Matching in Asymmetrical Domains","date":"2020-10-15","arxiv_id":"2010.07717","repositories_listed":1,"syntology":{"n":13,"n_ran":6,"n_constructed":4,"n_ran_checked":6,"n_instrument":0,"n_unverified":7,"n_honours":2,"n_violates":0,"n_no_contract":4,"n_pointer_only":13,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/wasserstein-distance-regularized-sequence#ran","syntology_url":"https://syntology.ai/paper/2010.07717","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07717"}},"official":{"repos":["RUC-WSM/WD-Match"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":4,"n_ran_no_instrument_failure":6,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/medicat-a-dataset-of-medical-images-captions","slug":"medicat-a-dataset-of-medical-images-captions","title":"MedICaT: A Dataset of Medical Images, Captions, and Textual References","date":"2020-10-12","arxiv_id":"2010.06000","repositories_listed":1,"syntology":null},{"url":"/paper/multicqa-zero-shot-transfer-of-self","slug":"multicqa-zero-shot-transfer-of-self","title":"MultiCQA: Zero-Shot Transfer of Self-Supervised Text Matching Models on a Massive Scale","date":"2020-10-02","arxiv_id":"2010.00980","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparison-of-supervised-learning-to-match","slug":"a-comparison-of-supervised-learning-to-match","title":"A Comparison of Supervised Learning to Match Methods for Product Search","date":"2020-07-20","arxiv_id":"2007.10296","repositories_listed":1,"syntology":null},{"url":"/paper/consensus-aware-visual-semantic-embedding-for","slug":"consensus-aware-visual-semantic-embedding-for","title":"Consensus-Aware Visual-Semantic Embedding for Image-Text Matching","date":"2020-07-17","arxiv_id":"2007.08883","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/consensus-aware-visual-semantic-embedding-for#ran","syntology_url":"https://syntology.ai/paper/2007.08883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.08883"}},"official":{"repos":["BruceW91/CVSE"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/rationalizing-text-matching-learning-sparse","slug":"rationalizing-text-matching-learning-sparse","title":"Rationalizing Text Matching: Learning Sparse Alignments via Optimal Transport","date":"2020-05-27","arxiv_id":"2005.13111","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rationalizing-text-matching-learning-sparse#ran","syntology_url":"https://syntology.ai/paper/2005.13111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.13111"}},"official":{"repos":["asappresearch/rationale-alignment"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/machine-reading-comprehension-the-role-of","slug":"machine-reading-comprehension-the-role-of","title":"Machine Reading Comprehension: The Role of Contextualized Language Models and Beyond","date":"2020-05-13","arxiv_id":"2005.06249","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-512-tokens-siamese-multi-depth","slug":"beyond-512-tokens-siamese-multi-depth","title":"Beyond 512 Tokens: Siamese Multi-depth Transformer-based Hierarchical Encoder for Long-Form Document Matching","date":"2020-04-26","arxiv_id":"2004.12297","repositories_listed":1,"syntology":null},{"url":"/paper/deep-multimodal-neural-architecture-search","slug":"deep-multimodal-neural-architecture-search","title":"Deep Multimodal Neural Architecture Search","date":"2020-04-25","arxiv_id":"2004.12070","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-reasoning-network-for-image-text","slug":"transformer-reasoning-network-for-image-text","title":"Transformer Reasoning Network for Image-Text Matching and Retrieval","date":"2020-04-20","arxiv_id":"2004.09144","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/transformer-reasoning-network-for-image-text#ran","syntology_url":"https://syntology.ai/paper/2004.09144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.09144"}},"official":{"repos":["mesnico/TERN"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-image-inpainting-guided-with","slug":"neural-image-inpainting-guided-with","title":"Text-Guided Neural Image Inpainting","date":"2020-04-07","arxiv_id":"2004.03212","repositories_listed":1,"syntology":null},{"url":"/paper/pixel-bert-aligning-image-pixels-with-text-by","slug":"pixel-bert-aligning-image-pixels-with-text-by","title":"Pixel-BERT: Aligning Image Pixels with Text by Deep Multi-Modal Transformers","date":"2020-04-02","arxiv_id":"2004.00849","repositories_listed":1,"syntology":null},{"url":"/paper/graph-structured-network-for-image-text","slug":"graph-structured-network-for-image-text","title":"Graph Structured Network for Image-Text Matching","date":"2020-04-01","arxiv_id":"2004.00277","repositories_listed":1,"syntology":null},{"url":"/paper/more-grounded-image-captioning-by-distilling","slug":"more-grounded-image-captioning-by-distilling","title":"More Grounded Image Captioning by Distilling Image-Text Matching Model","date":"2020-04-01","arxiv_id":"2004.00390","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/more-grounded-image-captioning-by-distilling#ran","syntology_url":"https://syntology.ai/paper/2004.00390","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.00390"}},"official":{"repos":["YuanEZhou/Grounded-Image-Captioning"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/keyword-attentive-deep-semantic-matching","slug":"keyword-attentive-deep-semantic-matching","title":"Keyword-Attentive Deep Semantic Matching","date":"2020-03-11","arxiv_id":"2003.11516","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-offline-quintuplet-loss-for-image","slug":"adaptive-offline-quintuplet-loss-for-image","title":"Adaptive Offline Quintuplet Loss for Image-Text Matching","date":"2020-03-07","arxiv_id":"2003.03669","repositories_listed":1,"syntology":null},{"url":"/paper/query-bag-matching-with-mutual-coverage-for","slug":"query-bag-matching-with-mutual-coverage-for","title":"Query-bag Matching with Mutual Coverage for Information-seeking Conversations in E-commerce","date":"2019-11-07","arxiv_id":"1911.02747","repositories_listed":1,"syntology":null},{"url":"/paper/w2vv-fully-deep-learning-for-ad-hoc-video","slug":"w2vv-fully-deep-learning-for-ad-hoc-video","title":"W2VV++: Fully Deep Learning for Ad-hoc Video Search","date":"2019-10-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-fragment-self-attention-embeddings","slug":"learning-fragment-self-attention-embeddings","title":"Learning fragment self-attention embeddings for image-text matching","date":"2019-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/tiger-text-to-image-grounding-for-image","slug":"tiger-text-to-image-grounding-for-image","title":"TIGEr: Text-to-Image Grounding for Image Caption Evaluation","date":"2019-09-04","arxiv_id":"1909.02050","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tiger-text-to-image-grounding-for-image#ran","syntology_url":"https://syntology.ai/paper/1909.02050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.02050"}},"official":{"repos":["SeleenaJM/CapEval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/position-focused-attention-network-for-image","slug":"position-focused-attention-network-for-image","title":"Position Focused Attention Network for Image-Text Matching","date":"2019-07-23","arxiv_id":"1907.09748","repositories_listed":1,"syntology":null},{"url":"/paper/constructing-interpretive-spatio-temporal","slug":"constructing-interpretive-spatio-temporal","title":"Constructing Interpretive Spatio-Temporal Features for Multi-Turn Responses Selection","date":"2019-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/matchzoo-a-learning-practicing-and-developing","slug":"matchzoo-a-learning-practicing-and-developing","title":"MatchZoo: A Learning, Practicing, and Developing System for Neural Text Matching","date":"2019-05-24","arxiv_id":"1905.10289","repositories_listed":1,"syntology":null},{"url":"/paper/document-similarity-for-texts-of-varying-1","slug":"document-similarity-for-texts-of-varying-1","title":"Document Similarity for Texts of Varying Lengths via Hidden Topics","date":"2019-03-26","arxiv_id":"1903.10675","repositories_listed":1,"syntology":null},{"url":"/paper/lattice-cnns-for-matching-based-chinese","slug":"lattice-cnns-for-matching-based-chinese","title":"Lattice CNNs for Matching Based Chinese Question Answering","date":"2019-02-25","arxiv_id":"1902.09087","repositories_listed":1,"syntology":null},{"url":"/paper/deep-cross-modal-projection-learning-for","slug":"deep-cross-modal-projection-learning-for","title":"Deep Cross-Modal Projection Learning for Image-Text Matching","date":"2018-09-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/response-ranking-with-deep-matching-networks","slug":"response-ranking-with-deep-matching-networks","title":"Response Ranking with Deep Matching Networks and External Knowledge in Information-seeking Conversation Systems","date":"2018-05-01","arxiv_id":"1805.00188","repositories_listed":1,"syntology":null},{"url":"/paper/matching-long-text-documents-via-graph","slug":"matching-long-text-documents-via-graph","title":"Matching Article Pairs with Graphical Decomposition and Convolutions","date":"2018-02-21","arxiv_id":"1802.07459","repositories_listed":1,"syntology":null},{"url":"/paper/matching-with-text-data-an-experimental","slug":"matching-with-text-data-an-experimental","title":"Matching with Text Data: An Experimental Evaluation of Methods for Matching Documents and of Measuring Match Quality","date":"2018-01-02","arxiv_id":"1801.00644","repositories_listed":1,"syntology":null},{"url":"/paper/matchzoo-a-toolkit-for-deep-text-matching","slug":"matchzoo-a-toolkit-for-deep-text-matching","title":"MatchZoo: A Toolkit for Deep Text Matching","date":"2017-07-23","arxiv_id":"1707.07270","repositories_listed":1,"syntology":null},{"url":"/paper/learning-two-branch-neural-networks-for-image","slug":"learning-two-branch-neural-networks-for-image","title":"Learning Two-Branch Neural Networks for Image-Text Matching Tasks","date":"2017-04-11","arxiv_id":"1704.03470","repositories_listed":1,"syntology":null},{"url":"/paper/luandri-a-clean-lua-interface-to-the-indri","slug":"luandri-a-clean-lua-interface-to-the-indri","title":"Luandri: a Clean Lua Interface to the Indri Search Engine","date":"2017-02-16","arxiv_id":"1702.05042","repositories_listed":1,"syntology":null},{"url":"/paper/a-study-of-matchpyramid-models-on-ad-hoc","slug":"a-study-of-matchpyramid-models-on-ad-hoc","title":"A Study of MatchPyramid Models on Ad-hoc Retrieval","date":"2016-06-15","arxiv_id":"1606.04648","repositories_listed":1,"syntology":null},{"url":null,"slug":"tng-clip-training-time-negation-data","title":"TNG-CLIP:Training-Time Negation Data Generation for Negation Awareness of CLIP","date":"2025-05-24","arxiv_id":"2505.18434","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-computer-use-grounding-via-user","title":"Scaling Computer-Use Grounding via User Interface Decomposition and Synthesis","date":"2025-05-19","arxiv_id":"2505.13227","repositories_listed":0,"syntology":null},{"url":null,"slug":"descriptive-image-text-matching-with-graded","title":"Descriptive Image-Text Matching with Graded Contextual Similarity","date":"2025-05-15","arxiv_id":"2505.09997","repositories_listed":0,"syntology":null},{"url":null,"slug":"compositional-image-text-matching-and","title":"Compositional Image-Text Matching and Retrieval by Grounding Entities","date":"2025-05-04","arxiv_id":"2505.02278","repositories_listed":0,"syntology":null},{"url":null,"slug":"lgd-leveraging-generative-descriptions-for","title":"LGD: Leveraging Generative Descriptions for Zero-Shot Referring Image Segmentation","date":"2025-04-20","arxiv_id":"2504.14467","repositories_listed":0,"syntology":null},{"url":null,"slug":"instruction-augmented-multimodal-alignment","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","date":"2025-04-16","arxiv_id":"2504.12018","repositories_listed":0,"syntology":null},{"url":null,"slug":"dependency-structure-augmented-contextual","title":"Dependency Structure Augmented Contextual Scoping Framework for Multimodal Aspect-Based Sentiment Analysis","date":"2025-04-15","arxiv_id":"2504.11331","repositories_listed":0,"syntology":null},{"url":null,"slug":"prectr-a-synergistic-framework-for","title":"PRECTR: A Synergistic Framework for Integrating Personalized Search Relevance Matching and CTR Prediction","date":"2025-03-24","arxiv_id":"2503.18395","repositories_listed":0,"syntology":null},{"url":null,"slug":"codereviewqa-the-code-review-comprehension","title":"CodeReviewQA: The Code Review Comprehension Assessment for Large Language Models","date":"2025-03-20","arxiv_id":"2503.16167","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-01019","title":"MedUnifier: Unifying Vision-and-Language Pre-training on Medical Data with Vision Generation Task using Discrete Visual Representations","date":"2025-03-02","arxiv_id":"2503.01019","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-centric-binding-in-contrastive","title":"Object-centric Binding in Contrastive Language-Image Pretraining","date":"2025-02-19","arxiv_id":"2502.14113","repositories_listed":0,"syntology":null},{"url":null,"slug":"cgi-identifying-conditional-generative-models","title":"CGI: Identifying Conditional Generative Models with Example Images","date":"2025-01-23","arxiv_id":"2501.13991","repositories_listed":0,"syntology":null},{"url":null,"slug":"mass-overcoming-language-bias-in-image-text","title":"MASS: Overcoming Language Bias in Image-Text Matching","date":"2025-01-20","arxiv_id":"2501.11469","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-textual-prompts-for-open-world-semi","title":"Learning Textual Prompts for Open-World Semi-Supervised Learning","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-head-attention-driven-dynamic-visual","title":"Multi-Head Attention Driven Dynamic Visual-Semantic Embedding for Enhanced Image-Text Matching","date":"2024-12-26","arxiv_id":"2412.19184","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-concept-centric-approach-to-multi-modality","title":"A Concept-Centric Approach to Multi-Modality Learning","date":"2024-12-18","arxiv_id":"2412.13847","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-only-submit-one-image-to-find-the-most","title":"You Only Submit One Image to Find the Most Suitable Generative Model","date":"2024-12-16","arxiv_id":"2412.12232","repositories_listed":0,"syntology":null},{"url":null,"slug":"viunit-visual-unit-tests-for-more-robust","title":"ViUniT: Visual Unit Tests for More Robust Visual Programming","date":"2024-12-12","arxiv_id":"2412.08859","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-prompt-generation-and-grounding","title":"Automatic Prompt Generation and Grounding Object Detection for Zero-Shot Image Anomaly Detection","date":"2024-11-28","arxiv_id":"2411.19220","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlm-hoi-vision-language-models-for","title":"VLM-HOI: Vision Language Models for Interpretable Human-Object Interaction Analysis","date":"2024-11-27","arxiv_id":"2411.18038","repositories_listed":0,"syntology":null},{"url":null,"slug":"entityclip-entity-centric-image-text-matching","title":"EntityCLIP: Entity-Centric Image-Text Matching via Multimodal Attentive Contrastive Learning","date":"2024-10-23","arxiv_id":"2410.17810","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-modality-gap-dimension","title":"Bridging the Modality Gap: Dimension Information Alignment and Sparse Spatial Constraint for Image-Text Matching","date":"2024-10-22","arxiv_id":"2410.16853","repositories_listed":0,"syntology":null},{"url":null,"slug":"storyboard-guided-alignment-for-fine-grained","title":"Storyboard guided Alignment for Fine-grained Video Action Recognition","date":"2024-10-18","arxiv_id":"2410.14238","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhance-graph-alignment-for-large-language","title":"Enhance Graph Alignment for Large Language Models","date":"2024-10-15","arxiv_id":"2410.11370","repositories_listed":0,"syntology":null},{"url":null,"slug":"dare-diverse-visual-question-answering-with","title":"DARE: Diverse Visual Question Answering with Robustness Evaluation","date":"2024-09-26","arxiv_id":"2409.18023","repositories_listed":0,"syntology":null},{"url":null,"slug":"jina-embeddings-v3-multilingual-embeddings","title":"jina-embeddings-v3: Multilingual Embeddings With Task LoRA","date":"2024-09-16","arxiv_id":"2409.10173","repositories_listed":0,"syntology":null},{"url":null,"slug":"nevlp-noise-robust-framework-for-efficient","title":"NEVLP: Noise-Robust Framework for Efficient Vision-Language Pre-training","date":"2024-09-15","arxiv_id":"2409.09582","repositories_listed":0,"syntology":null},{"url":null,"slug":"vidlpro-a-underline-vid-eo-underline-l","title":"VidLPRO: A $\\underline{Vid}$eo-$\\underline{L}$anguage $\\underline{P}$re-training Framework for $\\underline{Ro}$botic and Laparoscopic Surgery","date":"2024-09-07","arxiv_id":"2409.04732","repositories_listed":0,"syntology":null},{"url":null,"slug":"hound-hunting-supervision-signals-for-few-and","title":"Hound: Hunting Supervision Signals for Few and Zero Shot Node Classification on Text-attributed Graph","date":"2024-09-01","arxiv_id":"2409.00727","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-deconfounded-image-text-matching-with","title":"Towards Deconfounded Image-Text Matching with Causal Inference","date":"2024-08-22","arxiv_id":"2408.12292","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-and-compressive-adaptation-of","title":"Dynamic and Compressive Adaptation of Transformers From Images to Videos","date":"2024-08-13","arxiv_id":"2408.06840","repositories_listed":0,"syntology":null},{"url":null,"slug":"qaea-dr-a-unified-text-augmentation-framework","title":"QAEA-DR: A Unified Text Augmentation Framework for Dense Retrieval","date":"2024-07-29","arxiv_id":"2407.20207","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-bag-of-words-model-an-efficient-and","title":"Deep Bag-of-Words Model: An Efficient and Interpretable Relevance Architecture for Chinese E-Commerce","date":"2024-07-12","arxiv_id":"2407.09395","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-multimodal-deep-learning","title":"Advanced Multimodal Deep Learning Architecture for Image-Text Matching","date":"2024-06-13","arxiv_id":"2406.15306","repositories_listed":0,"syntology":null},{"url":null,"slug":"hire-hybrid-modal-interaction-with-multiple","title":"Hire: Hybrid-modal Interaction with Multiple Relational Enhancements for Image-Text Matching","date":"2024-06-05","arxiv_id":"2406.18579","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-learning-video-moment-retrieval-across","title":"Hybrid-Learning Video Moment Retrieval across Multi-Domain Labels","date":"2024-06-03","arxiv_id":"2406.01791","repositories_listed":0,"syntology":null},{"url":null,"slug":"demo-a-statistical-perspective-for-efficient","title":"DEMO: A Statistical Perspective for Efficient Image-Text Matching","date":"2024-05-19","arxiv_id":"2405.11496","repositories_listed":0,"syntology":null},{"url":null,"slug":"content-based-image-retrieval-for-multi-class","title":"Content-Based Image Retrieval for Multi-Class Volumetric Radiology Images: A Benchmark Study","date":"2024-05-15","arxiv_id":"2405.09334","repositories_listed":0,"syntology":null},{"url":null,"slug":"clip-powered-tass-target-aware-single-stream","title":"CLIP-Powered TASS: Target-Aware Single-Stream Network for Audio-Visual Question Answering","date":"2024-05-13","arxiv_id":"2405.07451","repositories_listed":0,"syntology":null}],"record_sha256":"ef846b95b01cf930afcb0293819c2de197ac681c4a5dbb417ecafdd936be4af7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}