{"url":"/task/text-similarity","name":"text similarity","slug":"text-similarity","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":271,"papers_with_code":100,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/italian-crime-news","name":"DICE: a Dataset of Italian Crime Event news","full_name":"from Gazzetta di Modena [2011-2021]","num_papers_in_archive":3},{"url":"/dataset/pubmed-abstracts","name":"PubMed Cognitive Control Abstracts","full_name":"CogText","num_papers_in_archive":2},{"url":"/dataset/phrase-in-context","name":"Phrase-in-Context","full_name":"Phrase-in-Context","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":100,"tagged_in_all":271,"items":[{"url":"/paper/stacked-cross-attention-for-image-text","title":"Stacked Cross Attention for Image-Text Matching","date":"2018-03-21","arxiv_id":"1803.08024","repositories_listed":6,"syntology":{"n":16,"n_ran":7,"n_unverified":9,"n_pointer_only":1}},{"url":"/paper/retsim-resilient-and-efficient-text","title":"RETSim: Resilient and Efficient Text Similarity","date":"2023-11-28","arxiv_id":"2311.17264","repositories_listed":3,"syntology":null},{"url":"/paper/cat-seg-cost-aggregation-for-open-vocabulary","title":"CAT-Seg: Cost Aggregation for Open-Vocabulary Semantic Segmentation","date":"2023-03-21","arxiv_id":"2303.11797","repositories_listed":3,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/actalign-zero-shot-fine-grained-video","title":"ActAlign: Zero-Shot Fine-Grained Video Classification via Language-Guided Sequence Alignment","date":"2025-06-28","arxiv_id":"2506.22967","repositories_listed":2,"syntology":null},{"url":"/paper/extensive-self-contrast-enables-feedback-free","title":"Extensive Self-Contrast Enables Feedback-Free Language Model Alignment","date":"2024-03-31","arxiv_id":"2404.00604","repositories_listed":2,"syntology":null},{"url":"/paper/tiefake-title-text-similarity-and-emotion","title":"TieFake: Title-Text Similarity and Emotion-Aware Fake News Detection","date":"2023-04-19","arxiv_id":"2304.09421","repositories_listed":2,"syntology":null},{"url":"/paper/infocse-information-aggregated-contrastive","title":"InfoCSE: Information-aggregated Contrastive Learning of Sentence Embeddings","date":"2022-10-08","arxiv_id":"2210.06432","repositories_listed":2,"syntology":null},{"url":"/paper/esimcse-enhanced-sample-building-method-for","title":"ESimCSE: Enhanced Sample Building Method for Contrastive Learning of Unsupervised Sentence Embedding","date":"2021-09-09","arxiv_id":"2109.04380","repositories_listed":2,"syntology":null},{"url":"/paper/smoothed-contrastive-learning-for","title":"Smoothed Contrastive Learning for Unsupervised Sentence Embedding","date":"2021-09-09","arxiv_id":"2109.04321","repositories_listed":2,"syntology":null},{"url":"/paper/effective-crowd-annotation-of-participants","title":"Effective Crowd-Annotation of Participants, Interventions, and Outcomes in the Text of Clinical Trial Reports","date":"2020-11-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/hhh-an-online-medical-chatbot-system-based-on-1","title":"HHH: An Online Medical Chatbot System based on Knowledge Graph and Hierarchical Bi-Directional Attention","date":"2020-02-08","arxiv_id":"2002.03140","repositories_listed":2,"syntology":null},{"url":"/paper/matching-images-and-text-with-multi-modal","title":"Matching Images and Text with Multi-modal Tensor Fusion and Re-ranking","date":"2019-08-12","arxiv_id":"1908.04011","repositories_listed":2,"syntology":null},{"url":"/paper/query-based-attention-cnn-for-text-similarity","title":"Query-based Attention CNN for Text Similarity Map","date":"2017-09-15","arxiv_id":"1709.05036","repositories_listed":2,"syntology":null},{"url":"/paper/adding-simple-structure-at-inference-improves","title":"Adding simple structure at inference improves Vision-Language Compositionality","date":"2025-06-11","arxiv_id":"2506.09691","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11493","title":"GIE-Bench: Towards Grounded Evaluation for Text-Guided Image Editing","date":"2025-05-16","arxiv_id":"2505.11493","repositories_listed":1,"syntology":null},{"url":"/paper/jtcse-joint-tensor-modulus-constraints-and","title":"JTCSE: Joint Tensor-Modulus Constraints and Cross-Attention for Unsupervised Contrastive Learning of Sentence Embeddings","date":"2025-05-05","arxiv_id":"2505.02366","repositories_listed":1,"syntology":null},{"url":"/paper/can-llms-generate-tabular-summaries-of","title":"Can LLMs Generate Tabular Summaries of Science Papers? Rethinking the Evaluation Protocol","date":"2025-04-14","arxiv_id":"2504.10284","repositories_listed":1,"syntology":null},{"url":"/paper/comac-conversational-agent-for-multi-source","title":"CoMAC: Conversational Agent for Multi-Source Auxiliary Context with Sparse and Symmetric Latent Interactions","date":"2025-03-25","arxiv_id":"2503.19274","repositories_listed":1,"syntology":null},{"url":"/paper/textinplace-indoor-visual-place-recognition","title":"TextInPlace: Indoor Visual Place Recognition in Repetitive Structures with Scene Text Spotting and Verification","date":"2025-03-09","arxiv_id":"2503.06501","repositories_listed":1,"syntology":null},{"url":"/paper/filo-zero-few-shot-anomaly-detection-by-fused","title":"FiLo++: Zero-/Few-Shot Anomaly Detection by Fused Fine-Grained Descriptions and Deformable Localization","date":"2025-01-17","arxiv_id":"2501.10067","repositories_listed":1,"syntology":null},{"url":"/paper/speechprune-context-aware-token-pruning-for","title":"SpeechPrune: Context-aware Token Pruning for Speech Information Retrieval","date":"2024-12-16","arxiv_id":"2412.12009","repositories_listed":1,"syntology":null},{"url":"/paper/vulnerability-of-text-matching-in-ml-ai","title":"Vulnerability of Text-Matching in ML/AI Conference Reviewer Assignments to Collusions","date":"2024-12-09","arxiv_id":"2412.06606","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/2d-matryoshka-training-for-information","title":"2D Matryoshka Training for Information Retrieval","date":"2024-11-26","arxiv_id":"2411.17299","repositories_listed":1,"syntology":null},{"url":"/paper/towards-cross-modal-text-molecule-retrieval","title":"Towards Cross-Modal Text-Molecule Retrieval with Better Modality Alignment","date":"2024-10-31","arxiv_id":"2410.23715","repositories_listed":1,"syntology":null},{"url":"/paper/geneol-harnessing-the-generative-power-of","title":"GenEOL: Harnessing the Generative Power of LLMs for Training-Free Sentence Embeddings","date":"2024-10-18","arxiv_id":"2410.14635","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/starbucks-improved-training-for-2d-matryoshka","title":"Starbucks: Improved Training for 2D Matryoshka Embeddings","date":"2024-10-17","arxiv_id":"2410.13230","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/slam-aac-enhancing-audio-captioning-with","title":"SLAM-AAC: Enhancing Audio Captioning with Paraphrasing Augmentation and CLAP-Refine through LLMs","date":"2024-10-12","arxiv_id":"2410.09503","repositories_listed":1,"syntology":null},{"url":"/paper/pooling-and-attention-what-are-effective","title":"Pooling And Attention: What Are Effective Designs For LLM-Based Embedding Models?","date":"2024-09-04","arxiv_id":"2409.02727","repositories_listed":1,"syntology":null},{"url":"/paper/positive-text-reframing-under-multi-strategy","title":"Positive Text Reframing under Multi-strategy Optimization","date":"2024-07-25","arxiv_id":"2407.17940","repositories_listed":1,"syntology":null},{"url":"/paper/modular-sentence-encoders-separating-language","title":"Modular Sentence Encoders: Separating Language Specialization from Cross-Lingual Alignment","date":"2024-07-20","arxiv_id":"2407.14878","repositories_listed":1,"syntology":null}],"syntology_records":5,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}