{"url":"/task/text-clustering","name":"Text Clustering","slug":"text-clustering","description_markdown":"Grouping a set of texts in such a way that objects in the same group (called a cluster) are more similar (in some sense) to each other than to those in other groups (clusters). (Source: Adapted from Wikipedia)","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":123,"papers_with_code":38,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":5,"subtasks":3,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/text-clustering-on-mteb","slug":"text-clustering-on-mteb","dataset":"MTEB","dataset_url":"/dataset/mteb","rows_in_archive":31,"metrics":["V-Measure"],"first_row_in_archive_order":{"model":"ST5-XXL","paper_title":"MTEB: Massive Text Embedding Benchmark","paper_url":"/paper/mteb-massive-text-embedding-benchmark","paper_date":"2022-10-13","arxiv_id":"2210.07316","code_links":[{"title":"embeddings-benchmark/mteb","url":"https://github.com/embeddings-benchmark/mteb"},{"title":"lyon-nlp/mteb-french","url":"https://github.com/lyon-nlp/mteb-french"},{"title":"climsocana/tecb-de","url":"https://github.com/climsocana/tecb-de"},{"title":"wadoodabdul/clinical_ner_benchmark","url":"https://github.com/wadoodabdul/clinical_ner_benchmark"},{"title":"basf/chemteb","url":"https://github.com/basf/chemteb"}],"syntology":{"n":13,"n_ran":3,"n_unverified":10,"n_pointer_only":0}}},{"leaderboard":"/sota/text-clustering-on-20-newsgroups","slug":"text-clustering-on-20-newsgroups","dataset":"20 Newsgroups","dataset_url":"/dataset/20-newsgroups","rows_in_archive":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"G-BAT","paper_title":"Neural Topic Modeling with Bidirectional Adversarial Training","paper_url":"/paper/neural-topic-modeling-with-bidirectional","paper_date":"2020-04-26","arxiv_id":"2004.12331","code_links":[{"title":"zll17/Neural_Topic_Models","url":"https://github.com/zll17/Neural_Topic_Models"}],"syntology":null}},{"leaderboard":"/sota/text-clustering-on-urdu-news-headlines","slug":"text-clustering-on-urdu-news-headlines","dataset":"Urdu News Headlines Dataset","dataset_url":"/dataset/urdu-news-headlines-dataset","rows_in_archive":1,"metrics":["Related Headlines"],"first_row_in_archive_order":{"model":"Vector Space Model","paper_title":"Clustering Urdu News Using Headlines","paper_url":"/paper/clustering-urdu-news-using-headlines","paper_date":"2015-09-27","arxiv_id":null,"code_links":[{"title":"SyedMuhammadFaheem/Urdu-News-Clustering","url":"https://github.com/SyedMuhammadFaheem/Urdu-News-Clustering"}],"syntology":null}}],"datasets":[{"url":"/dataset/mteb","name":"MTEB","full_name":"Massive Text Embedding Benchmark","num_papers_in_archive":155},{"url":"/dataset/20-newsgroups","name":"20 Newsgroups","full_name":"","num_papers_in_archive":27},{"url":"/dataset/covid-19-twitter-chatter-dataset","name":"COVID-19 Twitter Chatter Dataset","full_name":"","num_papers_in_archive":10},{"url":"/dataset/italian-crime-news","name":"DICE: a Dataset of Italian Crime Event news","full_name":"from Gazzetta di Modena [2011-2021]","num_papers_in_archive":3},{"url":"/dataset/urdu-news-headlines-dataset","name":"Urdu News Headlines Dataset","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/hierarchical-text-clustering","name":"Hierarchical Text Clustering"},{"url":"/task/open-intent-discovery","name":"Open Intent Discovery"},{"url":"/task/short-text-clustering","name":"Short Text Clustering"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":38,"tagged_in_all":123,"items":[{"url":"/paper/mteb-massive-text-embedding-benchmark","title":"MTEB: Massive Text Embedding Benchmark","date":"2022-10-13","arxiv_id":"2210.07316","repositories_listed":5,"syntology":{"n":13,"n_ran":3,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/short-text-clustering-via-convolutional","title":"Short Text Clustering via Convolutional Neural Networks","date":"2015-06-01","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/a-proposition-level-clustering-approach-for","title":"Proposition-Level Clustering for Multi-Document Summarization","date":"2021-12-16","arxiv_id":"2112.08770","repositories_listed":2,"syntology":null},{"url":"/paper/supporting-clustering-with-contrastive","title":"Supporting Clustering with Contrastive Learning","date":"2021-03-24","arxiv_id":"2103.12953","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/discovering-new-intents-with-deep-aligned","title":"Discovering New Intents with Deep Aligned Clustering","date":"2020-12-16","arxiv_id":"2012.08987","repositories_listed":2,"syntology":null},{"url":"/paper/dissimilarity-mixture-autoencoder-for-deep","title":"Dissimilarity Mixture Autoencoder for Deep Clustering","date":"2020-06-15","arxiv_id":"2006.08177","repositories_listed":2,"syntology":null},{"url":"/paper/cse-sfp-enabling-unsupervised-sentence","title":"CSE-SFP: Enabling Unsupervised Sentence Representation Learning via a Single Forward Pass","date":"2025-05-01","arxiv_id":"2505.00389","repositories_listed":1,"syntology":null},{"url":"/paper/reliable-pseudo-labeling-via-optimal","title":"Reliable Pseudo-labeling via Optimal Transport with Attention for Short Text Clustering","date":"2025-01-25","arxiv_id":"2501.15194","repositories_listed":1,"syntology":null},{"url":"/paper/discriminative-representation-learning-via","title":"Discriminative Representation learning via Attention-Enhanced Contrastive Learning for Short Text Clustering","date":"2025-01-07","arxiv_id":"2501.03584","repositories_listed":1,"syntology":null},{"url":"/paper/text-clustering-as-classification-with-llms","title":"Text Clustering as Classification with LLMs","date":"2024-09-30","arxiv_id":"2410.00927","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_unverified":0,"n_pointer_only":6}},{"url":"/paper/neurcam-interpretable-neural-clustering-via","title":"NeurCAM: Interpretable Neural Clustering via Additive Models","date":"2024-08-23","arxiv_id":"2408.13361","repositories_listed":1,"syntology":null},{"url":"/paper/guiding-sentiment-analysis-with-hierarchical","title":"Guiding Sentiment Analysis with Hierarchical Text Clustering: Analyzing the German X/Twitter Discourse on Face Masks in the 2020 COVID-19 Pandemic","date":"2024-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/human-interpretable-clustering-of-short-text","title":"Human-interpretable clustering of short-text using large language models","date":"2024-05-12","arxiv_id":"2405.07278","repositories_listed":1,"syntology":null},{"url":"/paper/more-discriminative-sentence-embeddings-via","title":"More Discriminative Sentence Embeddings via Semantic Graph Smoothing","date":"2024-02-20","arxiv_id":"2402.12890","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-enable-few-shot","title":"Large Language Models Enable Few-Shot Clustering","date":"2023-07-02","arxiv_id":"2307.00524","repositories_listed":1,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/clusterllm-large-language-models-as-a-guide","title":"ClusterLLM: Large Language Models as a Guide for Text Clustering","date":"2023-05-24","arxiv_id":"2305.14871","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/robust-representation-learning-with-reliable","title":"Robust Representation Learning with Reliable Pseudo-labels Generation via Self-Adaptive Optimal Transport for Short Text Clustering","date":"2023-05-23","arxiv_id":"2305.16335","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/influence-of-various-text-embeddings-on","title":"Influence of various text embeddings on clustering performance in NLP","date":"2023-05-04","arxiv_id":"2305.03144","repositories_listed":1,"syntology":null},{"url":"/paper/deeplens-interactive-out-of-distribution-data","title":"DeepLens: Interactive Out-of-distribution Data Detection in NLP Models","date":"2023-03-02","arxiv_id":"2303.01577","repositories_listed":1,"syntology":null},{"url":"/paper/very-large-language-model-as-a-unified","title":"Very Large Language Model as a Unified Methodology of Text Mining","date":"2022-12-19","arxiv_id":"2212.09271","repositories_listed":1,"syntology":null},{"url":"/paper/training-effective-neural-sentence-encoders","title":"Training Effective Neural Sentence Encoders from Automatically Mined Paraphrases","date":"2022-07-26","arxiv_id":"2207.12759","repositories_listed":1,"syntology":null},{"url":"/paper/clustering-similar-amendments-at-the-italian","title":"Clustering Similar Amendments at the Italian Senate","date":"2022-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ease-entity-aware-contrastive-learning-of-1","title":"EASE: Entity-Aware Contrastive Learning of Sentence Embedding","date":"2022-05-09","arxiv_id":"2205.04260","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/subspace-co-clustering-with-two-way-graph","title":"Subspace Co-clustering with Two-Way Graph Convolution","date":"2022-02-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/proposition-level-clustering-for-multi","title":"Proposition-Level Clustering for Multi-Document Summarization","date":"2022-01-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/task-oriented-clustering-for-dialogues","title":"Task-Oriented Clustering for Dialogues","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/translation-transformers-rediscover-inherent","title":"Translation Transformers Rediscover Inherent Data Domains","date":"2021-09-16","arxiv_id":"2109.07864","repositories_listed":1,"syntology":null},{"url":"/paper/learn-the-big-picture-representation-learning","title":"Learn The Big Picture: Representation Learning for Clustering","date":"2021-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/efficient-sparse-spherical-k-means-for","title":"Efficient Sparse Spherical k-Means for Document Clustering","date":"2021-07-30","arxiv_id":"2108.00895","repositories_listed":1,"syntology":null},{"url":"/paper/comstreamclust-a-communicative-text","title":"ComStreamClust: a communicative multi-agent approach to text clustering in streaming data","date":"2020-10-11","arxiv_id":"2010.05349","repositories_listed":1,"syntology":null}],"syntology_records":7,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}