{"url":"/task/topic-models","name":"Topic Models","slug":"topic-models","description_markdown":"A topic model is a type of statistical model for discovering the abstract \"topics\" that occur in a collection of documents. Topic modeling is a frequently used text-mining tool for the discovery of hidden semantic structures in a text body.","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":881,"papers_with_code":229,"benchmarks":6,"benchmark_tables_in_archive":6,"benchmark_tables_shown":6,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":17,"subtasks":2,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/topic-models-on-20newsgroups","slug":"topic-models-on-20newsgroups","dataset":"20NewsGroups","dataset_url":"/dataset/20newsgroups","rows_in_archive":6,"metrics":["C_v"],"first_row_in_archive_order":{"model":"vONTSS","paper_title":"vONTSS: vMF based semi-supervised neural topic modeling with optimal transport","paper_url":"/paper/vontss-vmf-based-semi-supervised-neural-topic","paper_date":"2023-07-03","arxiv_id":"2307.01226","code_links":[],"syntology":null}},{"leaderboard":"/sota/topic-models-on-ag-news","slug":"topic-models-on-ag-news","dataset":"AG News","dataset_url":"/dataset/ag-news","rows_in_archive":6,"metrics":["C_v","NPMI"],"first_row_in_archive_order":{"model":"vONTSS","paper_title":"vONTSS: vMF based semi-supervised neural topic modeling with optimal transport","paper_url":"/paper/vontss-vmf-based-semi-supervised-neural-topic","paper_date":"2023-07-03","arxiv_id":"2307.01226","code_links":[],"syntology":null}},{"leaderboard":"/sota/topic-models-on-20-newsgroups","slug":"topic-models-on-20-newsgroups","dataset":"20 Newsgroups","dataset_url":"/dataset/20-newsgroups","rows_in_archive":2,"metrics":["Test perplexity"],"first_row_in_archive_order":{"model":"Bayesian SMM","paper_title":"Learning document embeddings along with their uncertainties","paper_url":"/paper/190807599","paper_date":"2019-08-20","arxiv_id":"1908.07599","code_links":[{"title":"skesiraju/BaySMM","url":"https://github.com/skesiraju/BaySMM"},{"title":"BUTSpeechFIT/BaySMM","url":"https://github.com/BUTSpeechFIT/BaySMM"}],"syntology":null}},{"leaderboard":"/sota/topic-models-on-arxiv","slug":"topic-models-on-arxiv","dataset":"Arxiv HEP-TH citation graph","dataset_url":"/dataset/arxiv","rows_in_archive":2,"metrics":["MACC","Topic Coherence@50","Topic coherence@5"],"first_row_in_archive_order":{"model":"JoSH","paper_title":"Hierarchical Topic Mining via Joint Spherical Tree and Text Embedding","paper_url":"/paper/hierarchical-topic-mining-via-joint-spherical","paper_date":"2020-07-18","arxiv_id":"2007.09536","code_links":[{"title":"yumeng5/JoSH","url":"https://github.com/yumeng5/JoSH"}],"syntology":null}},{"leaderboard":"/sota/topic-models-on-agnews","slug":"topic-models-on-agnews","dataset":"AgNews","dataset_url":null,"rows_in_archive":1,"metrics":["C_v"],"first_row_in_archive_order":{"model":"vONTSS","paper_title":"vONTSS: vMF based semi-supervised neural topic modeling with optimal transport","paper_url":"/paper/vontss-vmf-based-semi-supervised-neural-topic","paper_date":"2023-07-03","arxiv_id":"2307.01226","code_links":[],"syntology":null}},{"leaderboard":"/sota/topic-models-on-nyt","slug":"topic-models-on-nyt","dataset":"NYT","dataset_url":"/dataset/new-york-times-annotated-corpus","rows_in_archive":1,"metrics":["MACC","Topic coherence@5"],"first_row_in_archive_order":{"model":"JoSH","paper_title":"Hierarchical Topic Mining via Joint Spherical Tree and Text Embedding","paper_url":"/paper/hierarchical-topic-mining-via-joint-spherical","paper_date":"2020-07-18","arxiv_id":"2007.09536","code_links":[{"title":"yumeng5/JoSH","url":"https://github.com/yumeng5/JoSH"}],"syntology":null}}],"datasets":[{"url":"/dataset/ag-news","name":"AG News","full_name":"AG’s News Corpus","num_papers_in_archive":969},{"url":"/dataset/reddit","name":"Reddit","full_name":"","num_papers_in_archive":699},{"url":"/dataset/new-york-times-annotated-corpus","name":"New York Times Annotated Corpus","full_name":"","num_papers_in_archive":262},{"url":"/dataset/arxiv","name":"Arxiv HEP-TH citation graph","full_name":"","num_papers_in_archive":35},{"url":"/dataset/20-newsgroups","name":"20 Newsgroups","full_name":"","num_papers_in_archive":27},{"url":"/dataset/20newsgroups","name":"20NewsGroups","full_name":"","num_papers_in_archive":20},{"url":"/dataset/fashion-144k","name":"Fashion 144K","full_name":"","num_papers_in_archive":11},{"url":"/dataset/covid-19-twitter-chatter-dataset","name":"COVID-19 Twitter Chatter Dataset","full_name":"","num_papers_in_archive":10},{"url":"/dataset/oposum","name":"OpoSum","full_name":"","num_papers_in_archive":7},{"url":"/dataset/postnauka","name":"PostNauka","full_name":"","num_papers_in_archive":2},{"url":"/dataset/ruwiki-good","name":"RuWiki-Good","full_name":"","num_papers_in_archive":2},{"url":"/dataset/covid-19-tweets-with-motivation-and-topics","name":"COVID-19 Tweets with Motivation and Topics","full_name":"","num_papers_in_archive":1},{"url":"/dataset/mapping-topics-in-100000-real-life-moral","name":"Mapping Topics in 100,000 Real-Life Moral Dilemmas","full_name":"Tuan Dung nguyen","num_papers_in_archive":1},{"url":"/dataset/mie-articles-dataset-1996-2024","name":"MIE Articles Dataset (1996-2024)","full_name":"","num_papers_in_archive":1},{"url":"/dataset/oagt","name":"OAGT","full_name":"Paper Topic Dataset","num_papers_in_archive":1},{"url":"/dataset/reddit-ideology-database","name":"Reddit Ideology Database","full_name":"","num_papers_in_archive":1},{"url":"/dataset/mie-articles-dataset","name":"MIE Articles Dataset","full_name":"","num_papers_in_archive":0}],"subtasks":[{"url":"/task/dynamic-topic-modeling","name":"Dynamic Topic Modeling"},{"url":"/task/topic-coverage","name":"Topic coverage"}],"parent_tasks":[{"url":"/task/text-classification","name":"Text Classification"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":229,"tagged_in_all":881,"items":[{"url":"/paper/topic-modeling-in-embedding-spaces","title":"Topic Modeling in Embedding Spaces","date":"2019-07-08","arxiv_id":"1907.04907","repositories_listed":12,"syntology":{"n":13,"n_ran":3,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/autoencoding-variational-inference-for-topic","title":"Autoencoding Variational Inference For Topic Models","date":"2017-03-04","arxiv_id":"1703.01488","repositories_listed":6,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":1}},{"url":"/paper/neural-variational-inference-for-text","title":"Neural Variational Inference for Text Processing","date":"2015-11-19","arxiv_id":"1511.06038","repositories_listed":6,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/mixing-dirichlet-topic-models-and-word","title":"Mixing Dirichlet Topic Models and Word Embeddings to Make lda2vec","date":"2016-05-06","arxiv_id":"1605.02019","repositories_listed":5,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/using-text-embeddings-for-causal-inference","title":"Adapting Text Embeddings for Causal Inference","date":"2019-05-29","arxiv_id":"1905.12741","repositories_listed":4,"syntology":null},{"url":"/paper/bertopic-neural-topic-modeling-with-a-class","title":"BERTopic: Neural topic modeling with a class-based TF-IDF procedure","date":"2022-03-11","arxiv_id":"2203.05794","repositories_listed":3,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/pre-training-is-a-hot-topic-contextualized","title":"Pre-training is a Hot Topic: Contextualized Document Embeddings Improve Topic Coherence","date":"2020-04-08","arxiv_id":"2004.03974","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/topic-discovery-in-massive-text-corpora-based","title":"Topic Discovery in Massive Text Corpora Based on Min-Hashing","date":"2018-07-03","arxiv_id":"1807.00938","repositories_listed":3,"syntology":null},{"url":"/paper/an-unsupervised-neural-attention-model-for","title":"An Unsupervised Neural Attention Model for Aspect Extraction","date":"2017-07-01","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/neural-models-for-documents-with-metadata","title":"Neural Models for Documents with Metadata","date":"2017-05-25","arxiv_id":"1705.09296","repositories_listed":3,"syntology":{"n":6,"n_ran":6,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/llm-reading-tea-leaves-automatically","title":"LLM Reading Tea Leaves: Automatically Evaluating Topic Models with Large Language Models","date":"2024-06-13","arxiv_id":"2406.09008","repositories_listed":2,"syntology":null},{"url":"/paper/fastopic-a-fast-adaptive-stable-and","title":"FASTopic: Pretrained Transformer is a Fast, Adaptive, Stable, and Transferable Topic Model","date":"2024-05-28","arxiv_id":"2405.17978","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/modeling-dynamic-topics-in-chain-free-fashion","title":"Modeling Dynamic Topics in Chain-Free Fashion by Evolution-Tracking Contrastive Learning and Unassociated Word Exclusion","date":"2024-05-28","arxiv_id":"2405.17957","repositories_listed":2,"syntology":null},{"url":"/paper/a-survey-on-neural-topic-models-methods","title":"A Survey on Neural Topic Models: Methods, Applications, and Challenges","date":"2024-01-27","arxiv_id":"2401.15351","repositories_listed":2,"syntology":null},{"url":"/paper/effective-neural-topic-modeling-with","title":"Effective Neural Topic Modeling with Embedding Clustering Regularization","date":"2023-06-07","arxiv_id":"2306.04217","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/infoctm-a-mutual-information-maximization","title":"InfoCTM: A Mutual Information Maximization Perspective of Cross-Lingual Topic Modeling","date":"2023-04-07","arxiv_id":"2304.03544","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/improving-neural-topic-models-with","title":"Improving Neural Topic Models with Wasserstein Knowledge Distillation","date":"2023-03-27","arxiv_id":"2303.15350","repositories_listed":2,"syntology":null},{"url":"/paper/improving-contextualized-topic-models-with","title":"Improving Contextualized Topic Models with Negative Sampling","date":"2023-03-27","arxiv_id":"2303.14951","repositories_listed":2,"syntology":null},{"url":"/paper/nonparametric-forest-structured-neural-topic","title":"Nonparametric Forest-Structured Neural Topic Modeling","date":"2022-10-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/principled-analysis-of-energy-discourse","title":"Principled Analysis of Energy Discourse across Domains with Thesaurus-based Automatic Topic Labeling","date":"2021-12-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/contrastive-learning-for-neural-topic-model","title":"Contrastive Learning for Neural Topic Model","date":"2021-10-25","arxiv_id":"2110.12764","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":2}},{"url":"/paper/phrase-bert-improved-phrase-embeddings-from","title":"Phrase-BERT: Improved Phrase Embeddings from BERT with an Application to Corpus Exploration","date":"2021-09-13","arxiv_id":"2109.06304","repositories_listed":2,"syntology":{"n":9,"n_ran":2,"n_unverified":7,"n_pointer_only":2}},{"url":"/paper/is-automated-topic-model-evaluation-broken","title":"Is Automated Topic Model Evaluation Broken?: The Incoherence of Coherence","date":"2021-07-05","arxiv_id":"2107.02173","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/using-transformer-based-ensemble-learning-to","title":"Using Transformer based Ensemble Learning to classify Scientific Articles","date":"2021-02-19","arxiv_id":"2102.09991","repositories_listed":2,"syntology":null},{"url":"/paper/short-text-topic-modeling-with-topic","title":"Short Text Topic Modeling with Topic Distribution Quantization and Negative Sampling Decoder","date":"2020-11-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/top2vec-distributed-representations-of-topics","title":"Top2Vec: Distributed Representations of Topics","date":"2020-08-19","arxiv_id":"2008.09470","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/cross-lingual-contextualized-topic-models","title":"Cross-lingual Contextualized Topic Models with Zero-shot Learning","date":"2020-04-16","arxiv_id":"2004.07737","repositories_listed":2,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/a-coefficient-of-determination-for","title":"A Coefficient of Determination for Probabilistic Topic Models","date":"2019-11-20","arxiv_id":"1911.11061","repositories_listed":2,"syntology":null},{"url":"/paper/190807599","title":"Learning document embeddings along with their uncertainties","date":"2019-08-20","arxiv_id":"1908.07599","repositories_listed":2,"syntology":null},{"url":"/paper/dirichlet-belief-networks-for-topic-structure","title":"Dirichlet belief networks for topic structure learning","date":"2018-11-02","arxiv_id":"1811.00717","repositories_listed":2,"syntology":null}],"syntology_records":15,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}