{"url":"/task/topic-classification","name":"Topic Classification","slug":"topic-classification","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":186,"papers_with_code":75,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":10,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/yahoo-answers","name":"Yahoo! Answers","full_name":"","num_papers_in_archive":132},{"url":"/dataset/how2sign","name":"How2Sign","full_name":"A Large-scale Multimodal Dataset for Continuous American Sign Language","num_papers_in_archive":44},{"url":"/dataset/amazon-product-data","name":"Amazon Product Data","full_name":"rithik","num_papers_in_archive":41},{"url":"/dataset/klue","name":"KLUE","full_name":"Korean Language Understanding Evaluation","num_papers_in_archive":21},{"url":"/dataset/20newsgroup-10-tasks","name":"20Newsgroup (10 tasks)","full_name":"","num_papers_in_archive":11},{"url":"/dataset/taiga-corpus","name":"Taiga Corpus","full_name":"An open-source corpus for machine learning.","num_papers_in_archive":5},{"url":"/dataset/lleqa","name":"LLeQA","full_name":"Long-form Legal Question Answering","num_papers_in_archive":3},{"url":"/dataset/latamxix","name":"LatamXIX","full_name":"19th Century Latin American Spanish Newspaper Corpus with LLM OCR Correction","num_papers_in_archive":2},{"url":"/dataset/mapping-topics-in-100000-real-life-moral","name":"Mapping Topics in 100,000 Real-Life Moral Dilemmas","full_name":"Tuan Dung nguyen","num_papers_in_archive":1},{"url":"/dataset/spanish-corpus-xix","name":"Spanish Corpus XIX","full_name":"19th Century Spanish Corpus","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":75,"tagged_in_all":186,"items":[{"url":"/paper/active-learning-in-annotating-micro-blogs","title":"Active learning in annotating micro-blogs dealing with e-reputation","date":"2017-06-16","arxiv_id":"1706.05349","repositories_listed":5,"syntology":null},{"url":"/paper/klue-korean-language-understanding-evaluation","title":"KLUE: Korean Language Understanding Evaluation","date":"2021-05-20","arxiv_id":"2105.09680","repositories_listed":4,"syntology":null},{"url":"/paper/entailment-as-few-shot-learner","title":"Entailment as Few-Shot Learner","date":"2021-04-29","arxiv_id":"2104.14690","repositories_listed":3,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/hierarchical-transformers-for-long-document","title":"Hierarchical Transformers for Long Document Classification","date":"2019-10-23","arxiv_id":"1910.10781","repositories_listed":3,"syntology":null},{"url":"/paper/lexc-gen-generating-data-for-extremely-low","title":"LexC-Gen: Generating Data for Extremely Low-Resource Languages with Large Language Models and Bilingual Lexicons","date":"2024-02-21","arxiv_id":"2402.14086","repositories_listed":2,"syntology":null},{"url":"/paper/sib-200-a-simple-inclusive-and-big-evaluation","title":"SIB-200: A Simple, Inclusive, and Big Evaluation Dataset for Topic Classification in 200+ Languages and Dialects","date":"2023-09-14","arxiv_id":"2309.07445","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/intermediate-training-on-question-answering","title":"Leveraging QA Datasets to Improve Generative Data Augmentation","date":"2022-05-25","arxiv_id":"2205.12604","repositories_listed":2,"syntology":null},{"url":"/paper/controlling-the-interaction-between","title":"Controlling the Interaction Between Generation and Inference in Semi-Supervised Variational Autoencoders Using Importance Weighting","date":"2020-10-13","arxiv_id":"2010.06549","repositories_listed":2,"syntology":null},{"url":"/paper/cross-lingual-adaptation-using-structural","title":"Cross-Lingual Adaptation using Structural Correspondence Learning","date":"2010-08-04","arxiv_id":"1008.0716","repositories_listed":2,"syntology":null},{"url":"/paper/a-multi-task-benchmark-for-abusive-language","title":"A Multi-Task Benchmark for Abusive Language Detection in Low-Resource Settings","date":"2025-05-17","arxiv_id":"2505.12116","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11177","title":"Low-Resource Language Processing: An OCR-Driven Summarization and Translation Pipeline","date":"2025-05-16","arxiv_id":"2505.11177","repositories_listed":1,"syntology":null},{"url":"/paper/a-thorough-benchmark-of-automatic-text","title":"A thorough benchmark of automatic text classification: From traditional approaches to large language models","date":"2025-04-02","arxiv_id":"2504.01930","repositories_listed":1,"syntology":null},{"url":"/paper/reading-the-unreadable-creating-a-dataset-of","title":"Reading the unreadable: Creating a dataset of 19th century English newspapers using image-to-text language models","date":"2025-02-18","arxiv_id":"2502.14901","repositories_listed":1,"syntology":null},{"url":"/paper/llm-teacher-student-framework-for-text","title":"LLM Teacher-Student Framework for Text Classification With No Manually Annotated Data: A Case Study in IPTC News Topic Classification","date":"2024-11-29","arxiv_id":"2411.19638","repositories_listed":1,"syntology":null},{"url":"/paper/quickcharnet-an-efficient-url-classification","title":"QuickCharNet: An Efficient URL Classification Framework for Enhanced Search Engine Optimization","date":"2024-10-22","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/inference-and-verbalization-functions-during","title":"Inference and Verbalization Functions During In-Context Learning","date":"2024-10-12","arxiv_id":"2410.09349","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/lowrem-a-repository-of-word-embeddings-for-87","title":"GrEmLIn: A Repository of Green Baseline Embeddings for 87 Low-Resource Languages Injected with Multilingual Graph Knowledge","date":"2024-09-26","arxiv_id":"2409.18193","repositories_listed":1,"syntology":null},{"url":"/paper/2408-01969","title":"Optimal and efficient text counterfactuals using Graph Neural Networks","date":"2024-08-04","arxiv_id":"2408.01969","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-classification-of-news-subjects-in","title":"Automatic Classification of News Subjects in Broadcast News: Application to a Gender Bias Representation Analysis","date":"2024-07-19","arxiv_id":"2407.14180","repositories_listed":1,"syntology":null},{"url":"/paper/retrieval-augmented-zero-shot-text","title":"Retrieval Augmented Zero-Shot Text Classification","date":"2024-06-21","arxiv_id":"2406.15241","repositories_listed":1,"syntology":null},{"url":"/paper/newswire-a-large-scale-structured-database-of","title":"Newswire: A Large-Scale Structured Database of a Century of Historical News","date":"2024-06-13","arxiv_id":"2406.09490","repositories_listed":1,"syntology":null},{"url":"/paper/topic-modelling-case-law-using-a-large","title":"Topic Classification of Case Law Using a Large Language Model and a New Taxonomy for UK Law: AI Insights into Summary Judgment","date":"2024-05-21","arxiv_id":"2405.12910","repositories_listed":1,"syntology":null},{"url":"/paper/synthesizrr-generating-diverse-datasets-with","title":"SynthesizRR: Generating Diverse Datasets with Retrieval Augmentation","date":"2024-05-16","arxiv_id":"2405.10040","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/addressing-topic-granularity-and","title":"Addressing Topic Granularity and Hallucination in Large Language Models for Topic Modelling","date":"2024-05-01","arxiv_id":"2405.00611","repositories_listed":1,"syntology":null},{"url":"/paper/what-drives-performance-in-multilingual","title":"What Drives Performance in Multilingual Language Models?","date":"2024-04-29","arxiv_id":"2404.19159","repositories_listed":1,"syntology":null},{"url":"/paper/l3cube-mahanews-news-based-short-text-and","title":"L3Cube-MahaNews: News-based Short Text and Long Document Classification Datasets in Marathi","date":"2024-04-28","arxiv_id":"2404.18216","repositories_listed":1,"syntology":null},{"url":"/paper/forget-nli-use-a-dictionary-zero-shot-topic","title":"Forget NLI, Use a Dictionary: Zero-Shot Topic Classification for Low-Resource Languages with Application to Luxembourgish","date":"2024-04-05","arxiv_id":"2404.03912","repositories_listed":1,"syntology":null},{"url":"/paper/text-classification-of-column-headers-with-a","title":"Zero-Shot Topic Classification of Column Headers: Leveraging LLMs for Metadata Enrichment","date":"2024-03-01","arxiv_id":"2403.00884","repositories_listed":1,"syntology":null},{"url":"/paper/l3cube-indicnews-news-based-short-text-and","title":"L3Cube-IndicNews: News-based Short Text and Long Document Classification Datasets in Indic Languages","date":"2024-01-04","arxiv_id":"2401.02254","repositories_listed":1,"syntology":null},{"url":"/paper/draft-dense-retrieval-augmented-few-shot","title":"DRAFT: Dense Retrieval Augmented Few-shot Topic classifier Framework","date":"2023-12-05","arxiv_id":"2312.02532","repositories_listed":1,"syntology":null}],"syntology_records":4,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}