{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/topic-classification/papers/2","list_of":"/task/topic-classification","task":"Topic Classification","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,186],"of":186,"counts":{"archive_papers_tagged":186,"with_a_code_link":75,"where_syntology_ran_a_sample":8,"not_listed_spam_title":0,"listed":186,"listed_where_code_ran":8,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":5,"every_run_a_failure_of_syntologys_instrument":3,"listed_with_a_run_with_no_instrument_failure":5,"listed_every_run_a_failure_of_syntologys_instrument":3,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/topic-classification","prev":"/task/topic-classification","next":null,"papers":[{"url":null,"slug":"metroberta-leveraging-traditional-customer","title":"MetRoBERTa: Leveraging Traditional Customer Relationship Management Data to Develop a Transit-Topic-Aware Language Model","date":"2023-08-09","arxiv_id":"2308.05012","repositories_listed":0,"syntology":null},{"url":"/paper/vontss-vmf-based-semi-supervised-neural-topic","slug":"vontss-vmf-based-semi-supervised-neural-topic","title":"vONTSS: vMF based semi-supervised neural topic modeling with optimal transport","date":"2023-07-03","arxiv_id":"2307.01226","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-open-domain-topic-classification-1","title":"Towards Open-Domain Topic Classification","date":"2023-06-29","arxiv_id":"2306.17290","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-few-shot-learning-via-language","title":"Multilingual Few-Shot Learning via Language Model Retrieval","date":"2023-06-19","arxiv_id":"2306.10964","repositories_listed":0,"syntology":null},{"url":null,"slug":"monolingual-and-cross-lingual-knowledge","title":"Monolingual and Cross-Lingual Knowledge Transfer for Topic Classification","date":"2023-06-13","arxiv_id":"2306.07797","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-large-language-models-for-topic","title":"Leveraging Large Language Models for Topic Classification in the Domain of Public Affairs","date":"2023-06-05","arxiv_id":"2306.02864","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-document-embeddings-via-self","title":"Efficient Document Embeddings via Self-Contrastive Bregman Divergence Learning","date":"2023-05-25","arxiv_id":"2305.16031","repositories_listed":0,"syntology":null},{"url":null,"slug":"regex-augmented-domain-transfer-topic","title":"Regex-augmented Domain Transfer Topic Classification based on a Pre-trained Language Model: An application in Financial Domain","date":"2023-05-23","arxiv_id":"2305.18324","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-for-opinion-mining-and-topic","title":"Deep Learning for Opinion Mining and Topic Classification of Course Reviews","date":"2023-04-06","arxiv_id":"2304.03394","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-crowd-meets-persona-creating-a-large","title":"When Crowd Meets Persona: Creating a Large-Scale Open-Domain Persona Dialogue Corpus","date":"2023-04-01","arxiv_id":"2304.00350","repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-segmentation-model-focusing-on-local","title":"Topic Segmentation Model Focusing on Local Context","date":"2023-01-05","arxiv_id":"2301.01935","repositories_listed":0,"syntology":null},{"url":null,"slug":"qbert-generalist-model-for-processing","title":"QBERT: Generalist Model for Processing Questions","date":"2022-12-05","arxiv_id":"2212.01967","repositories_listed":0,"syntology":null},{"url":null,"slug":"tcbert-a-technical-report-for-chinese-topic","title":"TCBERT: A Technical Report for Chinese Topic Classification BERT","date":"2022-11-21","arxiv_id":"2211.11304","repositories_listed":0,"syntology":null},{"url":null,"slug":"ccprompt-counterfactual-contrastive-prompt","title":"CCPrefix: Counterfactual Contrastive Prefix-Tuning for Many-Class Classification","date":"2022-11-11","arxiv_id":"2211.05987","repositories_listed":0,"syntology":null},{"url":null,"slug":"conformal-predictor-for-improving-zero-shot","title":"Conformal Predictor for Improving Zero-shot Text Classification Efficiency","date":"2022-10-23","arxiv_id":"2210.12619","repositories_listed":0,"syntology":null},{"url":null,"slug":"twitter-topic-classification","title":"Twitter Topic Classification","date":"2022-09-20","arxiv_id":"2209.09824","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-parametric-temporal-adaptation-for-social","title":"Non-Parametric Temporal Adaptation for Social Media Topic Classification","date":"2022-09-13","arxiv_id":"2209.05706","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctm-a-model-for-large-scale-multi-view-tweet-2","title":"CTM - A Model for Large-Scale Multi-View Tweet Topic Classification","date":"2022-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"realistic-zero-shot-cross-lingual-transfer-in-1","title":"Realistic Zero-Shot Cross-Lingual Transfer in Legal Topic Classification","date":"2022-06-08","arxiv_id":"2206.03785","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-term-advances-in-quantum-natural","title":"Near-Term Advances in Quantum Natural Language Processing","date":"2022-06-05","arxiv_id":"2206.02171","repositories_listed":0,"syntology":null},{"url":null,"slug":"estimating-confidence-of-predictions-of-1","title":"Estimating Confidence of Predictions of Individual Classifiers and TheirEnsembles for the Genre Classification Task","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-short-text-classification-with","title":"Improving Short Text Classification With Augmented Data Using GPT-3","date":"2022-05-23","arxiv_id":"2205.10981","repositories_listed":0,"syntology":null},{"url":null,"slug":"sentence-level-privacy-for-document-1","title":"Sentence-level Privacy for Document Embeddings","date":"2022-05-10","arxiv_id":"2205.04605","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctm-a-model-for-large-scale-multi-view-tweet-1","title":"CTM -- A Model for Large-Scale Multi-View Tweet Topic Classification","date":"2022-05-03","arxiv_id":"2205.01603","repositories_listed":0,"syntology":null},{"url":null,"slug":"clause-topic-classification-in-german-and","title":"Clause Topic Classification in German and English Standard Form Contracts","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-label-topic-classification-for-covid-19","title":"Multi-label topic classification for COVID-19 literature with Bioformer","date":"2022-04-14","arxiv_id":"2204.06758","repositories_listed":0,"syntology":null},{"url":null,"slug":"realistic-zero-shot-cross-lingual-transfer-in","title":"Realistic Zero-Shot Cross-Lingual Transfer in Legal Topic Classification","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unlearnable-text-for-neural-classifiers","title":"Unlearnable Text for Neural Classifiers","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multilingual-bag-of-entities-model-for-zero-1","title":"A Multilingual Bag-of-Entities Model for Zero-Shot Cross-Lingual Text Classification","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ctm-a-model-for-large-scale-multi-view-tweet","title":"CTM - A Model for Large-Scale Multi-View Tweet Topic Classification","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"prototypical-verbalizer-for-prompt-based-few","title":"Prototypical Verbalizer for Prompt-based Few-shot Tuning","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sentence-level-privacy-for-document","title":"Sentence-level Privacy for Document Embeddings","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"using-listeners-interpretations-in-topic","title":"Using Listeners’ Interpretations in Topic Classification of Song Lyrics","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multilingual-bag-of-entities-model-for-zero","title":"A Multilingual Bag-of-Entities Model for Zero-Shot Cross-Lingual Text Classification","date":"2021-10-15","arxiv_id":"2110.07792","repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-classification-on-spoken-documents","title":"Topic Classification on Spoken Documents Using Deep Acoustic and Linguistic Features","date":"2021-06-16","arxiv_id":"2106.08637","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-emotional-comfort-framework-for-improving","title":"An Emotional Comfort Framework for Improving User Satisfaction in E-Commerce Customer Service Chatbots","date":"2021-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mathbert-a-pre-trained-model-for-mathematical","title":"MathBERT: A Pre-Trained Model for Mathematical Formula Understanding","date":"2021-05-02","arxiv_id":"2105.00377","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-classes-clusters","title":"Are Classes Clusters?","date":"2021-04-16","arxiv_id":"2104.07840","repositories_listed":0,"syntology":null},{"url":null,"slug":"shuftext-a-simple-black-box-approach-to","title":"ShufText: A Simple Black Box Approach to Evaluate the Fragility of Text Classification Models","date":"2021-01-30","arxiv_id":"2102.00238","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-synthetic-oversampling","title":"A Comparison of Synthetic Oversampling Methods for Multi-class Text Classification","date":"2020-08-11","arxiv_id":"2008.04636","repositories_listed":0,"syntology":null},{"url":null,"slug":"variants-of-bert-random-forests-and-svm","title":"Variants of BERT, Random Forests and SVM approach for Multimodal Emotion-Target Sub-challenge","date":"2020-07-28","arxiv_id":"2007.13928","repositories_listed":0,"syntology":null},{"url":null,"slug":"aws-cord19-search-a-scientific-literature","title":"AWS CORD-19 Search: A Neural Search Engine for COVID-19 Literature","date":"2020-07-17","arxiv_id":"2007.09186","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-language-modeling-enough-evaluating","title":"Is Language Modeling Enough? Evaluating Effective Embedding Combinations","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-language-dataset-creation","title":"Low resource language dataset creation, curation and classification: Setswana and Sepedi -- Extended Abstract","date":"2020-03-30","arxiv_id":"2004.13842","repositories_listed":0,"syntology":null},{"url":null,"slug":"ontology-extraction-and-usage-in-the","title":"Ontology Extraction and Usage in the Scholarly Knowledge Domain","date":"2020-03-27","arxiv_id":"2003.12611","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-an-approach-for-low-resource","title":"Investigating an approach for low resource language dataset creation, curation and classification: Setswana and Sepedi","date":"2020-02-18","arxiv_id":"2003.04986","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneous-identification-of-tweet-purpose","title":"Simultaneous Identification of Tweet Purpose and Position","date":"2019-12-24","arxiv_id":"2001.00051","repositories_listed":0,"syntology":null},{"url":null,"slug":"global-health-monitor-a-web-based-system-for","title":"Global Health Monitor: A Web-based System for Detecting and Mapping Infectious Diseases","date":"2019-11-21","arxiv_id":"1911.09735","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-versus-traditional-classifiers","title":"Deep Learning versus Traditional Classifiers on Vietnamese Students' Feedback Corpus","date":"2019-11-17","arxiv_id":"1911.07223","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-model-using-cross-task-embedding","title":"Multilingual Model Using Cross-Task Embedding Projection","date":"2019-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-news-recommendation-with-topic-aware","title":"Neural News Recommendation with Topic-Aware News Representation","date":"2019-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"low-dimensional-embodied-semantics-for-music","title":"Low-dimensional Embodied Semantics for Music and Language","date":"2019-06-20","arxiv_id":"1906.11759","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-strength-of-the-weakest-supervision-topic","title":"The Strength of the Weakest Supervision: Topic Classification Using Class Labels","date":"2019-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"190413213","title":"Topic Classification Method for Analyzing Effect of eWOM on Consumer Game Sales","date":"2019-04-23","arxiv_id":"1904.13213","repositories_listed":0,"syntology":null},{"url":null,"slug":"expanding-the-text-classification-toolbox","title":"Expanding the Text Classification Toolbox with Cross-Lingual Embeddings","date":"2019-03-23","arxiv_id":"1903.09878","repositories_listed":0,"syntology":null},{"url":null,"slug":"importance-of-self-attention-for-sentiment","title":"Importance of Self-Attention for Sentiment Analysis","date":"2018-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-topic-modeling-for-dialog-systems","title":"Contextual Topic Modeling For Dialog Systems","date":"2018-10-18","arxiv_id":"1810.08135","repositories_listed":0,"syntology":null},{"url":null,"slug":"super-characters-a-conversion-from-sentiment","title":"Super Characters: A Conversion from Sentiment Classification to Image Classification","date":"2018-10-15","arxiv_id":"1810.07653","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-opinion-topics-and-polarity-of","title":"Identifying Opinion-Topics and Polarity of Parliamentary Debate Motions","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ltv-labeled-topic-vector","title":"LTV: Labeled Topic Vector","date":"2018-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-contextual-topic-identification","title":"Low-Resource Contextual Topic Identification on Speech","date":"2018-07-17","arxiv_id":"1807.06204","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-of-representations-of-named","title":"Comparison of Representations of Named Entities for Document Classification","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"illustrative-language-understanding-large","title":"Illustrative Language Understanding: Large-Scale Visual Grounding with Image Search","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"jointly-trained-sequential-labeling-and","title":"Jointly Trained Sequential Labeling and Classification by Sparse Attention Neural Networks","date":"2017-09-28","arxiv_id":"1709.10191","repositories_listed":0,"syntology":null},{"url":null,"slug":"initializing-convolutional-filters-with","title":"Initializing Convolutional Filters with Semantic Features for Text Classification","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"data-sets-word-embeddings-learned-from-tweets","title":"Data Sets: Word Embeddings Learned from Tweets and General Data","date":"2017-08-14","arxiv_id":"1708.03994","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-classification-of-topics-in","title":"Cross-Lingual Classification of Topics in Political Texts","date":"2017-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"distant-supervision-for-topic-classification","title":"Distant Supervision for Topic Classification of Tweets in Curated Streams","date":"2017-04-22","arxiv_id":"1704.06726","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-and-off-topic-classification-and-semantic","title":"On- and Off-Topic Classification and Semantic Annotation of User-Generated Software Requirements","date":"2016-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-policy-agendas-lessons-learned","title":"Analysis of Policy Agendas: Lessons Learned from Automatic Topic Classification of Croatian Political Texts","date":"2016-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"intra-topic-variability-normalization-based","title":"Intra-Topic Variability Normalization based on Linear Projection for Topic Classification","date":"2016-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"longitudinal-analysis-of-discussion-topics-in","title":"Longitudinal Analysis of Discussion Topics in an Online Breast Cancer Community using Convolutional Neural Networks","date":"2016-03-28","arxiv_id":"1603.08458","repositories_listed":0,"syntology":null},{"url":null,"slug":"embracing-error-to-enable-rapid-crowdsourcing","title":"Embracing Error to Enable Rapid Crowdsourcing","date":"2016-02-14","arxiv_id":"1602.04506","repositories_listed":0,"syntology":null},{"url":null,"slug":"co-training-for-topic-classification-of","title":"Co-Training for Topic Classification of Scholarly Data","date":"2015-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/semi-supervised-convolutional-neural-networks-1","slug":"semi-supervised-convolutional-neural-networks-1","title":"Semi-supervised Convolutional Neural Networks for Text Categorization via Region Embedding","date":"2015-04-06","arxiv_id":"1504.01255","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-optimization-of-text-representations","title":"Bayesian Optimization of Text Representations","date":"2015-03-02","arxiv_id":"1503.00693","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-frame-identification-with","title":"Semantic Frame Identification with Distributed Word Representations","date":"2014-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-lda-based-topic-classification-approach","title":"A LDA-Based Topic Classification Approach From Highly Imperfect Automatic Transcriptions","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"nomad-linguistic-resources-and-tools-aimed-at","title":"NOMAD: Linguistic Resources and Tools Aimed at Policy Formulation and Validation","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-anatomy-of-a-modular-system-for-media","title":"The Anatomy of a Modular System for Media Content Analysis","date":"2014-02-25","arxiv_id":"1402.6208","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-high-reproducibility-and-high-accuracy","title":"A high-reproducibility and high-accuracy method for automated topic classification","date":"2014-02-03","arxiv_id":"1402.0422","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-text-and-link-analysis-with-mixed","title":"Scalable Text and Link Analysis with Mixed-Topic Link Models","date":"2013-03-28","arxiv_id":"1303.7264","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-semi-supervised-bayesian-network-model-for","title":"A Semi-Supervised Bayesian Network Model for Microblog Topic Classification","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-topic-classification-for-highly","title":"Improving Topic Classification for Highly Inflective Languages","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"baselines-and-bigrams-simple-good-sentiment","title":"Baselines and Bigrams: Simple, Good Sentiment and Topic Classification","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-classification-of-blog-posts-using","title":"Topic Classification of Blog Posts Using Distant Supervision","date":"2012-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"74100fd00428a828960445dc3d457898aca613fac3a7701295fbc2ab26809be4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}