{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-categorization/papers/3","list_of":"/task/text-categorization","task":"Text Categorization","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":3,"rows_per_page":100,"rows":[201,247],"of":247,"counts":{"archive_papers_tagged":247,"with_a_code_link":44,"where_syntology_ran_a_sample":0,"not_listed_spam_title":0,"listed":247,"listed_where_code_ran":0,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":0,"every_run_a_failure_of_syntologys_instrument":0,"listed_with_a_run_with_no_instrument_failure":0,"listed_every_run_a_failure_of_syntologys_instrument":0,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-categorization","prev":"/task/text-categorization/papers/2","next":null,"papers":[{"url":null,"slug":"collexen-automatically-generating-and","title":"ColLex.en: Automatically Generating and Evaluating a Full-form Lexicon for English","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"data-mining-with-shallow-vs-linguistic","title":"Data Mining with Shallow vs. Linguistic Features to Study Diversification of Scientific Registers","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dense-components-in-the-structure-of-wordnet","title":"Dense Components in the Structure of WordNet","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mapping-wordnet-domains-wordnet-topics-and","title":"Mapping WordNet Domains, WordNet Topics and Wikipedia Categories to Generate Multilingual Domain Specific Resources","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sentube-a-corpus-for-sentiment-analysis-on","title":"SenTube: A Corpus for Sentiment Analysis on YouTube Social Media","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"varclass-an-open-source-language","title":"VarClass: An Open-source Language Identification Tool for Language Varieties","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"wikipedia-based-semantic-interpretation-for","title":"Wikipedia-based Semantic Interpretation for Natural Language Processing","date":"2014-01-15","arxiv_id":"1401.5697","repositories_listed":0,"syntology":null},{"url":null,"slug":"creation-of-lexical-relations-for-indowordnet","title":"Creation of Lexical Relations for IndoWordNet","date":"2014-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"compressive-feature-learning","title":"Compressive Feature Learning","date":"2013-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-corpora-construction-for-text","title":"Automatic Corpora Construction for Text Classification","date":"2013-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-bregman-variational-dual-tree-framework","title":"The Bregman Variational Dual-Tree Framework","date":"2013-09-26","arxiv_id":"1309.6812","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-language-plagiarism-detection-methods","title":"Cross-Language Plagiarism Detection Methods","date":"2013-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"text-segmentation-for-language-identification","title":"Text segmentation for Language Identification in Greek Forums","date":"2013-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-basque-oral-poetry-analysis-a-machine","title":"Towards Basque Oral Poetry Analysis: A Machine Learning Approach","date":"2013-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-semi-supervised-approach-for-natural","title":"A Semi-supervised Approach for Natural Language Call Routing","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-languages-through-etymology-the-case","title":"Bridging Languages through Etymology: The case of cross language text categorization","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"categorization-of-turkish-news-documents-with","title":"Categorization of Turkish News Documents with Morphological Analysis","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"gpkex-genetically-programmed-keyphrase","title":"GPKEX: Genetically Programmed Keyphrase Extraction from Croatian Texts","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-semantic-matching-application-to-cross","title":"Latent Semantic Matching: Application to Cross-language Text Categorization without Alignment Information","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"text-classification-from-positive-and","title":"Text Classification from Positive and Unlabeled Data using Misclassified Data Correction","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"topicspam-a-topic-model-based-approach-for","title":"TopicSpam: a Topic-Model based approach for spam detection","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-examination-of-regret-in-bullying-tweets","title":"An Examination of Regret in Bullying Tweets","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-and-generic-text-categorization","title":"Cross-lingual and generic text categorization (Apprentissage d'une classification th\\'ematique g\\'en\\'erique et cross-langue \\`a partir des cat\\'egories de la Wikip\\'edia) [in French]","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deeppurple-lexical-string-and-affective","title":"DeepPurple: Lexical, String and Affective Feature Fusion for Sentence-Level Semantic Similarity Estimation","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ecnucs-measuring-short-text-semantic","title":"ECNUCS: Measuring Short Text Semantic Equivalence Using Multiple Similarity Measurements","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"from-high-heels-to-weed-attics-a-syntactic","title":"From high heels to weed attics: a syntactic investigation of chick lit and literature","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-term-informativeness-in-context","title":"Measuring Term Informativeness in Context","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"negative-deceptive-opinion-spam","title":"Negative Deceptive Opinion Spam","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-story-of-the-characters-the-dna-and-the","title":"The Story of the Characters, the DNA and the Native Language","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-selection-based-on-term-frequency-and","title":"Feature Selection Based on Term Frequency and T-Test for Text Categorization","date":"2013-05-03","arxiv_id":"1305.0638","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-similarity-computation-for-abstract","title":"Semantic Similarity Computation for Abstract and Concrete Nouns Using Network-based Distributional Semantic Models","date":"2013-03-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"baselines-and-bigrams-simple-good-sentiment","title":"Baselines and Bigrams: Simple, Good Sentiment and Topic Classification","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deeppurple-estimating-sentence-semantic","title":"DeepPurple: Estimating Sentence Semantic Similarity using N-gram Regression Models and Web Snippets","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-based-semi-supervised-learning","title":"Graph-based Semi-Supervised Learning Algorithms for NLP","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"langidpy-an-off-the-shelf-language","title":"langid.py: An Off-the-shelf Language Identification Tool","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-topic-dependencies-in-hierarchical","title":"Modeling Topic Dependencies in Hierarchical Text Categorization","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"state-of-the-art-kernels-for-natural-language","title":"State-of-the-Art Kernels for Natural Language Processing","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-building-a-multilingual-semantic","title":"Towards Building a Multilingual Semantic Network: Identifying Interlingual Links in Wikipedia","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"effect-of-small-sample-size-on-text","title":"Effect of small sample size on text categorization with support vector machines","date":"2012-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-good-space-lexical-predictors-in-word-space","title":"A good space: Lexical predictors in word space evaluation","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"french-and-german-corpora-for-audience-based","title":"French and German Corpora for Audience-based Text Type Classification","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-k-nearest-neighbor-efficacy-for","title":"Improving K-Nearest Neighbor Efficacy for Farsi Text Classification","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"irregularity-detection-in-categorized","title":"Irregularity Detection in Categorized Document Corpora","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"is-it-useful-to-support-users-with-lexical","title":"Is it Useful to Support Users with Lexical Resources? A User Study.","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"semsim-resources-for-normalized-semantic","title":"SemSim: Resources for Normalized Semantic Similarity Computation Using Lexical Networks","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/learning-from-multiple-partially-observed","slug":"learning-from-multiple-partially-observed","title":"Learning from Multiple Partially Observed Views - an Application to Multilingual Text Categorization","date":"2009-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-learning-with-weakly-related","title":"Semi-supervised Learning with Weakly-Related Unlabeled Data : Towards Better Text Categorization","date":"2008-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"70c3edf73641da8729e4227c3dd03a9c50c869157f6aa6e529e6ae116073cc58","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}