{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-clustering/papers/2","list_of":"/task/text-clustering","task":"Text Clustering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,123],"of":123,"counts":{"archive_papers_tagged":123,"with_a_code_link":38,"where_syntology_ran_a_sample":9,"not_listed_spam_title":0,"listed":123,"listed_where_code_ran":9,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":7,"every_run_a_failure_of_syntologys_instrument":2,"listed_with_a_run_with_no_instrument_failure":7,"listed_every_run_a_failure_of_syntologys_instrument":2,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-clustering","prev":"/task/text-clustering","next":null,"papers":[{"url":null,"slug":"robust-multi-relational-clustering-via-l1","title":"Robust Multi-Relational Clustering via $\\ell_1$-Norm Symmetric Nonnegative Matrix Factorization","date":"2015-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-method-of-accounting-bigrams-in-topic","title":"A Method of Accounting Bigrams in Topic Models","date":"2015-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-better-understanding-of-burrowss","title":"Towards a better understanding of Burrows's Delta in literary authorship attribution","date":"2015-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-models-accounting-component-structure","title":"Topic Models: Accounting Component Structure of Bigrams","date":"2015-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"experimental-estimation-of-number-of-clusters","title":"Experimental Estimation of Number of Clusters Based on Cluster Quality","date":"2015-03-10","arxiv_id":"1503.03168","repositories_listed":0,"syntology":null},{"url":null,"slug":"lmsim-computing-domain-specific-semantic-word","title":"LMSim : Computing Domain-specific Semantic Word Similarities Using a Language Modeling Approach","date":"2014-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"notes-on-using-determinantal-point-processes","title":"Notes on using Determinantal Point Processes for Clustering with Applications to Text Clustering","date":"2014-10-26","arxiv_id":"1410.6975","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-of-conversion-aided","title":"A Comparative Study of Conversion Aided Methods for WordNet Sentence Textual Similarity","date":"2014-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dirichlet-multinomial-mixture-model-based","title":"A Dirichlet Multinomial Mixture Model-based Approach for Short Text Clustering","date":"2014-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"clustering-tweets-usingwikipedia-concepts","title":"Clustering tweets usingWikipedia concepts","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"thematic-cohesion-measuring-terms","title":"Thematic Cohesion: measuring terms discriminatory power toward themes","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-evaluation-metrics-via-the","title":"Combining Evaluation Metrics via the Unanimous Improvement Ratio and its Application to Clustering Tasks","date":"2014-01-18","arxiv_id":"1401.4590","repositories_listed":0,"syntology":null},{"url":null,"slug":"which-clustering-do-you-want-inducing-your","title":"Which Clustering Do You Want? Inducing Your Ideal Clustering with Minimal Feedback","date":"2014-01-16","arxiv_id":"1401.5389","repositories_listed":0,"syntology":null},{"url":null,"slug":"post-retrieval-clustering-using-third-order","title":"Post-Retrieval Clustering Using Third-Order Similarity Measures","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-symmetric-rank-one-quasi-newton-method-for","title":"A Symmetric Rank-one Quasi Newton Method for Non-negative Matrix Factorization","date":"2013-05-24","arxiv_id":"1305.5829","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-template-based-hybrid-model-for-chinese","title":"A Template Based Hybrid Model for Chinese Personal Name Disambiguation","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-based-punjabi-text-document-clustering","title":"Domain Based Punjabi Text Document Clustering","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-discourse-relations-between","title":"Exploiting Discourse Relations between Sentences for Text Clustering","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-feature-rich-clustering","title":"Unsupervised Feature-Rich Clustering","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"subgroup-detection-in-ideological-discussions","title":"Subgroup Detection in Ideological Discussions","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-weighting-scheme-for-open-information","title":"A Weighting Scheme for Open Information Extraction","date":"2012-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cltc-a-chinese-english-cross-lingual-topic","title":"CLTC: A Chinese-English Cross-lingual Topic Corpus","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"qursim-a-corpus-for-evaluation-of-relatedness","title":"QurSim: A corpus for evaluation of relatedness in short texts","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"86c0b4369f83069967c80ef5e9ec170b61a3ccd8a83f802f10affb31ff26b146","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}