{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-identification/papers/8","list_of":"/task/language-identification","task":"Language Identification","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":8,"rows_per_page":100,"rows":[701,794],"of":794,"counts":{"archive_papers_tagged":794,"with_a_code_link":143,"where_syntology_ran_a_sample":14,"not_listed_spam_title":0,"listed":794,"listed_where_code_ran":14,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":10,"every_run_a_failure_of_syntologys_instrument":4,"listed_with_a_run_with_no_instrument_failure":10,"listed_every_run_a_failure_of_syntologys_instrument":4,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-identification","prev":"/task/language-identification/papers/7","next":null,"papers":[{"url":null,"slug":"pos-tagging-of-english-hindi-code-mixed","title":"POS Tagging of English-Hindi Code-Mixed Social Media Content","date":"2014-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-code-switching-in-multilingual","title":"Predicting Code-switching in Multilingual Communication for Immigrant Communities","date":"2014-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-cmu-submission-for-the-shared-task-on","title":"The CMU Submission for the Shared Task on Language Identification in Code-Switched Data","date":"2014-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-iucl-system-word-level-language","title":"The IUCL+ System: Word-Level Language Identification via Extended Markov Models","date":"2014-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"word-level-language-identification-using-crf","title":"Word-level Language Identification using CRF: Code-switching Shared Task Report of MSR India System","date":"2014-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-report-on-the-dsl-shared-task-2014","title":"A Report on the DSL Shared Task 2014","date":"2014-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"experiments-in-sentence-language","title":"Experiments in Sentence Language Identification with Groups of Similar Languages","date":"2014-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-methods-and-resources-for","title":"Exploring Methods and Resources for Discriminating Similar Languages","date":"2014-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-syntactic-features-for-native","title":"Exploring Syntactic Features for Native Language Identification: A Variationist Perspective on Feature Encoding and Ensemble Optimization","date":"2014-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"language-family-relationship-preserved-in-non","title":"Language Family Relationship Preserved in Non-native English","date":"2014-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"subsegmental-language-detection-in-celtic","title":"Subsegmental language detection in Celtic language text","date":"2014-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-nrc-system-for-discriminating-similar","title":"The NRC System for Discriminating Similar Languages","date":"2014-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"using-maximum-entropy-models-to-discriminate","title":"Using Maximum Entropy Models to Discriminate between Similar Languages and Varieties","date":"2014-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-phrase-structure-learning-methods","title":"A survey on phrase structure learning methods for text classification","date":"2014-06-21","arxiv_id":"1406.5598","repositories_listed":0,"syntology":null},{"url":null,"slug":"dkpro-tc-a-java-based-framework-for","title":"DKPro TC: A Java-based Framework for Supervised Learning Experiments on Textual Data","date":"2014-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"does-the-phonology-of-l1-show-up-in-l2-texts","title":"Does the Phonology of L1 Show Up in L2 Texts?","date":"2014-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"short-term-projects-long-term-benefits-four","title":"Short-Term Projects, Long-Term Benefits: Four Student NLP Projects for Low-Resource Languages","date":"2014-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-feature-learning-for-visual-sign","title":"Unsupervised Feature Learning for Visual Sign Language Identification","date":"2014-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-language-identification-using-deep","title":"AUTOMATIC LANGUAGE IDENTIFICATION USING DEEP NEURAL NETWORKS","date":"2014-05-04","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-language-identity-tagging-on-word","title":"Automatic language identity tagging on word and sentence-level in multilingual text sources: a case-study on Luxembourgish","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"facing-the-identification-problem-in-language","title":"Facing the Identification Problem in Language-Related Scientific Data Analysis.","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-romanized-arabic-dialect-in-code","title":"Finding Romanized Arabic Dialect in Code-Mixed Tweets","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"globalphone-pronunciation-dictionaries-in-20","title":"GlobalPhone: Pronunciation Dictionaries in 20 Languages","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-exploitation-of-linguistic","title":"Improving the exploitation of linguistic annotations in ELAN","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"koko-an-l1-learner-corpus-for-german","title":"KoKo: an L1 Learner Corpus for German","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"native-language-identification-using-large-1","title":"Native Language Identification Using Large, Longitudinal Data","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"statistical-analysis-of-multilingual-text","title":"Statistical Analysis of Multilingual Text Corpus and Development of Language Models","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"swissadmin-a-multilingual-tagged-parallel","title":"SwissAdmin: A multilingual tagged parallel corpus of press releases","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-merlin-corpus-learner-language-and-the","title":"The MERLIN corpus: Learner language and the CEFR","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-rats-collection-supporting-hlt-research","title":"The RATS Collection: Supporting HLT Research with Degraded Audio Data","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tlaxcala-a-multilingual-corpus-of-independent","title":"TLAXCALA: a multilingual corpus of independent news","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"varclass-an-open-source-language","title":"VarClass: An Open-source Language Identification Tool for Language Varieties","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"vocabulary-based-language-similarity-using","title":"Vocabulary-Based Language Similarity using Web Corpora","date":"2014-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"accurate-language-identification-of-twitter","title":"Accurate Language Identification of Twitter Messages","date":"2014-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrapping-a-historical-commodities","title":"Bootstrapping a historical commodities lexicon with SKOS and DBpedia","date":"2014-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bshrsrwac-web-corpora-of-bosnian-croatian-and","title":"bs,hr,srWaC - Web Corpora of Bosnian, Croatian and Serbian","date":"2014-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"chinese-native-language-identification","title":"Chinese Native Language Identification","date":"2014-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-detection-and-language","title":"Automatic Detection and Language Identification of Multilingual Documents","date":"2014-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-identification-of-learners-language","title":"Automatic Identification of Learners' Language Background Based on Their Writing in Czech","date":"2013-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"word-level-language-identification-in-online","title":"Word Level Language Identification in Online Multilingual Communication","date":"2013-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"linguistic-profiling-of-texts-across-textual","title":"Linguistic Profiling of Texts Across Textual Genres and Readability Levels. An Exploratory Study on Italian Fictional Prose","date":"2013-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"text-segmentation-for-language-identification","title":"Text segmentation for Language Identification in Greek Forums","date":"2013-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-mysterious-letter-j","title":"The Mysterious Letter J","date":"2013-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"categorization-of-turkish-news-documents-with","title":"Categorization of Turkish News Documents with Morphological Analysis","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reconstructing-an-indo-european-family-tree","title":"Reconstructing an Indo-European Family Tree from Non-native English Texts","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"wordnet-based-cross-language-identification","title":"Wordnet-Based Cross-Language Identification of Semantic Relations","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-report-on-the-first-native-language","title":"A Report on the First Native Language Identification Shared Task","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cognate-and-misspelling-features-for-natural","title":"Cognate and Misspelling Features for Natural Language Identification","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-shallow-and-linguistically","title":"Combining Shallow and Linguistically Motivated Features in Native Language Identification","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"discriminating-non-native-english-with-350","title":"Discriminating Non-Native English with 350 Words","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"experimental-results-on-the-native-language","title":"Experimental Results on the Native Language Identification Shared Task","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-syntactic-representations-for","title":"Exploring Syntactic Representations for Native Language Identification","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"extracting-the-native-language-signal-for","title":"Extracting the Native Language Signal for Second Language Acquisition","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-engineering-in-the-nli-shared-task","title":"Feature Engineering in the NLI Shared Task 2013: Charles University Submission Report","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-space-selection-and-combination-for","title":"Feature Space Selection and Combination for Native Language Identification","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"from-language-to-family-and-back-native","title":"From Language to Family and Back: Native Language and Language Family Identification from English Text","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-native-language-identification-with","title":"Improving Native Language Identification with TF-IDF Weighting","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"labeling-the-languages-of-words-in-mixed","title":"Labeling the Languages of Words in Mixed-Language Documents using Weakly Supervised Methods","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"limsis-participation-to-the-2013-shared-task","title":"LIMSI's participation to the 2013 shared task on Native Language Identification","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"linguistic-profiling-based-on-generalpurpose","title":"Linguistic Profiling based on General--purpose Features and Native Language Identification","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"maximizing-classification-accuracy-in-native","title":"Maximizing Classification Accuracy in Native Language Identification","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"naist-at-the-nli-2013-shared-task","title":"NAIST at the NLI 2013 Shared Task","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"native-language-identification-a-key-n-gram","title":"Native Language Identification: A Key N-gram Category Approach","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"native-language-identification-a-simple-n","title":"Native Language Identification: a Simple n-gram Based Approach","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"native-language-identification-using-large","title":"Native Language Identification using large scale lexical features","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"native-language-identification-with-ppm","title":"Native Language Identification with PPM","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"nli-shared-task-2013-mq-submission","title":"NLI Shared Task 2013: MQ Submission","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"recognizing-english-learners-native-language","title":"Recognizing English Learners' Native Language from Their Writings","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-yet-powerful-native-language","title":"Simple Yet Powerful Native Language Identification on TOEFL11","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-story-of-the-characters-the-dna-and-the","title":"The Story of the Characters, the DNA and the Native Language","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"using-n-gram-and-word-network-features-for","title":"Using N-gram and Word Network Features for Native Language Identification","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"using-other-learner-corpora-in-the-2013-nli","title":"Using Other Learner Corpora in the 2013 NLI Shared Task","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"vtex-system-description-for-the-nli-2013","title":"VTEX System Description for the NLI 2013 Shared Task","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-discrimination-between-closely","title":"Efficient Discrimination Between Closely Related Languages","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"native-language-identification-using","title":"Native Language Identification using Recurring $n$-grams -- Investigating Abstraction and Domain Dependence","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"native-tongues-lost-and-found-resources-and","title":"Native Tongues, Lost and Found: Resources and Empirical Evaluations in Native Language Identification","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-lexicalized-native-language","title":"Robust, Lexicalized Native Language Identification","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"beefmoves-dissemination-diversity-and","title":"Beefmoves: Dissemination, Diversity, and Dynamics of English Borrowings in a German Hip Hop Forum","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"characterizing-stylistic-elements-in","title":"Characterizing Stylistic Elements in Syntactic Structure","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-adaptor-grammars-for-native","title":"Exploring Adaptor Grammars for Native Language Identification","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"langidpy-an-off-the-shelf-language","title":"langid.py: An Off-the-shelf Language Identification Tool","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"language-identification-for-creating-language","title":"Language Identification for Creating Language-Specific Twitter Collections","date":"2012-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reference-scope-identification-in-citing","title":"Reference Scope Identification in Citing Sentences","date":"2012-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"vers-la-correction-automatique-de-textes","title":"Vers la correction automatique de textes bruit\\'es: Architecture g\\'en\\'erale et d\\'etermination de la langue d'un mot inconnu (Towards Automatic Spell-Checking of Noisy Texts : General Architecture and Language Identification for Unknown Words) [in French]","date":"2012-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-mandarin-english-code-switching-corpus","title":"A Mandarin-English Code-Switching Corpus","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"development-of-text-and-speech-database-for","title":"Development of Text and Speech database for Hindi and Indian English specific to Mobile Communication environment","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-lexical-analysis","title":"Large Scale Lexical Analysis","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"linguagrid-a-network-of-linguistic-and","title":"Linguagrid: a network of Linguistic and Semantic Services for the Italian Language.","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-interlanguage-native-language","title":"Measuring Interlanguage: Native Language Identification with L1-influence Metrics","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"query-log-analysis-with-langlog","title":"Query log analysis with LangLog","date":"2012-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"yet-another-language-identifier","title":"Yet Another Language Identifier","date":"2012-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-domain-feature-selection-for-language","title":"Cross-domain Feature Selection for Language Identification","date":"2011-11-11","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-spoken-language-identification","title":"Automatic Spoken Language Identification Utilizing Acoustic and Phonetic Speech Information","date":"2004-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-language-identification","title":"Automatic language identification","date":"2001-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"db314ec78b0f008696ee5c73d3baccdff8ec0875431f46636bcac385754f7783","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}