{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/lemmatization/papers/2","list_of":"/task/lemmatization","task":"Lemmatization","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":351,"counts":{"archive_papers_tagged":351,"with_a_code_link":68,"where_syntology_ran_a_sample":3,"not_listed_spam_title":0,"listed":351,"listed_where_code_ran":3,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2,"every_run_a_failure_of_syntologys_instrument":1,"listed_with_a_run_with_no_instrument_failure":2,"listed_every_run_a_failure_of_syntologys_instrument":1,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/lemmatization","prev":"/task/lemmatization","next":"/task/lemmatization/papers/3","papers":[{"url":null,"slug":"distant-reading-in-digital-humanities-case","title":"Distant Reading in Digital Humanities: Case Study on the Serbian Part of the ELTeC Collection","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"overview-of-the-evalatin-2022-evaluation","title":"Overview of the EvaLatin 2022 Evaluation Campaign","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tarc-tunisian-arabish-corpus-first-complete-1","title":"TArC: Tunisian Arabish Corpus, First complete release","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-the-creation-of-a-diachronic-corpus","title":"Towards the Creation of a Diachronic Corpus for Italian: A Case Study on the GDLI Quotations","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-based-part-of-speech-tagging-and","title":"Transformer-based Part-of-Speech Tagging and Lemmatization for Latin","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"zaebuc-an-annotated-arabic-english-bilingual","title":"ZAEBUC: An Annotated Arabic-English Bilingual Writer Corpus","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"abusive-and-threatening-language-detection-in-1","title":"Abusive and Threatening Language Detection in Urdu using Supervised Machine Learning and Feature Combinations","date":"2022-04-06","arxiv_id":"2204.03062","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-2021-urdu-fake-news-detection-task-using","title":"The 2021 Urdu Fake News Detection Task using Supervised Machine Learning and Feature Combinations","date":"2022-04-06","arxiv_id":"2204.03064","repositories_listed":0,"syntology":null},{"url":null,"slug":"supervised-and-unsupervised-categorization-of","title":"Supervised and Unsupervised Categorization of an Imbalanced Italian Crime News Dataset","date":"2022-03-22","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"biaffine-dependency-and-semantic-graph","title":"Biaffine Dependency and Semantic Graph Parsing for EnhancedUniversal Dependencies","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-mbert-based-seq2seq-enhanced","title":"End-to-end mBERT based Seq2seq Enhanced Dependency Parser with Linguistic Typology knowledge","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"linguistic-change-and-historical","title":"Linguistic change and historical periodization of Old Literary Finnish","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"morphological-analysis-corpus-construction-of","title":"Morphological Analysis Corpus Construction of Uyghur","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-reading-machine-a-versatile-framework-for","title":"The Reading Machine: A Versatile Framework for Studying Incremental Parsing Strategies","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pos-tagging-lemmatization-and-dependency","title":"POS tagging, lemmatization and dependency parsing of West Frisian","date":"2021-07-16","arxiv_id":"2107.07974","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-case-study-of-spanish-text-transformations","title":"A Case Study of Spanish Text Transformations for Twitter Sentiment Analysis","date":"2021-06-03","arxiv_id":"2106.02009","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-de-methodes-et-doutils-pour-la","title":"Évaluation de méthodes et d’outils pour la lemmatisation automatique du français médiéval (Evaluation of methods and tools for automatic lemmatization in Old French)","date":"2021-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"how-low-is-too-low-a-monolingual-take-on","title":"How low is too low? A monolingual take on lemmatisation in Indian languages","date":"2021-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-lemmatize-in-the-word","title":"Learning to Lemmatize in the Word Representation Space","date":"2021-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"named-entity-recognition-and-linking","title":"Named Entity Recognition and Linking Augmented with Large-Scale Structured Data","date":"2021-04-27","arxiv_id":"2104.13456","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effectiveness-of-dataset-embeddings-in","title":"On the Effectiveness of Dataset Embeddings in Mono-lingual,Multi-lingual and Zero-shot Conditions","date":"2021-03-01","arxiv_id":"2103.01273","repositories_listed":0,"syntology":null},{"url":null,"slug":"constraint-2021-machine-learning-models-for","title":"Constraint 2021: Machine Learning Models for COVID-19 Fake News Detection Shared Task","date":"2021-01-11","arxiv_id":"2101.03717","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysing-cross-lingual-transfer-in","title":"Analysing cross-lingual transfer in lemmatisation for Indian languages","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"recycling-and-comparing-morphological","title":"Recycling and Comparing Morphological Annotation Models for Armenian Diachronic-Variational Corpus Processing","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"utilizing-subword-entities-in-character-level","title":"Utilizing Subword Entities in Character-Level Sequence-to-Sequence Lemmatization Models","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cure-collection-for-urdu-information","title":"CURE: Collection for Urdu Information Retrieval Evaluation and Ranking","date":"2020-11-01","arxiv_id":"2011.00565","repositories_listed":0,"syntology":null},{"url":null,"slug":"lemmed-fast-and-effective-neural","title":"LemMED: Fast and Effective Neural Morphological Analysis with Short Context Windows","date":"2020-10-21","arxiv_id":"2010.10921","repositories_listed":0,"syntology":null},{"url":null,"slug":"turku-enhanced-parser-pipeline-from-raw-text","title":"Turku Enhanced Parser Pipeline: From Raw Text to Enhanced Graphs in the IWPT 2020 Shared Task","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"udpipe-at-evalatin-2020-contextualized-1","title":"UDPipe at EvaLatin 2020: Contextualized Embeddings and Treebank Embeddings","date":"2020-06-05","arxiv_id":"2006.03687","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-polysynthetic-language-modelling","title":"Neural Polysynthetic Language Modelling","date":"2020-05-11","arxiv_id":"2005.05477","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-gradient-boosting-seq2seq-system-for-latin","title":"A Gradient Boosting-Seq2Seq System for Latin POS Tagging and Lemmatization","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"babyfst-towards-a-finite-state-based","title":"BabyFST - Towards a Finite-State Based Computational Model of Ancient Babylonian","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"better-together-modern-methods-plus","title":"Better Together: Modern Methods Plus Traditional Thinking in NP Alignment","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"empirist-corpus-2-0-adding-manual","title":"EmpiriST Corpus 2.0: Adding Manual Normalization, Lemmatization and Semantic Tagging to a German Web and CMC Corpus","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"jhubc-s-submission-to-lt4hala-evalatin-2020","title":"JHUBC's Submission to LT4HALA EvaLatin 2020","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"lemmatization-and-pos-tagging-process-by","title":"Lemmatization and POS-tagging process by using joint learning approach. Experimental results on Classical Armenian, Old Georgian, and Syriac","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-learning-and-deep-neural-network","title":"Machine Learning and Deep Neural Network-Based Lemmatization and Morphosyntactic Tagging for Serbian","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"material-philology-meets-digital-onomastic","title":"Material Philology Meets Digital Onomastic Lexicography: The NordiCon Database of Medieval Nordic Personal Names in Continental Sources","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"overview-of-the-evalatin-2020-evaluation","title":"Overview of the EvaLatin 2020 Evaluation Campaign","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"voting-for-pos-tagging-of-latin-texts-using","title":"Voting for POS tagging of Latin texts: Using the flair of FLAIR to better Ensemble Classifiers by Example of Latin","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-resource-for-studying-chatino-verbal","title":"A Resource for Studying Chatino Verbal Morphology","date":"2020-04-05","arxiv_id":"2004.02083","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-sigmorphon-2019-shared-task-morphological-1","title":"The SIGMORPHON 2019 Shared Task: Morphological Analysis in Context and Cross-Lingual Transfer for Inflection","date":"2019-10-25","arxiv_id":"1910.11493","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-diacritization-lemmatization","title":"Joint Diacritization, Lemmatization, Normalization, and Fine-Grained Morphological Tagging","date":"2019-10-05","arxiv_id":"1910.02267","repositories_listed":0,"syntology":null},{"url":null,"slug":"czech-text-processing-with-contextual","title":"Czech Text Processing with Contextual Embeddings: POS Tagging, Lemmatization, Parsing and NER","date":"2019-09-08","arxiv_id":"1909.03544","repositories_listed":0,"syntology":null},{"url":null,"slug":"to-lemmatize-or-not-to-lemmatize-how-word","title":"To lemmatize or not to lemmatize: how word normalisation affects ELMo performance in word sense disambiguation","date":"2019-09-06","arxiv_id":"1909.03135","repositories_listed":0,"syntology":null},{"url":null,"slug":"corpora-and-processing-tools-for-non-standard","title":"Corpora and Processing Tools for Non-standard Contemporary and Diachronic Balkan Slavic","date":"2019-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/evaluating-contextualized-embeddings-on-54","slug":"evaluating-contextualized-embeddings-on-54","title":"Evaluating Contextualized Embeddings on 54 Languages in POS Tagging, Lemmatization and Dependency Parsing","date":"2019-08-20","arxiv_id":"1908.07448","repositories_listed":0,"syntology":null},{"url":null,"slug":"udpipe-at-sigmorphon-2019-contextualized-1","title":"UDPipe at SIGMORPHON 2019: Contextualized Embeddings, Regularization with Morphological Categories, Corpora Merging","date":"2019-08-19","arxiv_id":"1908.06931","repositories_listed":0,"syntology":null},{"url":null,"slug":"cbnu-system-for-sigmorphon-2019-shared-task-2","title":"CBNU System for SIGMORPHON 2019 Shared Task 2: a Pipeline Model","date":"2019-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cuni-malta-system-at-sigmorphon-2019-shared","title":"CUNI--Malta system at SIGMORPHON 2019 Shared Task on Morphological Analysis and Lemmatization in context: Operation-based word formation","date":"2019-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"harmonizing-different-lemmatization","title":"Harmonizing Different Lemmatization Strategies for Building a Knowledge Base of Linguistic Resources for Latin","date":"2019-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-sub-word-embedding-strategies","title":"Investigating Sub-Word Embedding Strategies for the Morphologically Rich and Free Phrase-Order Hungarian","date":"2019-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-team-a-multi-attention-multi-decoder","title":"Multi-Team: A Multi-attention, Multi-decoder Approach to Morphological Analysis.","date":"2019-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-lemmatization-of-multiword-expressions","title":"Neural Lemmatization of Multiword Expressions","date":"2019-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sigmorphon-2019-task-2-system-description","title":"Sigmorphon 2019 Task 2 system description paper: Morphological analysis in context for many languages, with supervision from only a few","date":"2019-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"nefnir-a-high-accuracy-lemmatizer-for","title":"Nefnir: A high accuracy lemmatizer for Icelandic","date":"2019-07-27","arxiv_id":"1907.11907","repositories_listed":0,"syntology":null},{"url":null,"slug":"development-of-email-classifier-in-brazilian","title":"Development of email classifier in Brazilian Portuguese using feature selection for automatic response","date":"2019-07-08","arxiv_id":"1907.04905","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-morphosyntactic-analyzers-from-the","title":"Learning Morphosyntactic Analyzers from the Bible via Iterative Annotation Projection across 26 Languages","date":"2019-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"usf-at-semeval-2019-task-6-offensive-language","title":"USF at SemEval-2019 Task 6: Offensive Language Detection Using LSTM With Word Embeddings","date":"2019-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"producing-corpora-of-medieval-and-premodern","title":"Producing Corpora of Medieval and Premodern Occitan","date":"2019-04-26","arxiv_id":"1904.11815","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simple-joint-model-for-improved-contextual","title":"A Simple Joint Model for Improved Contextual Neural Lemmatization","date":"2019-04-04","arxiv_id":"1904.02306","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilevel-text-normalization-with-sequence","title":"Multilevel Text Normalization with Sequence-to-Sequence Networks and Multisource Learning","date":"2019-03-27","arxiv_id":"1903.11340","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-and-zero-shot-learning-for","title":"Few-Shot and Zero-Shot Learning for Historical Text Normalization","date":"2019-03-12","arxiv_id":"1903.04870","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-lemmatizer-a-sequence-to-sequence","title":"Universal Lemmatizer: A Sequence to Sequence Model for Lemmatizing Universal Dependencies Treebanks","date":"2019-02-03","arxiv_id":"1902.00972","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-morphological-analysis-for-uralic","title":"Data-Driven Morphological Analysis for Uralic Languages","date":"2019-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-morphological-analyzer-for-shipibo-konibo","title":"A Morphological Analyzer for Shipibo-Konibo","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-free-encoder-decoder-for","title":"Attention-free encoder decoder for morphological processing","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/turku-neural-parser-pipeline-an-end-to-end","slug":"turku-neural-parser-pipeline-an-end-to-end","title":"Turku Neural Parser Pipeline: An End-to-End System for the CoNLL 2018 Shared Task","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/udpipe-20-prototype-at-conll-2018-ud-shared","slug":"udpipe-20-prototype-at-conll-2018-ud-shared","title":"UDPipe 2.0 Prototype at CoNLL 2018 UD Shared Task","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"uzhsmm4h-system-descriptions","title":"UZH@SMM4H: System Descriptions","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-lemmatizer-and-a-spell-checker-for","title":"Building a Lemmatizer and a Spell-checker for Sorani Kurdish","date":"2018-09-27","arxiv_id":"1809.10763","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-evaluation-of-lexicon-based-sentiment","title":"An Evaluation of Lexicon-based Sentiment Analysis Techniques for the Plays of Gotthold Ephraim Lessing","date":"2018-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"local-string-transduction-as-sequence","title":"Local String Transduction as Sequence Labeling","date":"2018-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"character-level-supervision-for-low-resource","title":"Character-level Supervision for Low-resource POS Tagging","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"context-sensitive-neural-lemmatization-with","title":"Context Sensitive Neural Lemmatization with Lematus","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-query-expansion-on-an-accounting-corpus","title":"Fast Query Expansion on an Accounting Corpus using Sub-Word Embeddings","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tw-star-at-semeval-2018-task-1-preprocessing","title":"Tw-StAR at SemEval-2018 Task 1: Preprocessing Impact on Multi-label Emotion Classification","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-of-sentence-length-measures-in","title":"Robustness of sentence length measures in written texts","date":"2018-05-02","arxiv_id":"1805.01460","repositories_listed":0,"syntology":null},{"url":"/paper/a-morphologically-annotated-corpus-of-emirati","slug":"a-morphologically-annotated-corpus-of-emirati","title":"A Morphologically Annotated Corpus of Emirati Arabic","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bioro-the-biomedical-corpus-for-the-romanian","title":"BioRo: The Biomedical Corpus for the Romanian Language","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"coreference-resolution-in-freeling-40","title":"Coreference Resolution in FreeLing 4.0","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-new-linguistic-resources-and-tools","title":"Developing New Linguistic Resources and Tools for the Galician Language","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-a-gold-standard-for-a-swedish","title":"Generating a Gold Standard for a Swedish Sentiment Lexicon","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"moving-tiger-beyond-sentence-level","title":"Moving TIGER beyond Sentence-Level","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"parser-combinators-for-tigrinya-and-oromo","title":"Parser combinators for Tigrinya and Oromo morphology","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sentiarabic-a-sentiment-analyzer-for-standard","title":"SentiArabic: A Sentiment Analyzer for Standard Arabic","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-use-of-text-alignment-in-semi-automatic","title":"The Use of Text Alignment in Semi-Automatic Error Analysis: Use Case in the Development of the Corpus of the Latvian Language Learners","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"treeannotator-versatile-visual-annotation-of","title":"TreeAnnotator: Versatile Visual Annotation of Hierarchical Text Relations","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-morphologies-for-the-caucasus","title":"Universal Morphologies for the Caucasus region","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"very-large-scale-lexical-resources-to-enhance","title":"Very Large-Scale Lexical Resources to Enhance Chinese and Japanese Machine Translation","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-categorization-of-tagalog-documents","title":"Automatic Categorization of Tagalog Documents Using Support Vector Machines","date":"2017-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"build-fast-and-accurate-lemmatization-for","title":"Build Fast and Accurate Lemmatization for Arabic","date":"2017-10-18","arxiv_id":"1710.06700","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-the-ttl-romanian-pos-tagger-to-the","title":"Adapting the TTL Romanian POS Tagger to the Biomedical Domain","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-extensible-multilingual-open-source","title":"An Extensible Multilingual Open Source Lemmatizer","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatically-acquired-lexical-knowledge","title":"Automatically Acquired Lexical Knowledge Improves Japanese Joint Morphological and Dependency Analysis","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bleu2vec-the-painfully-familiar-metric-on","title":"bleu2vec: the Painfully Familiar Metric on Continuous Vector Space Steroids","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-finite-state-morphological","title":"Evaluation of Finite State Morphological Analyzers Based on Paradigm Extraction from Wiktionary","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-and-accurate-decision-trees-for-natural","title":"Fast and Accurate Decision Trees for Natural Language Processing Tasks","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"lemmatization-of-multi-word-common-noun","title":"Lemmatization of Multi-word Common Noun Phrases and Named Entities in Polish","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"impact-of-feature-selection-on-micro-text","title":"Impact of Feature Selection on Micro-Text Classification","date":"2017-08-27","arxiv_id":"1708.08123","repositories_listed":0,"syntology":null}],"record_sha256":"6be8e13c1dfd29d5a761a7a59c23915690ed1f0139f77842d520a8ad2dd87a62","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}