{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/lemmatization/papers/4","list_of":"/task/lemmatization","task":"Lemmatization","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":4,"rows_per_page":100,"rows":[301,351],"of":351,"counts":{"archive_papers_tagged":351,"with_a_code_link":68,"where_syntology_ran_a_sample":3,"not_listed_spam_title":0,"listed":351,"listed_where_code_ran":3,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2,"every_run_a_failure_of_syntologys_instrument":1,"listed_with_a_run_with_no_instrument_failure":2,"listed_every_run_a_failure_of_syntologys_instrument":1,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/lemmatization","prev":"/task/lemmatization/papers/3","next":null,"papers":[{"url":null,"slug":"the-effects-of-syntactic-features-in","title":"The Effects of Syntactic Features in Automatic Prediction of Morphology","date":"2013-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-lexical-processing-framework-based","title":"A unified lexical processing framework based on the Margin Infused Relaxed Algorithm. A case study on the Romanian Language","date":"2013-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-language-plagiarism-detection-methods","title":"Cross-Language Plagiarism Detection Methods","date":"2013-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"purepos-20-a-hybrid-tool-for-morphological","title":"PurePos 2.0: a hybrid tool for morphological disambiguation","date":"2013-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"chimera-three-heads-for-english-to-czech","title":"Chimera -- Three Heads for English-to-Czech Translation","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dkpro-similarity-an-open-source-framework-for","title":"DKPro Similarity: An Open Source Framework for Text Similarity","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"factored-machine-translation-systems-for","title":"Factored Machine Translation Systems for Russian-English","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"lemmatization-and-morphosyntactic-tagging-of","title":"Lemmatization and Morphosyntactic Tagging of Croatian and Serbian","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"modernizing-historical-slovene-words-with","title":"Modernizing historical Slovene words with character-based SMT","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"uzh-in-bionlp-2013","title":"UZH in BioNLP 2013","date":"2013-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-extended-morphological-analyzer-of-german","title":"An extended morphological analyzer of German handling verbal forms with separated separable particles (Un analyseur morphologique \\'etendu de l'allemand traitant les formes verbales \\`a particule s\\'epar\\'ee) [in French]","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cngl-core-referential-translation-machines","title":"CNGL-CORE: Referential Translation Machines for Measuring Semantic Similarity","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"klue-core-a-regression-model-of-semantic","title":"KLUE-CORE: A regression model of semantic textual similarity","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"lvic-limsi-using-syntactic-features-and-multi","title":"[LVIC-LIMSI]: Using Syntactic Features and Multi-polarity Words for Sentiment Analysis in Twitter","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"morphological-analysis-and-disambiguation-for","title":"Morphological Analysis and Disambiguation for Dialectal Arabic","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"nrc-a-machine-translation-approach-to-cross","title":"NRC: A Machine Translation Approach to Cross-Lingual Word Sense Disambiguation (SemEval-2013 Task 10)","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneous-word-morpheme-alignment-for","title":"Simultaneous Word-Morpheme Alignment for Statistical Machine Translation","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ssa-uo-unsupervised-sentiment-analysis-in","title":"SSA-UO: Unsupervised Sentiment Analysis in Twitter","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-an-automatic-identification-of","title":"Towards an automatic identification of chiasmus of words (Vers une identification automatique du chiasme de mots) [in French]","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ubc_uos-typed-regression-for-typed-similarity","title":"UBC\\_UOS-TYPED: Regression for typed-similarity","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-lemmatization-for-mongolian-and-its","title":"Enhancing Lemmatization for Mongolian and its Application to Statistical Machine Translation","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"lexical-categories-for-improved-parsing-of","title":"Lexical Categories for Improved Parsing of Web Data","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-floating-arabic-dictionary-an-automatic","title":"The Floating Arabic Dictionary: An Automatic Method for Updating a Lexical Database through the Detection and Lemmatization of Unknown Words","date":"2012-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"annlor-a-naive-notation-system-for-lexical","title":"ANNLOR: A Na\\\"\\ive Notation-system for Lexical Outputs Ranking","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"celi-an-experiment-with-cross-language","title":"CELI: An Experiment with Cross Language Textual Entailment","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"handling-unknown-words-in-arabic-fst","title":"Handling Unknown Words in Arabic FST Morphology","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-lexical-generalization-for","title":"Probabilistic Lexical Generalization for French Dependency Parsing","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"samar-a-system-for-subjectivity-and-sentiment","title":"SAMAR: A System for Subjectivity and Sentiment Analysis of Arabic Social Media","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"statistical-parsing-of-spanish-and-data","title":"Statistical Parsing of Spanish and Data Driven Lemmatization","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"wsd-for-n-best-reranking-and-local-language","title":"WSD for n-best reranking and local language modeling in SMT","date":"2012-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"enrichir-et-raisonner-sur-des-espaces","title":"Enrichir et raisonner sur des espaces s\\'emantiques pour l'attribution de mots-cl\\'es (Enriching and reasoning on semantic spaces for keyword extraction) [in French]","date":"2012-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"indexation-libre-et-controlee-darticles","title":"Indexation libre et contr\\^ol\\'ee d'articles scientifiques. Pr\\'esentation et r\\'esultats du d\\'efi fouille de textes DEFT2012 (Controlled and free indexing of scientific papers. Presentation and results of the DEFT2012 text-mining challenge) [in French]","date":"2012-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mining-wisdom","title":"Mining wisdom","date":"2012-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-and-evaluating-a-generic-term","title":"Adapting and evaluating a generic term extraction tool","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-and-aligning-german-compound-nouns","title":"Analyzing and Aligning German compound nouns","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-multilingual-parallel-corpus-for","title":"Building a multilingual parallel corpus for human users","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"building-large-monolingual-dictionaries-at","title":"Building Large Monolingual Dictionaries at the Leipzig Corpora Collection: From 100 to 200 Languages","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"first-steps-towards-the-semi-automatic","title":"First Steps towards the Semi-automatic Development of a Wordformation-based Lexicon of Latin","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"holaaa-writin-like-u-talk-is-kewl-but-kinda","title":"Holaaa!! writin like u talk is kewl but kinda hard 4 NLP","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"iula2standoff-a-tool-for-creating-standoff","title":"Iula2Standoff: a tool for creating standoff documents for the IULACT","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"korp-the-corpus-infrastructure-of-sprakbanken","title":"Korp --- the corpus infrastructure of Spr\\aakbanken","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"linguistic-analysis-processing-line-for","title":"Linguistic Analysis Processing Line for Bulgarian","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neotag-a-pos-tagger-for-grammatical-neologism","title":"NeoTag: a POS Tagger for Grammatical Neologism Detection","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"rombac-the-romanian-balanced-annotated-corpus","title":"ROMBAC: The Romanian Balanced Annotated Corpus","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-annotation-of-the-c-oral-brasil-oral","title":"The annotation of the C-ORAL-BRASIL oral through the implementation of the Palavras Parser","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-goo300k-corpus-of-historical-slovene","title":"The goo300k corpus of historical Slovene","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-netlog-corpus-a-resource-for-the-study-of","title":"The Netlog Corpus. A Resource for the Study of Flemish Dutch Internet Language","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-political-speech-corpus-of-bulgarian","title":"The Political Speech Corpus of Bulgarian","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ubiquitous-usage-of-a-broad-coverage-french","title":"Ubiquitous Usage of a Broad Coverage French Corpus: Processing the Est Republicain corpus","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"vreselijk-mooi-terribly-beautiful-a","title":"``Vreselijk mooi!'' (terribly beautiful): A Subjectivity Lexicon for Dutch Adjectives.","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-annotated-english-child-language-database","title":"An annotated English child language database","date":"2012-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"c6e27d59479769524bbc6ea9ea5312ddb241cfd153bad483b8bf46357211ad0c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}