{"url":"/dataset/unimorph","name":"UniMorph 4.0","full_name":"Universal Morphology","description_markdown":"The Universal Morphology (UniMorph) project is a collaborative effort to improve how NLP handles complex morphology in the world’s languages. The goal of UniMorph is to annotate morphological data in a universal schema that allows an inflected word from any language to be defined by its lexical meaning, typically carried by the lemma, and by a rendering of its inflectional form in terms of a bundle of morphological features from our schema. The specification of the schema is described here and in Sylak-Glassman (2016).","description_withheld":null,"homepage":"https://unimorph.github.io/","introduced_date":"2016-05-07","introduced_date_note":null,"introduced_by":{"paper":"/paper/unimorph-4-0-universal-morphology","title":"UniMorph 4.0: Universal Morphology","first_author":"Khuyagbaatar Batsuren","url":null},"license":{"name":"CC-BY-SA","url":null},"modalities":[],"tasks":[{"name":"Morpheme Segmentaiton","url":"/task/morpheme-segmentaiton","datasets_with_task":"/datasets/task/morpheme-segmentaiton"},{"name":"Morphological Inflection","url":"/task/morphological-inflection","datasets_with_task":"/datasets/task/morphological-inflection"}],"languages":[],"variants":["UniMorph 4.0"],"data_loaders":[],"num_papers_in_archive":9,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/morpheme-segmentaiton-on-unimorph-4-0","task":"Morpheme Segmentaiton","dataset_variant":"UniMorph 4.0","rows":19,"metrics":["macro avg (subtask 1)","f1 macro avg (subtask 2)","lev dist (subtask 2)"],"first_row_in_archive_order":{"model":"Subword-ULM transformer (DeepSPIN-3; soft-attention, 1-5 entmax)","paper":"/paper/beyond-characters-subword-level-morpheme","metrics":{"macro avg (subtask 1)":"97.29"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/sigmorphon-2022-shared-task-on-morpheme","title":"SIGMORPHON 2022 Shared Task on Morpheme Segmentation Submission Description: Sequence Labelling for Word-Level Morpheme Segmentation","date":"2022-07-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/morfessor-enriched-features-and-multilingual","title":"Morfessor-enriched features and multilingual training for canonical morphological segmentation","date":"2022-07-01","rows_on_this_dataset":6,"code_links":0,"syntology":null},{"paper":"/paper/jb132-submission-to-the-sigmorphon-2022","title":"JB132 submission to the SIGMORPHON 2022 Shared Task 3 on Morphological Segmentation","date":"2022-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cluzh-at-sigmorphon-2022-shared-tasks-on","title":"CLUZH at SIGMORPHON 2022 Shared Tasks on Morpheme Segmentation and Inflection Generation","date":"2022-07-01","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/beyond-characters-subword-level-morpheme","title":"Beyond Characters: Subword-level Morpheme Segmentation","date":"2022-07-01","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/the-sigmorphon-2022-shared-task-on-morpheme","title":"The SIGMORPHON 2022 Shared Task on Morpheme Segmentation","date":"2022-06-15","rows_on_this_dataset":3,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}