{"url":"/dataset/italki-nli","name":"italki NLI","full_name":"italki NLI","description_markdown":"A large, crowd-sourced dataset for the Native Language Identification (NLI) task. People learning English as a second language write practice Notebooks which can be used to classify the author's native language using word choice, spelling mistakes and other language features.\r\n\r\nThe dataset has:\r\n\r\n- 11 languages (Arabic, Chinese, French, German, Hindi, Italian, Japanese, Korean, Spanish, Telagu, Turkish)\r\n- 111,917 documents","description_withheld":null,"homepage":"https://github.com/ghomasHudson/italkiCorpus","introduced_date":"2018-12-10","introduced_date_note":null,"introduced_by":{"paper":"/paper/on-the-development-of-a-large-scale-corpus","title":"On the Development of a Large Scale Corpus for Native Language Identification","first_author":"Thomas Hudson","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Native Language Identification","url":"/task/native-language-identification","datasets_with_task":"/datasets/task/native-language-identification"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["italki NLI"],"data_loaders":[{"repo":"https://github.com/ghomasHudson/italkiCorpus","url":"https://github.com/ghomasHudson/italkiCorpus","frameworks":[]}],"num_papers_in_archive":3,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/native-language-identification-on-italki-nli","task":"Native Language Identification","dataset_variant":"italki NLI","rows":2,"metrics":["Average F1"],"first_row_in_archive_order":{"model":"Tubasfs","paper":"/paper/fewer-features-perform-well-at-native","metrics":{"Average F1":"0.5807"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/fewer-features-perform-well-at-native","title":"Fewer features perform well at Native Language Identification task","date":"2017-09-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-study-of-n-gram-and-embedding","title":"A study of N-gram and Embedding Representations for Native Language Identification","date":"2017-09-01","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}