{"url":"/dataset/lti-langid-corpus","name":"LTI LangID Corpus","full_name":null,"description_markdown":"The LTI LangID Corpus is a dataset used for language identification (LangID) tasks. It contains text data in various languages. The dataset has had multiple releases, with the first release containing 781 \"core\" languages and 1091 languages overall.","description_withheld":null,"homepage":"https://www.cs.cmu.edu/~ralf/langid.html","introduced_date":"2020-10-27","introduced_date_note":null,"introduced_by":{"paper":"/paper/language-id-in-the-wild-unexpected-challenges","title":"Language ID in the Wild: Unexpected Challenges on the Path to a Thousand-Language Web Text Corpus","first_author":"Isaac Caswell","url":null},"license":null,"modalities":[],"tasks":[],"languages":[],"variants":["LTI LangID Corpus"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}