{"url":"/dataset/oteannv3","name":"OTEANNv3","full_name":null,"description_markdown":"This dataset contains orthographic samples of words in 19 languages (ar, br, de, en, eno, ent, eo, es, fi, fr, fro, it, ko, nl, pt, ru, sh, tr, zh). Each sample contains two text features: a Word (the textual representation of the word according to its orthography) and a Pronunciation (the highest-surface IPA pronunciation of the word as pronunced in its language).","description_withheld":null,"homepage":"https://github.com/marxav/oteann3/","introduced_date":"2019-12-31","introduced_date_note":null,"introduced_by":{"paper":"/paper/oteann-estimating-the-transparency-of","title":"OTEANN: Estimating the Transparency of Orthographies with an Artificial Neural Network","first_author":"Xavier Marjou","url":null},"license":null,"modalities":[],"tasks":[],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"French","url":"/datasets/language/french"},{"name":"Spanish","url":"/datasets/language/spanish"},{"name":"German","url":"/datasets/language/german"},{"name":"Italian","url":"/datasets/language/italian"},{"name":"Russian","url":"/datasets/language/russian"},{"name":"Portuguese","url":"/datasets/language/portuguese"},{"name":"Arabic","url":"/datasets/language/arabic"},{"name":"Breton","url":"/datasets/language/breton"},{"name":"Dutch","url":"/datasets/language/dutch"},{"name":"Finnish","url":"/datasets/language/finnish"},{"name":"Korean","url":"/datasets/language/korean"},{"name":"Turkish","url":"/datasets/language/turkish"},{"name":"Mandarin Chinese","url":"/datasets/language/mandarin-chinese"},{"name":"Esperanto","url":"/datasets/language/esperanto"},{"name":"Serbo-Croatian","url":"/datasets/language/serbo-croatian"}],"variants":["OTEANNv3"],"data_loaders":[{"repo":"https://github.com/marxav/oteann3","url":"https://github.com/marxav/oteann3","frameworks":["pytorch"]}],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}