{"url":"/dataset/turkish-punctuation-restoration","name":"Turkish Punctuation Restoration","full_name":null,"description_markdown":"we have prepared a dataset using publicly available TED Talks transcripts [27] and selected the Turkish corpus. The resulting Turkish punctuation restoration dataset currently consists of 146K sentences and 1.8M tokens. The ratio of the train, validation, and test splits are 0.8, 0.1, and 0.1, respectively. Data files contain two columns. The first column has the tokens separated by white space. The second column includes tags for each token.","description_withheld":null,"homepage":"https://github.com/uygarkurt/Turkish-Punctuation-Restoration/tree/main/multitarget-ted","introduced_date":"2023-09-15","introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Punctuation Restoration","url":"/task/punctuation-restoration","datasets_with_task":"/datasets/task/punctuation-restoration"}],"languages":[{"name":"Turkish","url":"/datasets/language/turkish"}],"variants":["Turkish Punctuation Restoration"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}