{"url":"/dataset/tat","name":"TAT","full_name":"Taiwanese Across Taiwan","description_markdown":"**Taiwanese Across Taiwan (TAT)** corpus is a Large-Scale database of Native Taiwanese Article/Reading Speech collected across Taiwan. This corpus contains native Taiwanese speech of various accent across Taiwan. The corpus is annotated twice for use in voice recognition research. The corpus contains recording from 100 native speakers, each with length of 30 minutes making a total of 100 hours of speech data.\r\n\r\nSource: [https://sites.google.com/speech.ntut.edu.tw/fsw/home/tat-corpus?authuser=0](https://sites.google.com/speech.ntut.edu.tw/fsw/home/tat-corpus?authuser=0)\r\n\r\nImage Source: [https://sites.google.com/speech.ntut.edu.tw/fsw/home/tat-corpus?authuser=0](https://sites.google.com/speech.ntut.edu.tw/fsw/home/tat-corpus?authuser=0)","description_withheld":null,"homepage":"https://sites.google.com/nycu.edu.tw/speechlabx/tat_s2st_benchmark?authuser=0","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":{"name":"CC BY-NC 4.0","url":"https://creativecommons.org/licenses/by-nc/4.0/"},"modalities":[{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Speech-to-Speech Translation","url":"/task/speech-to-speech-translation","datasets_with_task":"/datasets/task/speech-to-speech-translation"},{"name":"Voice Query Recognition","url":"/task/voice-query-recognition","datasets_with_task":"/datasets/task/voice-query-recognition"}],"languages":[],"variants":["TAT"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/speech-to-speech-translation-on-tat","task":"Speech-to-Speech Translation","dataset_variant":"TAT","rows":8,"metrics":["ASR-BLEU (Dev)","ASR-BLEU (Test)"],"first_row_in_archive_order":{"model":"Hokkien→En (Two-pass decoding)","paper":"/paper/speech-to-speech-translation-for-a-real-world","metrics":{"ASR-BLEU (Dev)":"13.6","ASR-BLEU (Test)":"12.5"},"code_links":[{"title":"facebookresearch/fairseq","url":"https://github.com/facebookresearch/fairseq/tree/ust/examples/hokkien"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/speech-to-speech-translation-for-a-real-world","title":"Speech-to-speech translation for a real-world unwritten language","date":null,"rows_on_this_dataset":8,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}