{"url":"/dataset/jesc","name":"JESC","full_name":"Japanese-English Subtitle Corpus","description_markdown":"Japanese-English Subtitle Corpus is a large Japanese-English parallel corpus covering the underrepresented domain of conversational dialogue. It consists of more than 3.2 million examples, making it the largest freely available dataset of its kind. The corpus was assembled by crawling and aligning subtitles found on the web. \r\n\r\nSource: [JESC: Japanese-English Subtitle Corpus](https://arxiv.org/pdf/1710.10639v4.pdf)","description_withheld":null,"homepage":"https://nlp.stanford.edu/projects/jesc/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/jesc-japanese-english-subtitle-corpus","title":"JESC: Japanese-English Subtitle Corpus","first_author":"Reid Pryzant","url":null},"license":{"name":"CC BY-SA 4.0","url":"https://creativecommons.org/licenses/by-sa/4.0/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Domain Adaptation","url":"/task/domain-adaptation","datasets_with_task":"/datasets/task/domain-adaptation"},{"name":"Cross-Lingual Transfer","url":"/task/cross-lingual-transfer","datasets_with_task":"/datasets/task/cross-lingual-transfer"}],"languages":[{"name":"Japanese","url":"/datasets/language/japanese"}],"variants":["JESC"],"data_loaders":[],"num_papers_in_archive":17,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}