{"url":"/dataset/jparacrawl","name":"JParaCrawl","full_name":null,"description_markdown":"JParaCrawl is a parallel corpus for English-Japanese, for which the amount of publicly available parallel corpora is still limited. The parallel corpus was constructed by broadly crawling the web and automatically aligning parallel sentences. The corpus amassed over 8.7 million sentence pairs.\r\n\r\nSource: [JParaCrawl: A Large Scale Web-Based English-Japanese Parallel Corpus](https://arxiv.org/pdf/1911.10668)","description_withheld":null,"homepage":"http://www.kecl.ntt.co.jp/icl/lirg/jparacrawl/","introduced_date":"2019-11-25","introduced_date_note":null,"introduced_by":{"paper":"/paper/jparacrawl-a-large-scale-web-based-english","title":"JParaCrawl: A Large Scale Web-Based English-Japanese Parallel Corpus","first_author":"Makoto Morishita","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Machine Translation","url":"/task/machine-translation","datasets_with_task":"/datasets/task/machine-translation"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"Japanese","url":"/datasets/language/japanese"}],"variants":["JParaCrawl"],"data_loaders":[],"num_papers_in_archive":6,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}