{"url":"/dataset/opus","name":"OPUS","full_name":"open parallel corpus","description_markdown":"OPUS is a growing collection of translated texts from the web. In the OPUS project we try to convert and align free online data, to add linguistic annotation, and to provide the community with a publicly available parallel corpus. OPUS is based on open source products and the corpus is also delivered as an open content package. We used several tools to compile the current collection. All pre-processing is done automatically. No manual corrections have been carried out.","description_withheld":null,"homepage":"https://opus.nlpl.eu/","introduced_date":"2016-05-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/opus-parallel-corpora-for-everyone","title":"OPUS – parallel corpora for everyone","first_author":"Jörg Tiedemann","url":null},"license":null,"modalities":[],"tasks":[{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Paraphrase Generation","url":"/task/paraphrase-generation","datasets_with_task":"/datasets/task/paraphrase-generation"},{"name":"Sentence Embedding","url":"/task/sentence-embedding","datasets_with_task":"/datasets/task/sentence-embedding"}],"languages":[],"variants":["Opusparcus","opus_books","opus_infopankki","OPUS"],"data_loaders":[],"num_papers_in_archive":12,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}