{"url":"/dataset/opusparcus","name":"Opusparcus","full_name":null,"description_markdown":"Opusparcus is a paraphrase corpus for six European languages: German, English, Finnish, French, Russian, and Swedish. The paraphrases are extracted from the OpenSubtitles2016 corpus, which contains subtitles from movies and TV shows.\r\n\r\nFor each target language, the Opusparcus data have been partitioned into three types of data sets: training, development and test sets. The training sets are large, consisting of millions of sentence pairs, and have been compiled automatically, with the help of probabilistic ranking functions. The development and test sets consist of sentence pairs that have been annotated manually; each set contains approximately 1000 sentence pairs that have been verified to be acceptable paraphrases by two annotators.\r\n\r\nSource: [Opusparcus](http://urn.fi/urn:nbn:fi:lb-2018021221)","description_withheld":null,"homepage":"http://urn.fi/urn:nbn:fi:lb-2018021221","introduced_date":"2018-09-17","introduced_date_note":null,"introduced_by":{"paper":"/paper/open-subtitles-paraphrase-corpus-for-six","title":"Open Subtitles Paraphrase Corpus for Six Languages","first_author":"Mathias Creutz","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Paraphrase Generation","url":"/task/paraphrase-generation","datasets_with_task":"/datasets/task/paraphrase-generation"},{"name":"Sentence Embedding","url":"/task/sentence-embedding","datasets_with_task":"/datasets/task/sentence-embedding"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"French","url":"/datasets/language/french"},{"name":"German","url":"/datasets/language/german"},{"name":"Russian","url":"/datasets/language/russian"},{"name":"Finnish","url":"/datasets/language/finnish"},{"name":"Swedish","url":"/datasets/language/swedish"}],"variants":["Opusparcus"],"data_loaders":[],"num_papers_in_archive":15,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}