{"url":"/dataset/opensubtitles","name":"OpenSubtitles","full_name":null,"description_markdown":"OpenSubtitles is collection of multilingual parallel corpora. The dataset is compiled from a large database of movie and TV subtitles and includes a total of 1689 bitexts spanning 2.6 billion sentences across 60 languages.","description_withheld":null,"homepage":"http://opus.nlpl.eu/OpenSubtitles2018.php","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/opensubtitles2016-extracting-large-parallel","title":"OpenSubtitles2016: Extracting Large Parallel Corpora from Movie and TV Subtitles","first_author":"Pierre Lison","url":null},"license":{"name":"Unknown","url":null},"modalities":[],"tasks":[{"name":"Domain Adaptation","url":"/task/domain-adaptation","datasets_with_task":"/datasets/task/domain-adaptation"},{"name":"Machine Translation","url":"/task/machine-translation","datasets_with_task":"/datasets/task/machine-translation"},{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"},{"name":"Dialogue Generation","url":"/task/dialogue-generation","datasets_with_task":"/datasets/task/dialogue-generation"},{"name":"Language Identification","url":"/task/language-identification","datasets_with_task":"/datasets/task/language-identification"}],"languages":[{"name":"Russian","url":"/datasets/language/russian"}],"variants":["OpenSubtitles"," OpenSubtitles"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Helsinki-NLP/open_subtitles","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/open_subtitles","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/facebookresearch/ParlAI","url":"https://parl.ai/docs/tasks.html#open-subtitles","frameworks":["pytorch"]}],"num_papers_in_archive":214,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/language-identification-on-opensubtitles","task":"Language Identification","dataset_variant":"OpenSubtitles","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Apple bi-LSTM","paper":"/paper/a-reproduction-of-apple-s-bi-directional-lstm","metrics":{"Accuracy":"91.37"},"code_links":[{"title":"AU-DIS/LSTM_langid","url":"https://github.com/AU-DIS/LSTM_langid"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/language-modelling-on-opensubtitles","task":"Language Modelling","dataset_variant":"OpenSubtitles","rows":1,"metrics":["BPB"],"first_row_in_archive_order":{"model":"Gopher","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"BPB":"0.899"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/machine-translation-on-opensubtitles","task":"Machine Translation","dataset_variant":"OpenSubtitles","rows":1,"metrics":["BLEU score","METEOR"],"first_row_in_archive_order":{"model":"Fine tuned MarianMT","paper":"/paper/crossing-language-borders-a-pipeline-for","metrics":{"BLEU score":"27","METEOR":"61"},"code_links":[{"title":"Sagarika-Singh-99/indonesian_manhwa_translation","url":"https://github.com/Sagarika-Singh-99/indonesian_manhwa_translation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/crossing-language-borders-a-pipeline-for","title":"Crossing Language Borders: A Pipeline for Indonesian Manhwa Translation","date":"2025-01-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/scaling-language-models-methods-analysis-1","title":"Scaling Language Models: Methods, Analysis & Insights from Training Gopher","date":"2021-12-08","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/a-reproduction-of-apple-s-bi-directional-lstm","title":"A reproduction of Apple's bi-directional LSTM models for language identification in short strings","date":"2021-02-11","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}