{"url":"/dataset/muld","name":"MuLD","full_name":"Multitask Long Document Benchmark","description_markdown":"**MuLD** (**Multitask Long Document Benchmark**) is a set of 6 NLP tasks where the inputs consist of at least 10,000 words. The benchmark covers a wide variety of task types including translation, summarization, question answering, and classification. Additionally there is a range of output lengths from a single word classification label all the way up to an output longer than the input text.","description_withheld":null,"homepage":"https://github.com/ghomashudson/muld","introduced_date":"2022-02-15","introduced_date_note":null,"introduced_by":{"paper":"/paper/muld-the-multitask-long-document-benchmark","title":"MuLD: The Multitask Long Document Benchmark","first_author":"G Thomas Hudson","url":null},"license":{"name":"Custom","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Text Classification","url":"/task/text-classification","datasets_with_task":"/datasets/task/text-classification"},{"name":"Summarization","url":"/task/summarization","datasets_with_task":"/datasets/task/summarization"},{"name":"Natural Language Understanding","url":"/task/natural-language-understanding","datasets_with_task":"/datasets/task/natural-language-understanding"},{"name":"Translation","url":"/task/translation","datasets_with_task":"/datasets/task/translation"},{"name":"Long-range modeling","url":"/task/long-range-modeling","datasets_with_task":"/datasets/task/long-range-modeling"},{"name":"Style change detection","url":"/task/style-change-detection","datasets_with_task":"/datasets/task/style-change-detection"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["MuLD","MuLD (NarrativeQA)","MuLD (HotpotQA)","MuLD (Style Change)","MuLD (Character Type)","MuLD (VLSP)","MuLD (OpenSubtitles)"],"data_loaders":[{"repo":"https://github.com/ghomashudson/muld","url":"https://github.com/ghomashudson/muld","frameworks":[]}],"num_papers_in_archive":3,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-answering-on-muld-hotpotqa","task":"Question Answering","dataset_variant":"MuLD (HotpotQA)","rows":2,"metrics":["BLEU-1","BLEU-4","METEOR","Rouge-L"],"first_row_in_archive_order":{"model":"Longformer","paper":"/paper/muld-the-multitask-long-document-benchmark","metrics":{"BLEU-1":"30.38","BLEU-4":"16.76","METEOR":"4.98","Rouge-L":"30.49"},"code_links":[{"title":"ghomashudson/muld","url":"https://github.com/ghomashudson/muld"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/question-answering-on-muld-narrativeqa","task":"Question Answering","dataset_variant":"MuLD (NarrativeQA)","rows":2,"metrics":["BLEU-1","BLEU-4","METEOR","Rouge-L"],"first_row_in_archive_order":{"model":"Longformer","paper":"/paper/muld-the-multitask-long-document-benchmark","metrics":{"BLEU-1":"19.84","BLEU-4":"62","METEOR":"4.52","Rouge-L":"22.09"},"code_links":[{"title":"ghomashudson/muld","url":"https://github.com/ghomashudson/muld"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/summarization-on-muld-vlsp","task":"Summarization","dataset_variant":"MuLD (VLSP)","rows":2,"metrics":["BLEU-1","BLEU-4","METEOR","Rouge-L"],"first_row_in_archive_order":{"model":"Longformer","paper":"/paper/muld-the-multitask-long-document-benchmark","metrics":{"BLEU-1":"46.74","BLEU-4":"3.05","METEOR":"9.58","Rouge-L":"19.52"},"code_links":[{"title":"ghomashudson/muld","url":"https://github.com/ghomashudson/muld"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-classification-on-muld-character-type","task":"Text Classification","dataset_variant":"MuLD (Character Type)","rows":2,"metrics":["F1"],"first_row_in_archive_order":{"model":"Longformer","paper":"/paper/muld-the-multitask-long-document-benchmark","metrics":{"F1":"82.58"},"code_links":[{"title":"ghomashudson/muld","url":"https://github.com/ghomashudson/muld"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/translation-on-muld-opensubtitles","task":"Translation","dataset_variant":"MuLD (OpenSubtitles)","rows":2,"metrics":["BLEU-1","BLEU-4","METEOR","Rouge-L"],"first_row_in_archive_order":{"model":"T5","paper":"/paper/muld-the-multitask-long-document-benchmark","metrics":{"BLEU-1":"34.07","BLEU-4":"1.63","METEOR":"38.53","Rouge-L":"35.35"},"code_links":[{"title":"ghomashudson/muld","url":"https://github.com/ghomashudson/muld"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/muld-the-multitask-long-document-benchmark","title":"MuLD: The Multitask Long Document Benchmark","date":"2022-02-15","rows_on_this_dataset":10,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}