{"url":"/dataset/lambada","name":"LAMBADA","full_name":null,"description_markdown":"The **LAMBADA** (LAnguage Modeling Broadened to Account for Discourse Aspects) benchmark is an open-ended cloze task which consists of about 10,000 passages from BooksCorpus where a missing target word is predicted in the last sentence of each passage. The missing word is constrained to always be the last word of the last sentence and there are no candidate words to choose from. Examples were filtered by humans to ensure they were possible to guess given the context, i.e., the sentences in the passage leading up to the last sentence. Examples were further filtered to ensure that missing words could not be guessed without the context, ensuring that models attempting the dataset would need to reason over the entire paragraph to answer questions.\r\n\r\nSource: [Recent Advances in Natural Language Inference:A Survey of Benchmarks, Resources, and Approaches](https://arxiv.org/abs/1904.01172)\r\nImage Source: [https://arxiv.org/pdf/1606.06031.pdf](https://arxiv.org/pdf/1606.06031.pdf)","description_withheld":null,"homepage":"https://zenodo.org/record/2630551#.YFJVaWT7S_w","introduced_date":"2016-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/the-lambada-dataset-word-prediction-requiring","title":"The LAMBADA dataset: Word prediction requiring a broad discourse context","first_author":"Denis Paperno","url":null},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/legalcode"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["LAMBADA"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/cimec/lambada","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/lambada","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/lambada","frameworks":["tf","jax"]}],"num_papers_in_archive":293,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/language-modelling-on-lambada","task":"Language Modelling","dataset_variant":"LAMBADA","rows":37,"metrics":["Accuracy","Perplexity"],"first_row_in_archive_order":{"model":"PaLM-540B (Few-Shot)","paper":"/paper/palm-scaling-language-modeling-with-pathways-1","metrics":{"Accuracy":"89.7"},"code_links":[{"title":"lucidrains/CoCa-pytorch","url":"https://github.com/lucidrains/CoCa-pytorch"},{"title":"lucidrains/PaLM-pytorch","url":"https://github.com/lucidrains/PaLM-pytorch"},{"title":"google/paxml","url":"https://github.com/google/paxml"},{"title":"foundation-model-stack/fms-fsdp","url":"https://github.com/foundation-model-stack/fms-fsdp"},{"title":"lucidrains/PaLM-jax","url":"https://github.com/lucidrains/PaLM-jax"},{"title":"chrisociepa/allamo","url":"https://github.com/chrisociepa/allamo"},{"title":"conceptofmind/PaLM-flax","url":"https://github.com/conceptofmind/PaLM-flax"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/mamba-linear-time-sequence-modeling-with","title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","date":"2023-12-01","rows_on_this_dataset":1,"code_links":35,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":62,"samples_ran":18,"samples_unverified":44,"pointer_only_for_licence":28,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/stay-on-topic-with-classifier-free-guidance","title":"Stay on topic with Classifier-Free Guidance","date":"2023-06-30","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/palm-2-technical-report-1","title":"PaLM 2 Technical Report","date":"2023-05-17","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/pythia-a-suite-for-analyzing-large-language","title":"Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling","date":"2023-04-03","rows_on_this_dataset":4,"code_links":4,"syntology":null},{"paper":"/paper/massive-language-models-can-be-accurately","title":"SparseGPT: Massive Language Models Can Be Accurately Pruned in One-Shot","date":"2023-01-02","rows_on_this_dataset":5,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":2,"samples_unverified":10,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/glm-130b-an-open-bilingual-pre-trained-model","title":"GLM-130B: An Open Bilingual Pre-trained Model","date":"2022-10-05","rows_on_this_dataset":1,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":5,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/palm-scaling-language-modeling-with-pathways-1","title":"PaLM: Scaling Language Modeling with Pathways","date":"2022-04-05","rows_on_this_dataset":3,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":37,"samples_ran":30,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/training-compute-optimal-large-language","title":"Training Compute-Optimal Large Language Models","date":"2022-03-29","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":8,"samples_unverified":3,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/using-deepspeed-and-megatron-to-train","title":"Using DeepSpeed and Megatron to Train Megatron-Turing NLG 530B, A Large-Scale Generative Language Model","date":"2022-01-28","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/glam-efficient-scaling-of-language-models","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","date":"2021-12-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/all-nlp-tasks-are-generation-tasks-a-general","title":"GLM: General Language Model Pretraining with Autoregressive Blank Infilling","date":"2021-03-18","rows_on_this_dataset":2,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/language-models-are-few-shot-learners","title":"Language Models are Few-Shot Learners","date":"2020-05-28","rows_on_this_dataset":5,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":65,"samples_ran":15,"samples_unverified":50,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/residual-shuffle-exchange-networks-for-fast","title":"Residual Shuffle-Exchange Networks for Fast Processing of Long Sequences","date":"2020-04-06","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/test-time-training-for-out-of-distribution-1","title":"Test-Time Training with Self-Supervision for Generalization under Distribution Shifts","date":"2019-09-29","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/language-models-are-unsupervised-multitask","title":"Language Models are Unsupervised Multitask Learners","date":"2019-02-14","rows_on_this_dataset":1,"code_links":21,"syntology":null},{"paper":"/paper/universal-transformers","title":"Universal Transformers","date":"2018-07-10","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":14,"samples_unverified":11,"pointer_only_for_licence":24,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/broad-context-language-modeling-as-reading","title":"Broad Context Language Modeling as Reading Comprehension","date":"2016-10-26","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":8,"samples_harvested":234,"samples_ran":93,"samples_unverified":141,"pointer_only_for_licence":69,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}