{"url":"/dataset/kleister-nda","name":"Kleister NDA","full_name":null,"description_markdown":"**Kleister NDA** is a dataset for Key Information Extraction (KIE). The dataset contains a mix of scanned and born-digital long formal English-language documents. For this datasets, an NLP system is expected to find or infer various types of entities by employing both textual and structural layout features.  The Kleister NDA dataset has 540 Non-disclosure Agreements, with 3,229 unique pages and 2,160 entities to extract.","description_withheld":null,"homepage":"https://github.com/applicaai/kleister-nda","introduced_date":"2021-05-12","introduced_date_note":null,"introduced_by":{"paper":"/paper/kleister-key-information-extraction-datasets","title":"Kleister: Key Information Extraction Datasets Involving Long Documents with Complex Layouts","first_author":"Tomasz Stanisławek","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Key Information Extraction","url":"/task/key-information-extraction","datasets_with_task":"/datasets/task/key-information-extraction"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Kleister NDA"],"data_loaders":[{"repo":"https://github.com/applicaai/kleister-nda","url":"https://github.com/applicaai/kleister-nda","frameworks":[]}],"num_papers_in_archive":17,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/key-information-extraction-on-kleister-nda","task":"Key Information Extraction","dataset_variant":"Kleister NDA","rows":3,"metrics":["F1"],"first_row_in_archive_order":{"model":"LayoutLMv2LARGE","paper":"/paper/layoutlmv2-multi-modal-pre-training-for","metrics":{"F1":"85.2"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"PaddlePaddle/PaddleOCR","url":"https://github.com/PaddlePaddle/PaddleOCR"},{"title":"microsoft/unilm","url":"https://github.com/microsoft/unilm"},{"title":"PaddlePaddle/PaddleNLP","url":"https://github.com/PaddlePaddle/PaddleNLP/tree/develop/paddlenlp/transformers/layoutlmv2"},{"title":"facebookresearch/data2vec_vision","url":"https://github.com/facebookresearch/data2vec_vision"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/layoutlmv2"},{"title":"MS-P3/code3","url":"https://github.com/MS-P3/code3/tree/main/layoutlmv2"},{"title":"MindSpore-scientific-2/code-14","url":"https://github.com/MindSpore-scientific-2/code-14/tree/main/layoutlmv2"},{"title":"MindSpore-scientific/code-7","url":"https://github.com/MindSpore-scientific/code-7/tree/main/LayoutLMv2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/layoutlmv2-multi-modal-pre-training-for","title":"LayoutLMv2: Multi-modal Pre-training for Visually-Rich Document Understanding","date":"2020-12-29","rows_on_this_dataset":2,"code_links":9,"syntology":null},{"paper":"/paper/lambert-layout-aware-language-modeling-using","title":"LAMBERT: Layout-Aware (Language) Modeling for information extraction","date":"2020-02-19","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}