{"url":"/task/key-information-extraction","name":"Key Information Extraction","slug":"key-information-extraction","description_markdown":"Key Information Extraction (KIE) is aimed at extracting structured information (e.g. key-value pairs) from form-style documents (e.g. invoices), which makes an important step towards intelligent document understanding.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":74,"papers_with_code":40,"benchmarks":6,"benchmark_tables_in_archive":6,"benchmark_tables_shown":6,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":12,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/key-information-extraction-on-cord","slug":"key-information-extraction-on-cord","dataset":"CORD","dataset_url":"/dataset/cord","rows_in_archive":9,"metrics":["F1"],"first_row_in_archive_order":{"model":"RORE (GeoLayoutLM)","paper_title":"Modeling Layout Reading Order as Ordering Relations for Visually-rich Document Understanding","paper_url":"/paper/modeling-layout-reading-order-as-ordering","paper_date":"2024-09-29","arxiv_id":"2409.19672","code_links":[{"title":"chongzhangFDU/ROOR","url":"https://github.com/chongzhangFDU/ROOR"}],"syntology":null}},{"leaderboard":"/sota/key-information-extraction-on-sroie","slug":"key-information-extraction-on-sroie","dataset":"SROIE","dataset_url":"/dataset/sroie","rows_in_archive":5,"metrics":["F1","Accuracy"],"first_row_in_archive_order":{"model":"LayoutLMv2LARGE (Excluding OCR mismatch)","paper_title":"LayoutLMv2: Multi-modal Pre-training for Visually-Rich Document Understanding","paper_url":"/paper/layoutlmv2-multi-modal-pre-training-for","paper_date":"2020-12-29","arxiv_id":"2012.14740","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"PaddlePaddle/PaddleOCR","url":"https://github.com/PaddlePaddle/PaddleOCR"},{"title":"microsoft/unilm","url":"https://github.com/microsoft/unilm"},{"title":"PaddlePaddle/PaddleNLP","url":"https://github.com/PaddlePaddle/PaddleNLP/tree/develop/paddlenlp/transformers/layoutlmv2"},{"title":"facebookresearch/data2vec_vision","url":"https://github.com/facebookresearch/data2vec_vision"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/layoutlmv2"},{"title":"MS-P3/code3","url":"https://github.com/MS-P3/code3/tree/main/layoutlmv2"},{"title":"MindSpore-scientific-2/code-14","url":"https://github.com/MindSpore-scientific-2/code-14/tree/main/layoutlmv2"},{"title":"MindSpore-scientific/code-7","url":"https://github.com/MindSpore-scientific/code-7/tree/main/LayoutLMv2"}],"syntology":null}},{"leaderboard":"/sota/key-information-extraction-on-kleister-nda","slug":"key-information-extraction-on-kleister-nda","dataset":"Kleister NDA","dataset_url":"/dataset/kleister-nda","rows_in_archive":3,"metrics":["F1"],"first_row_in_archive_order":{"model":"LayoutLMv2LARGE","paper_title":"LayoutLMv2: Multi-modal Pre-training for Visually-Rich Document Understanding","paper_url":"/paper/layoutlmv2-multi-modal-pre-training-for","paper_date":"2020-12-29","arxiv_id":"2012.14740","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"PaddlePaddle/PaddleOCR","url":"https://github.com/PaddlePaddle/PaddleOCR"},{"title":"microsoft/unilm","url":"https://github.com/microsoft/unilm"},{"title":"PaddlePaddle/PaddleNLP","url":"https://github.com/PaddlePaddle/PaddleNLP/tree/develop/paddlenlp/transformers/layoutlmv2"},{"title":"facebookresearch/data2vec_vision","url":"https://github.com/facebookresearch/data2vec_vision"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/layoutlmv2"},{"title":"MS-P3/code3","url":"https://github.com/MS-P3/code3/tree/main/layoutlmv2"},{"title":"MindSpore-scientific-2/code-14","url":"https://github.com/MindSpore-scientific-2/code-14/tree/main/layoutlmv2"},{"title":"MindSpore-scientific/code-7","url":"https://github.com/MindSpore-scientific/code-7/tree/main/LayoutLMv2"}],"syntology":null}},{"leaderboard":"/sota/key-information-extraction-on-ephoie","slug":"key-information-extraction-on-ephoie","dataset":"EPHOIE","dataset_url":"/dataset/ephoie","rows_in_archive":1,"metrics":["Average F1"],"first_row_in_archive_order":{"model":"LayoutLMv3","paper_title":"LayoutLMv3: Pre-training for Document AI with Unified Text and Image Masking","paper_url":"/paper/layoutlmv3-pre-training-for-document-ai-with","paper_date":"2022-04-18","arxiv_id":"2204.08387","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"microsoft/unilm","url":"https://github.com/microsoft/unilm/tree/master/layoutlmv3"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/layoutlmv3"},{"title":"MindSpore-scientific-2/code-14","url":"https://github.com/MindSpore-scientific-2/code-14/tree/main/layoutlmv3"}],"syntology":null}},{"leaderboard":"/sota/key-information-extraction-on-etd500","slug":"key-information-extraction-on-etd500","dataset":"ETD500","dataset_url":"/dataset/etd500","rows_in_archive":1,"metrics":["F1 (%)"],"first_row_in_archive_order":{"model":"CRF-visual","paper_title":"Automatic Metadata Extraction Incorporating Visual Features from Scanned Electronic Theses and Dissertations","paper_url":"/paper/automatic-metadata-extraction-incorporating","paper_date":"2021-07-01","arxiv_id":"2107.00516","code_links":[{"title":"lamps-lab/ETDMiner","url":"https://github.com/lamps-lab/ETDMiner"},{"title":"lamps-lab/AutoMeta","url":"https://github.com/lamps-lab/AutoMeta"}],"syntology":null}},{"leaderboard":"/sota/key-information-extraction-on-simara","slug":"key-information-extraction-on-simara","dataset":"SIMARA","dataset_url":"/dataset/simara","rows_in_archive":1,"metrics":["F1 (%)"],"first_row_in_archive_order":{"model":"DAN","paper_title":"SIMARA: a database for key-value information extraction from full pages","paper_url":"/paper/simara-a-database-for-key-value-information","paper_date":"2023-04-26","arxiv_id":"2304.13606","code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/sroie","name":"SROIE","full_name":"","num_papers_in_archive":105},{"url":"/dataset/cord","name":"CORD","full_name":"Consolidated Receipt Dataset for Post-OCR Parsing","num_papers_in_archive":100},{"url":"/dataset/ephoie","name":"EPHOIE","full_name":"phtnsantader@gmail.com","num_papers_in_archive":21},{"url":"/dataset/kleister-nda","name":"Kleister NDA","full_name":"","num_papers_in_archive":17},{"url":"/dataset/sibr","name":"SIBR","full_name":"SIBR Dataset for VIE in the Wild","num_papers_in_archive":14},{"url":"/dataset/docile","name":"DocILE","full_name":"","num_papers_in_archive":11},{"url":"/dataset/information-extraction-from-tables","name":"Information Extraction from Tables","full_name":"Extraction materials compositions from tables of materials science research papers","num_papers_in_archive":3},{"url":"/dataset/etd500","name":"ETD500","full_name":"","num_papers_in_archive":2},{"url":"/dataset/simara","name":"SIMARA","full_name":"SIMARA: a database for key-value information extraction from full-page handwritten documents","num_papers_in_archive":2},{"url":"/dataset/poie","name":"POIE","full_name":"Products for OCR and Information Extraction","num_papers_in_archive":1},{"url":"/dataset/somd","name":"SOMD","full_name":"SOftware Mention Detection","num_papers_in_archive":1},{"url":"/dataset/arf","name":"ARF","full_name":"Artificial Relationships in Fiction","num_papers_in_archive":0}],"subtasks":[{"url":"/task/key-value-pair-extraction","name":"Key-value Pair Extraction"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":40,"tagged_in_all":74,"items":[{"url":"/paper/layoutlm-pre-training-of-text-and-layout-for","title":"LayoutLM: Pre-training of Text and Layout for Document Image Understanding","date":"2019-12-31","arxiv_id":"1912.13318","repositories_listed":19,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/layoutlmv2-multi-modal-pre-training-for","title":"LayoutLMv2: Multi-modal Pre-training for Visually-Rich Document Understanding","date":"2020-12-29","arxiv_id":"2012.14740","repositories_listed":9,"syntology":null},{"url":"/paper/lilt-a-simple-yet-effective-language","title":"LiLT: A Simple yet Effective Language-Independent Layout Transformer for Structured Document Understanding","date":"2022-02-28","arxiv_id":"2202.13669","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/layoutlmv3-pre-training-for-document-ai-with","title":"LayoutLMv3: Pre-training for Document AI with Unified Text and Image Masking","date":"2022-04-18","arxiv_id":"2204.08387","repositories_listed":4,"syntology":null},{"url":"/paper/xformparser-a-simple-and-effective-multimodal","title":"XFormParser: A Simple and Effective Multimodal Multilingual Semi-structured Form Parser","date":"2024-05-27","arxiv_id":"2405.17336","repositories_listed":2,"syntology":null},{"url":"/paper/reading-order-matters-information-extraction","title":"Reading Order Matters: Information Extraction from Visually-rich Documents by Token Path Prediction","date":"2023-10-17","arxiv_id":"2310.11016","repositories_listed":2,"syntology":null},{"url":"/paper/ernie-layout-layout-knowledge-enhanced-pre","title":"ERNIE-Layout: Layout Knowledge Enhanced Pre-training for Visually-rich Document Understanding","date":"2022-10-12","arxiv_id":"2210.06155","repositories_listed":2,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/bros-a-layout-aware-pre-trained-language","title":"BROS: A Pre-trained Language Model Focusing on Text and Layout for Better Key Information Extraction from Documents","date":"2021-08-10","arxiv_id":"2108.04539","repositories_listed":2,"syntology":{"n":13,"n_ran":2,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/automatic-metadata-extraction-incorporating","title":"Automatic Metadata Extraction Incorporating Visual Features from Scanned Electronic Theses and Dissertations","date":"2021-07-01","arxiv_id":"2107.00516","repositories_listed":2,"syntology":null},{"url":"/paper/spatial-dual-modality-graph-reasoning-for-key","title":"Spatial Dual-Modality Graph Reasoning for Key Information Extraction","date":"2021-03-26","arxiv_id":"2103.14470","repositories_listed":2,"syntology":null},{"url":"/paper/pick-processing-key-information-extraction","title":"PICK: Processing Key Information Extraction from Documents using Improved Graph Learning-Convolutional Networks","date":"2020-04-16","arxiv_id":"2004.07464","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/paddleocr-3-0-technical-report","title":"PaddleOCR 3.0 Technical Report","date":"2025-07-08","arxiv_id":"2507.05595","repositories_listed":1,"syntology":null},{"url":"/paper/class-agnostic-region-of-interest-matching-in","title":"Class-Agnostic Region-of-Interest Matching in Document Images","date":"2025-06-26","arxiv_id":"2506.21055","repositories_listed":1,"syntology":null},{"url":"/paper/omniparser-v2-structured-points-of-thought","title":"OmniParser V2: Structured-Points-of-Thought for Unified Visual Text Parsing and Its Generality to Multimodal Large Language Models","date":"2025-02-22","arxiv_id":"2502.16161","repositories_listed":1,"syntology":null},{"url":"/paper/graphrevisedie-multimodal-information","title":"GraphRevisedIE: Multimodal Information Extraction with Graph-Revised Network","date":"2024-10-02","arxiv_id":"2410.01160","repositories_listed":1,"syntology":null},{"url":"/paper/modeling-layout-reading-order-as-ordering","title":"Modeling Layout Reading Order as Ordering Relations for Visually-rich Document Understanding","date":"2024-09-29","arxiv_id":"2409.19672","repositories_listed":1,"syntology":null},{"url":"/paper/information-extraction-from-visually-rich-1","title":"Information Extraction from Visually Rich Documents Using Directed Weighted Graph Neural Network","date":"2024-09-11","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-bounding-box-is-worth-one-token","title":"A Bounding Box is Worth One Token: Interleaving Layout and Text in a Large Language Model for Document Understanding","date":"2024-07-02","arxiv_id":"2407.01976","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_unverified":0,"n_pointer_only":7}},{"url":"/paper/kvp10k-a-comprehensive-dataset-for-key-value","title":"KVP10k : A Comprehensive Dataset for Key-Value Pair Extraction in Business Documents","date":"2024-05-01","arxiv_id":"2405.00505","repositories_listed":1,"syntology":null},{"url":"/paper/omniparser-a-unified-framework-for-text","title":"OmniParser: A Unified Framework for Text Spotting, Key Information Extraction and Table Recognition","date":"2024-03-28","arxiv_id":"2403.19128","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/textmonkey-an-ocr-free-large-multimodal-model","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","date":"2024-03-07","arxiv_id":"2403.04473","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/different-tastes-of-entities-investigating","title":"Different Tastes of Entities: Investigating Human Label Variation in Named Entity Annotations","date":"2024-02-02","arxiv_id":"2402.01423","repositories_listed":1,"syntology":null},{"url":"/paper/peneo-unifying-line-extraction-line-grouping","title":"PEneo: Unifying Line Extraction, Line Grouping, and Entity Linking for End-to-end Document Pair Extraction","date":"2024-01-07","arxiv_id":"2401.03472","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-weighted-graph-representation-for","title":"Multimodal weighted graph representation for information extraction from visually rich documents.","date":"2024-01-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/omniparser-a-unified-framework-for-text-1","title":"OmniParser: A Unified Framework for Text Spotting Key Information Extraction and Table Recognition","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/exploring-ocr-capabilities-of-gpt-4v-ision-a","title":"Exploring OCR Capabilities of GPT-4V(ision) : A Quantitative and In-depth Evaluation","date":"2023-10-25","arxiv_id":"2310.16809","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/genkie-robust-generative-multimodal-document","title":"GenKIE: Robust Generative Multimodal Document Key Information Extraction","date":"2023-10-24","arxiv_id":"2310.16131","repositories_listed":1,"syntology":null},{"url":"/paper/amurd-annotated-multilingual-receipts-dataset","title":"AMuRD: Annotated Arabic-English Receipt Dataset for Key Information Extraction and Classification","date":"2023-09-18","arxiv_id":"2309.09800","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-hidden-mystery-of-ocr-in-large","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","date":"2023-05-13","arxiv_id":"2305.07895","repositories_listed":1,"syntology":null},{"url":"/paper/information-redundancy-and-biases-in-public","title":"Information Redundancy and Biases in Public Document Information Extraction Benchmarks","date":"2023-04-28","arxiv_id":"2304.14936","repositories_listed":1,"syntology":null}],"syntology_records":9,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}