{"url":"/dataset/sroie","name":"SROIE","full_name":null,"description_markdown":"Consists of a dataset with 1000 whole scanned receipt images and annotations for the competition on scanned receipts OCR and key information extraction (SROIE).\r\n\r\nImage source: [https://arxiv.org/pdf/2103.10213.pdf](https://arxiv.org/pdf/2103.10213.pdf)","description_withheld":null,"homepage":"https://arxiv.org/pdf/2103.10213.pdf","introduced_date":"2021-03-18","introduced_date_note":null,"introduced_by":{"paper":"/paper/icdar2019-competition-on-scanned-receipt-ocr","title":"ICDAR2019 Competition on Scanned Receipt OCR and Information Extraction","first_author":"Zheng Huang","url":null},"license":null,"modalities":[],"tasks":[{"name":"Token Classification","url":"/task/token-classification","datasets_with_task":"/datasets/task/token-classification"},{"name":"Key Information Extraction","url":"/task/key-information-extraction","datasets_with_task":"/datasets/task/key-information-extraction"},{"name":"Task 2","url":"/task/task-2","datasets_with_task":"/datasets/task/task-2"}],"languages":[{"name":"Spanish","url":"/datasets/language/spanish"}],"variants":["SROIE"],"data_loaders":[{"repo":"https://github.com/mindee/doctr","url":"https://mindee.github.io/doctr/latest/datasets.html#doctr.datasets.SROIE","frameworks":["tf","pytorch"]}],"num_papers_in_archive":105,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/key-information-extraction-on-sroie","task":"Key Information Extraction","dataset_variant":"SROIE","rows":5,"metrics":["F1","Accuracy"],"first_row_in_archive_order":{"model":"LayoutLMv2LARGE (Excluding OCR mismatch)","paper":"/paper/layoutlmv2-multi-modal-pre-training-for","metrics":{"F1":"97.81"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"PaddlePaddle/PaddleOCR","url":"https://github.com/PaddlePaddle/PaddleOCR"},{"title":"microsoft/unilm","url":"https://github.com/microsoft/unilm"},{"title":"PaddlePaddle/PaddleNLP","url":"https://github.com/PaddlePaddle/PaddleNLP/tree/develop/paddlenlp/transformers/layoutlmv2"},{"title":"facebookresearch/data2vec_vision","url":"https://github.com/facebookresearch/data2vec_vision"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/layoutlmv2"},{"title":"MS-P3/code3","url":"https://github.com/MS-P3/code3/tree/main/layoutlmv2"},{"title":"MindSpore-scientific-2/code-14","url":"https://github.com/MindSpore-scientific-2/code-14/tree/main/layoutlmv2"},{"title":"MindSpore-scientific/code-7","url":"https://github.com/MindSpore-scientific/code-7/tree/main/LayoutLMv2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/token-classification-on-sroie","task":"Token Classification","dataset_variant":"SROIE","rows":0,"metrics":["Accuracy","F1","Precision","Recall"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/modeling-layout-reading-order-as-ordering","title":"Modeling Layout Reading Order as Ordering Relations for Visually-rich Document Understanding","date":"2024-09-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/lapdoc-layout-aware-prompting-for-documents","title":"LAPDoc: Layout-Aware Prompting for Documents","date":"2024-02-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/layoutlmv2-multi-modal-pre-training-for","title":"LayoutLMv2: Multi-modal Pre-training for Visually-Rich Document Understanding","date":"2020-12-29","rows_on_this_dataset":3,"code_links":9,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}