{"url":"/dataset/sibr","name":"SIBR","full_name":"SIBR Dataset for VIE in the Wild","description_markdown":"SIBR是面向自然场景视觉信息抽取的数据集。\r\n\r\n1）SIBR总的有1000张图片，400张测试，600张训练，包括中文、英文两种语言。\r\n2）包含images.zip、label.zip、train.txt、test.txt四个文件，images.zip、label.zip中包含所有图片和标签，通过train.txt和test.txt区分训练和测试。\r\n3）标注规则与FUNSD、XFUND一致，实体类别包括header、question、answer、other四类，但与之不同的是除了提供link id之外，还标注了link type。link type包括inter以及intra两种，inter表示实体间的link，也即是一对kv对之间的link；而intra表示同一个实体内部segment之间的link。通过intra将多行文字组成一个实体，通过inter将实体组成kv对。","description_withheld":null,"homepage":"https://www.modelscope.cn/datasets/iic/SIBR/summary","introduced_date":"2023-03-23","introduced_date_note":null,"introduced_by":{"paper":"/paper/modeling-entities-as-semantic-points-for","title":"Modeling Entities as Semantic Points for Visual Information Extraction in the Wild","first_author":"Zhibo Yang","url":null},"license":{"name":"Apache License Version 2.0","url":"https://www.apache.org/licenses/LICENSE-2.0.html"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Key Information Extraction","url":"/task/key-information-extraction","datasets_with_task":"/datasets/task/key-information-extraction"},{"name":"Key-value Pair Extraction","url":"/task/key-value-pair-extraction","datasets_with_task":"/datasets/task/key-value-pair-extraction"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":[],"data_loaders":[{"repo":"https://github.com/maybeyouorme/MEF_Data","url":"https://github.com/maybeyouorme/MEF_Data","frameworks":["tf"]}],"num_papers_in_archive":14,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/key-value-pair-extraction-on-sibr","task":"Key-value Pair Extraction","dataset_variant":"SIBR","rows":7,"metrics":["key-value pair F1"],"first_row_in_archive_order":{"model":"PEneo\n(LayoutLMv3_base_chinese)","paper":"/paper/peneo-unifying-line-extraction-line-grouping","metrics":{"key-value pair F1":"82.52"},"code_links":[{"title":"ZeningLin/PEneo","url":"https://github.com/ZeningLin/PEneo"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/peneo-unifying-line-extraction-line-grouping","title":"PEneo: Unifying Line Extraction, Line Grouping, and Entity Linking for End-to-end Document Pair Extraction","date":"2024-01-07","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/layoutlmv3-pre-training-for-document-ai-with","title":"LayoutLMv3: Pre-training for Document AI with Unified Text and Image Masking","date":"2022-04-18","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/lilt-a-simple-yet-effective-language","title":"LiLT: A Simple yet Effective Language-Independent Layout Transformer for Structured Document Understanding","date":"2022-02-28","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/donut-document-understanding-transformer","title":"OCR-free Document Understanding Transformer","date":"2021-11-30","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/layoutxlm-multimodal-pre-training-for","title":"LayoutXLM: Multimodal Pre-training for Multilingual Visually-rich Document Understanding","date":"2021-04-18","rows_on_this_dataset":1,"code_links":6,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":12,"samples_ran":2,"samples_unverified":10,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}