{"url":"/dataset/rfund-en","name":"RFUND-EN","full_name":null,"description_markdown":"English subset of RFUND","description_withheld":null,"homepage":"https://github.com/SCUT-DLVCLab/RFUND","introduced_date":"2024-01-07","introduced_date_note":null,"introduced_by":{"paper":"/paper/peneo-unifying-line-extraction-line-grouping","title":"PEneo: Unifying Line Extraction, Line Grouping, and Entity Linking for End-to-end Document Pair Extraction","first_author":"Zening Lin","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Key-value Pair Extraction","url":"/task/key-value-pair-extraction","datasets_with_task":"/datasets/task/key-value-pair-extraction"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["RFUND-EN"],"data_loaders":[],"num_papers_in_archive":8,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/key-value-pair-extraction-on-rfund-en","task":"Key-value Pair Extraction","dataset_variant":"RFUND-EN","rows":13,"metrics":["key-value pair F1"],"first_row_in_archive_order":{"model":"PEneo\n(LayoutLMv3_base)","paper":"/paper/peneo-unifying-line-extraction-line-grouping","metrics":{"key-value pair F1":"79.27"},"code_links":[{"title":"ZeningLin/PEneo","url":"https://github.com/ZeningLin/PEneo"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/peneo-unifying-line-extraction-line-grouping","title":"PEneo: Unifying Line Extraction, Line Grouping, and Entity Linking for End-to-end Document Pair Extraction","date":"2024-01-07","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/reading-order-matters-information-extraction","title":"Reading Order Matters: Information Extraction from Visually-rich Documents by Token Path Prediction","date":"2023-10-17","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/geolayoutlm-geometric-pre-training-for-visual","title":"GeoLayoutLM: Geometric Pre-training for Visual Information Extraction","date":"2023-04-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/layoutlmv3-pre-training-for-document-ai-with","title":"LayoutLMv3: Pre-training for Document AI with Unified Text and Image Masking","date":"2022-04-18","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/lilt-a-simple-yet-effective-language","title":"LiLT: A Simple yet Effective Language-Independent Layout Transformer for Structured Document Understanding","date":"2022-02-28","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/donut-document-understanding-transformer","title":"OCR-free Document Understanding Transformer","date":"2021-11-30","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/layoutxlm-multimodal-pre-training-for","title":"LayoutXLM: Multimodal Pre-training for Multilingual Visually-rich Document Understanding","date":"2021-04-18","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/layoutlmv2-multi-modal-pre-training-for","title":"LayoutLMv2: Multi-modal Pre-training for Visually-Rich Document Understanding","date":"2020-12-29","rows_on_this_dataset":1,"code_links":9,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":12,"samples_ran":2,"samples_unverified":10,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}