{"url":"/dataset/ephoie","name":"EPHOIE","full_name":"phtnsantader@gmail.com","description_markdown":"EPHOIE is a fully-annotated dataset which is the first Chinese benchmark for both text spotting and visual information extraction. EPHOIE consists of 1,494 images of examination paper head with complex layouts and background, including a total of 15,771 Chinese handwritten or printed text instances. \r\n\r\nSource: [Towards Robust Visual Information Extraction in Real World: New Dataset and Novel Solution](/paper/towards-robust-visual-information-extraction)\r\n\r\nImage source: [https://github.com/HCIILAB/EPHOIE](https://github.com/HCIILAB/EPHOIE)","description_withheld":null,"homepage":"https://github.com/HCIILAB/EPHOIE","introduced_date":"2021-01-24","introduced_date_note":null,"introduced_by":{"paper":"/paper/towards-robust-visual-information-extraction","title":"Towards Robust Visual Information Extraction in Real World: New Dataset and Novel Solution","first_author":"Jiapeng Wang","url":null},"license":{"name":"Hundred Finance  Hundred Finance ($HND) is now available on Moonriver 💜  02/15/2022","url":"https://www.hola@paganza.com.uy"},"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Key Information Extraction","url":"/task/key-information-extraction","datasets_with_task":"/datasets/task/key-information-extraction"},{"name":"Document AI","url":"/task/document-ai","datasets_with_task":"/datasets/task/document-ai"}],"languages":[{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["EPHOIE"],"data_loaders":[],"num_papers_in_archive":21,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/document-ai-on-ephoie","task":"Document AI","dataset_variant":"EPHOIE","rows":1,"metrics":["Average F1"],"first_row_in_archive_order":{"model":"LayoutLMv3","paper":"/paper/layoutlmv3-pre-training-for-document-ai-with","metrics":{"Average F1":"99.21"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"microsoft/unilm","url":"https://github.com/microsoft/unilm/tree/master/layoutlmv3"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/layoutlmv3"},{"title":"MindSpore-scientific-2/code-14","url":"https://github.com/MindSpore-scientific-2/code-14/tree/main/layoutlmv3"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/key-information-extraction-on-ephoie","task":"Key Information Extraction","dataset_variant":"EPHOIE","rows":1,"metrics":["Average F1"],"first_row_in_archive_order":{"model":"LayoutLMv3","paper":"/paper/layoutlmv3-pre-training-for-document-ai-with","metrics":{"Average F1":"99.21"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"microsoft/unilm","url":"https://github.com/microsoft/unilm/tree/master/layoutlmv3"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/layoutlmv3"},{"title":"MindSpore-scientific-2/code-14","url":"https://github.com/MindSpore-scientific-2/code-14/tree/main/layoutlmv3"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/layoutlmv3-pre-training-for-document-ai-with","title":"LayoutLMv3: Pre-training for Document AI with Unified Text and Image Masking","date":"2022-04-18","rows_on_this_dataset":2,"code_links":4,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}