{"url":"/task/document-ai","name":"Document AI","slug":"document-ai","description_markdown":null,"categories":[{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":40,"papers_with_code":24,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/document-ai-on-ephoie","slug":"document-ai-on-ephoie","dataset":"EPHOIE","dataset_url":"/dataset/ephoie","rows_in_archive":1,"metrics":["Average F1"],"first_row_in_archive_order":{"model":"LayoutLMv3","paper_title":"LayoutLMv3: Pre-training for Document AI with Unified Text and Image Masking","paper_url":"/paper/layoutlmv3-pre-training-for-document-ai-with","paper_date":"2022-04-18","arxiv_id":"2204.08387","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"microsoft/unilm","url":"https://github.com/microsoft/unilm/tree/master/layoutlmv3"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/layoutlmv3"},{"title":"MindSpore-scientific-2/code-14","url":"https://github.com/MindSpore-scientific-2/code-14/tree/main/layoutlmv3"}],"syntology":null}}],"datasets":[{"url":"/dataset/ephoie","name":"EPHOIE","full_name":"phtnsantader@gmail.com","num_papers_in_archive":21}],"subtasks":[{"url":"/task/document-understanding","name":"document understanding"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":24,"of":24,"tagged_in_all":40,"items":[{"url":"/paper/layoutlm-pre-training-of-text-and-layout-for","title":"LayoutLM: Pre-training of Text and Layout for Document Image Understanding","date":"2019-12-31","arxiv_id":"1912.13318","repositories_listed":19,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/unifying-vision-text-and-layout-for-universal","title":"Unifying Vision, Text, and Layout for Universal Document Processing","date":"2022-12-05","arxiv_id":"2212.02623","repositories_listed":5,"syntology":{"n":17,"n_ran":4,"n_unverified":13,"n_pointer_only":3}},{"url":"/paper/layoutlmv3-pre-training-for-document-ai-with","title":"LayoutLMv3: Pre-training for Document AI with Unified Text and Image Masking","date":"2022-04-18","arxiv_id":"2204.08387","repositories_listed":4,"syntology":null},{"url":"/paper/dit-self-supervised-pre-training-for-document","title":"DiT: Self-supervised Pre-training for Document Image Transformer","date":"2022-03-04","arxiv_id":"2203.02378","repositories_listed":4,"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/xformparser-a-simple-and-effective-multimodal","title":"XFormParser: A Simple and Effective Multimodal Multilingual Semi-structured Form Parser","date":"2024-05-27","arxiv_id":"2405.17336","repositories_listed":2,"syntology":null},{"url":"/paper/layoutllm-layout-instruction-tuning-with","title":"LayoutLLM: Layout Instruction Tuning with Large Language Models for Document Understanding","date":"2024-04-08","arxiv_id":"2404.05225","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/multimodal-machine-learning-for-extraction-of","title":"Modular Multimodal Machine Learning for Extraction of Theorems and Proofs in Long Scientific Documents (Extended Version)","date":"2023-07-18","arxiv_id":"2307.09047","repositories_listed":2,"syntology":null},{"url":"/paper/infinity-parser-layout-aware-reinforcement","title":"Infinity Parser: Layout Aware Reinforcement Learning for Scanned Document Parsing","date":"2025-06-01","arxiv_id":"2506.03197","repositories_listed":1,"syntology":null},{"url":"/paper/out-of-distribution-detection-with-attention","title":"Out-of-Distribution Detection with Attention Head Masking for Multimodal Document Classification","date":"2024-08-20","arxiv_id":"2408.11237","repositories_listed":1,"syntology":null},{"url":"/paper/design-of-a-quality-management-system-based","title":"Design of a Quality Management System based on the EU Artificial Intelligence Act","date":"2024-08-08","arxiv_id":"2408.04689","repositories_listed":1,"syntology":null},{"url":"/paper/officebench-benchmarking-language-agents","title":"OfficeBench: Benchmarking Language Agents across Multiple Applications for Office Automation","date":"2024-07-26","arxiv_id":"2407.19056","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/on-efficient-language-and-vision-assistants","title":"On Efficient Language and Vision Assistants for Visually-Situated Natural Language Understanding: What Matters in Reading and Reasoning","date":"2024-06-17","arxiv_id":"2406.11823","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_unverified":2,"n_pointer_only":8}},{"url":"/paper/docres-a-generalist-model-toward-unifying","title":"DocRes: A Generalist Model Toward Unifying Document Image Restoration Tasks","date":"2024-05-07","arxiv_id":"2405.04408","repositories_listed":1,"syntology":null},{"url":"/paper/doctrack-a-visually-rich-document-dataset","title":"DocTrack: A Visually-Rich Document Dataset Really Aligned with Human Eye Movement for Machine Reading","date":"2023-10-23","arxiv_id":"2310.14802","repositories_listed":1,"syntology":null},{"url":"/paper/docxchain-a-powerful-open-source-toolchain","title":"DocXChain: A Powerful Open-Source Toolchain for Document Parsing and Beyond","date":"2023-10-19","arxiv_id":"2310.12430","repositories_listed":1,"syntology":null},{"url":"/paper/vision-grid-transformer-for-document-layout","title":"Vision Grid Transformer for Document Layout Analysis","date":"2023-08-29","arxiv_id":"2308.14978","repositories_listed":1,"syntology":null},{"url":"/paper/document-ai-a-comparative-study-of","title":"Document AI: A Comparative Study of Transformer-Based, Graph-Based Models, and Convolutional Neural Networks For Document Layout Analysis","date":"2023-08-29","arxiv_id":"2308.15517","repositories_listed":1,"syntology":null},{"url":"/paper/document-understanding-dataset-and-evaluation","title":"Document Understanding Dataset and Evaluation (DUDE)","date":"2023-05-15","arxiv_id":"2305.08455","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/context-aware-chart-element-detection","title":"Context-Aware Chart Element Detection","date":"2023-05-07","arxiv_id":"2305.04151","repositories_listed":1,"syntology":null},{"url":"/paper/geolayoutlm-geometric-pre-training-for-visual","title":"GeoLayoutLM: Geometric Pre-training for Visual Information Extraction","date":"2023-04-21","arxiv_id":"2304.10759","repositories_listed":1,"syntology":null},{"url":"/paper/icl-d3ie-in-context-learning-with-diverse","title":"ICL-D3IE: In-Context Learning with Diverse Demonstrations Updating for Document Information Extraction","date":"2023-03-09","arxiv_id":"2303.05063","repositories_listed":1,"syntology":null},{"url":"/paper/dosa-a-system-to-accelerate-annotations-on","title":"DoSA : A System to Accelerate Annotations on Business Documents with Human-in-the-Loop","date":"2022-11-09","arxiv_id":"2211.04934","repositories_listed":1,"syntology":null},{"url":"/paper/dimsum-distributed-and-multilingual","title":"DiMSum: Distributed and Multilingual Summarization of Financial Narratives","date":"2022-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/document-intelligence-metrics-for-visually","title":"Document Intelligence Metrics for Visually Rich Document Evaluation","date":"2022-05-23","arxiv_id":"2205.11215","repositories_listed":1,"syntology":null}],"syntology_records":7,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}