{"url":"/task/document-understanding","name":"document understanding","slug":"document-understanding","description_markdown":"Document understanding involves document classification, layout analysis, information extraction, and DocQA.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":309,"papers_with_code":140,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":2,"subtasks":1,"parent_tasks":1},"benchmarks":[],"datasets":[{"url":"/dataset/pdfvqa","name":"PDFVQA","full_name":"","num_papers_in_archive":3},{"url":"/dataset/u-diads-bib","name":"U-DIADS-Bib","full_name":"","num_papers_in_archive":2}],"subtasks":[{"url":"/task/line-items-extraction","name":"Line Items Extraction"}],"parent_tasks":[{"url":"/task/document-ai","name":"Document AI"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":140,"tagged_in_all":309,"items":[{"url":"/paper/layoutlmv2-multi-modal-pre-training-for","title":"LayoutLMv2: Multi-modal Pre-training for Visually-Rich Document Understanding","date":"2020-12-29","arxiv_id":"2012.14740","repositories_listed":9,"syntology":null},{"url":"/paper/layoutxlm-multimodal-pre-training-for","title":"LayoutXLM: Multimodal Pre-training for Multilingual Visually-rich Document Understanding","date":"2021-04-18","arxiv_id":"2104.08836","repositories_listed":6,"syntology":null},{"url":"/paper/colpali-efficient-document-retrieval-with","title":"ColPali: Efficient Document Retrieval with Vision Language Models","date":"2024-06-27","arxiv_id":"2407.01449","repositories_listed":5,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/unifying-vision-text-and-layout-for-universal","title":"Unifying Vision, Text, and Layout for Universal Document Processing","date":"2022-12-05","arxiv_id":"2212.02623","repositories_listed":5,"syntology":{"n":17,"n_ran":4,"n_unverified":13,"n_pointer_only":3}},{"url":"/paper/lilt-a-simple-yet-effective-language","title":"LiLT: A Simple yet Effective Language-Independent Layout Transformer for Structured Document Understanding","date":"2022-02-28","arxiv_id":"2202.13669","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/donut-document-understanding-transformer","title":"OCR-free Document Understanding Transformer","date":"2021-11-30","arxiv_id":"2111.15664","repositories_listed":5,"syntology":{"n":9,"n_ran":0,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/chargrid-towards-understanding-2d-documents","title":"Chargrid: Towards Understanding 2D Documents","date":"2018-09-24","arxiv_id":"1809.08799","repositories_listed":5,"syntology":null},{"url":"/paper/qwen2-5-vl-technical-report","title":"Qwen2.5-VL Technical Report","date":"2025-02-19","arxiv_id":"2502.13923","repositories_listed":4,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/icdar-2021-competition-on-scientific","title":"ICDAR 2021 Competition on Scientific Literature Parsing","date":"2021-06-08","arxiv_id":"2106.14616","repositories_listed":3,"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":1}},{"url":"/paper/learned-compression-for-compressed-learning","title":"Learned Compression for Compressed Learning","date":"2024-12-12","arxiv_id":"2412.09405","repositories_listed":2,"syntology":null},{"url":"/paper/doclayout-yolo-enhancing-document-layout","title":"DocLayout-YOLO: Enhancing Document Layout Analysis through Diverse Synthetic Data and Global-to-Local Adaptive Perception","date":"2024-10-16","arxiv_id":"2410.12628","repositories_listed":2,"syntology":null},{"url":"/paper/layoutllm-layout-instruction-tuning-with","title":"LayoutLLM: Layout Instruction Tuning with Large Language Models for Document Understanding","date":"2024-04-08","arxiv_id":"2404.05225","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/on-the-affinity-rationality-and-diversity-of","title":"On the Affinity, Rationality, and Diversity of Hierarchical Topic Modeling","date":"2024-01-25","arxiv_id":"2401.14113","repositories_listed":2,"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":6}},{"url":"/paper/pali-x-on-scaling-up-a-multilingual-vision","title":"PaLI-X: On Scaling up a Multilingual Vision and Language Model","date":"2023-05-29","arxiv_id":"2305.18565","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/ernie-layout-layout-knowledge-enhanced-pre","title":"ERNIE-Layout: Layout Knowledge Enhanced Pre-training for Visually-rich Document Understanding","date":"2022-10-12","arxiv_id":"2210.06155","repositories_listed":2,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/end-to-end-document-recognition-and","title":"End-to-end Document Recognition and Understanding with Dessurt","date":"2022-03-30","arxiv_id":"2203.16618","repositories_listed":2,"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/markuplm-pre-training-of-text-and-markup","title":"MarkupLM: Pre-training of Text and Markup Language for Visually-rich Document Understanding","date":"2021-10-16","arxiv_id":"2110.08518","repositories_listed":2,"syntology":null},{"url":"/paper/message-passing-attention-networks-for","title":"Message Passing Attention Networks for Document Understanding","date":"2019-08-17","arxiv_id":"1908.06267","repositories_listed":2,"syntology":null},{"url":"/paper/paddleocr-3-0-technical-report","title":"PaddleOCR 3.0 Technical Report","date":"2025-07-08","arxiv_id":"2507.05595","repositories_listed":1,"syntology":null},{"url":"/paper/glm-4-1v-thinking-towards-versatile","title":"GLM-4.1V-Thinking: Towards Versatile Multimodal Reasoning with Scalable Reinforcement Learning","date":"2025-07-01","arxiv_id":"2507.01006","repositories_listed":1,"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/class-agnostic-region-of-interest-matching-in","title":"Class-Agnostic Region-of-Interest Matching in Document Images","date":"2025-06-26","arxiv_id":"2506.21055","repositories_listed":1,"syntology":null},{"url":"/paper/drishtikon-multi-granular-visual-grounding","title":"DrishtiKon: Multi-Granular Visual Grounding for Text-Rich Document Images","date":"2025-06-26","arxiv_id":"2506.21316","repositories_listed":1,"syntology":null},{"url":"/paper/pp-docbee2-improved-baselines-with-efficient","title":"PP-DocBee2: Improved Baselines with Efficient Data for Multimodal Document Understanding","date":"2025-06-22","arxiv_id":"2506.18023","repositories_listed":1,"syntology":null},{"url":"/paper/simpledoc-multi-modal-document-understanding","title":"SimpleDoc: Multi-Modal Document Understanding with Dual-Cue Page Retrieval and Iterative Refinement","date":"2025-06-16","arxiv_id":"2506.14035","repositories_listed":1,"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":11}},{"url":"/paper/lemonade-a-large-multilingual-expert","title":"LEMONADE: A Large Multilingual Expert-Annotated Abstractive Event Dataset for the Real World","date":"2025-06-01","arxiv_id":"2506.00980","repositories_listed":1,"syntology":null},{"url":"/paper/infinity-parser-layout-aware-reinforcement","title":"Infinity Parser: Layout Aware Reinforcement Learning for Scanned Document Parsing","date":"2025-06-01","arxiv_id":"2506.03197","repositories_listed":1,"syntology":null},{"url":"/paper/arb-a-comprehensive-arabic-multimodal","title":"ARB: A Comprehensive Arabic Multimodal Reasoning Benchmark","date":"2025-05-22","arxiv_id":"2505.17021","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-markup-language-generation-for","title":"Adaptive Markup Language Generation for Contextually-Grounded Visual Document Understanding","date":"2025-05-08","arxiv_id":"2505.05446","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/frag-frame-selection-augmented-generation-for","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","date":"2025-04-24","arxiv_id":"2504.17447","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-menu-ocr-and-translation-a","title":"Evaluating Menu OCR and Translation: A Benchmark for Aligning Human and Automated Evaluations in Large Vision-Language Models","date":"2025-04-16","arxiv_id":"2504.13945","repositories_listed":1,"syntology":null}],"syntology_records":14,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}