{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/document-understanding/papers/2","list_of":"/task/document-understanding","task":"document understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":309,"counts":{"archive_papers_tagged":309,"with_a_code_link":140,"where_syntology_ran_a_sample":39,"not_listed_spam_title":0,"listed":309,"listed_where_code_ran":39,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":30,"every_run_a_failure_of_syntologys_instrument":9,"listed_with_a_run_with_no_instrument_failure":30,"listed_every_run_a_failure_of_syntologys_instrument":9,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/document-understanding","prev":"/task/document-understanding","next":"/task/document-understanding/papers/3","papers":[{"url":"/paper/lineformer-rethinking-line-chart-data","slug":"lineformer-rethinking-line-chart-data","title":"LineFormer: Rethinking Line Chart Data Extraction as Instance Segmentation","date":"2023-05-03","arxiv_id":"2305.01837","repositories_listed":1,"syntology":null},{"url":"/paper/ccpdf-building-a-high-quality-corpus-for","slug":"ccpdf-building-a-high-quality-corpus-for","title":"CCpdf: Building a High Quality Corpus for Visually Rich Documents from Web Crawl Data","date":"2023-04-28","arxiv_id":"2304.14953","repositories_listed":1,"syntology":null},{"url":"/paper/information-redundancy-and-biases-in-public","slug":"information-redundancy-and-biases-in-public","title":"Information Redundancy and Biases in Public Document Information Extraction Benchmarks","date":"2023-04-28","arxiv_id":"2304.14936","repositories_listed":1,"syntology":null},{"url":"/paper/is-chatgpt-a-good-keyphrase-generator-a","slug":"is-chatgpt-a-good-keyphrase-generator-a","title":"Is ChatGPT A Good Keyphrase Generator? A Preliminary Study","date":"2023-03-23","arxiv_id":"2303.13001","repositories_listed":1,"syntology":null},{"url":"/paper/m6doc-a-large-scale-multi-format-multi-type","slug":"m6doc-a-large-scale-multi-format-multi-type","title":"M6Doc: A Large-Scale Multi-Format, Multi-Type, Multi-Layout, Multi-Language, Multi-Annotation Category Dataset for Modern Document Layout Analysis","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/wukong-reader-multi-modal-pre-training-for","slug":"wukong-reader-multi-modal-pre-training-for","title":"Wukong-Reader: Multi-modal Pre-training for Fine-grained Visual Document Understanding","date":"2022-12-19","arxiv_id":"2212.09621","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-tree-decoder-for-table-of-contents","slug":"multimodal-tree-decoder-for-table-of-contents","title":"Multimodal Tree Decoder for Table of Contents Extraction in Document Images","date":"2022-12-06","arxiv_id":"2212.02896","repositories_listed":1,"syntology":null},{"url":"/paper/technical-report-on-web-based-visual-corpus","slug":"technical-report-on-web-based-visual-corpus","title":"On Web-based Visual Corpus Construction for Visual Document Understanding","date":"2022-11-07","arxiv_id":"2211.03256","repositories_listed":1,"syntology":null},{"url":"/paper/kalm-knowledge-aware-integration-of-local","slug":"kalm-knowledge-aware-integration-of-local","title":"KALM: Knowledge-Aware Integration of Local, Document, and Global Contexts for Long Document Understanding","date":"2022-10-08","arxiv_id":"2210.04105","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/kalm-knowledge-aware-integration-of-local#ran","syntology_url":"https://syntology.ai/paper/2210.04105","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.04105"}},"official":{"repos":["bunsenfeng/kalm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/xdoc-unified-pre-training-for-cross-format","slug":"xdoc-unified-pre-training-for-cross-format","title":"XDoc: Unified Pre-training for Cross-Format Document Understanding","date":"2022-10-06","arxiv_id":"2210.02849","repositories_listed":1,"syntology":null},{"url":"/paper/docquerynet-value-retrieval-with-arbitrary","slug":"docquerynet-value-retrieval-with-arbitrary","title":"DocQueryNet: Value Retrieval with Arbitrary Queries for Form-like Documents","date":"2022-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/doc2graph-a-task-agnostic-document","slug":"doc2graph-a-task-agnostic-document","title":"Doc2Graph: a Task Agnostic Document Understanding Framework based on Graph Neural Networks","date":"2022-08-23","arxiv_id":"2208.11168","repositories_listed":1,"syntology":null},{"url":"/paper/knowing-where-and-what-unified-word-block","slug":"knowing-where-and-what-unified-word-block","title":"Knowing Where and What: Unified Word Block Pretraining for Document Understanding","date":"2022-07-28","arxiv_id":"2207.13979","repositories_listed":1,"syntology":null},{"url":"/paper/davarocr-a-toolbox-for-ocr-and-multi-modal","slug":"davarocr-a-toolbox-for-ocr-and-multi-modal","title":"DavarOCR: A Toolbox for OCR and Multi-Modal Document Understanding","date":"2022-07-14","arxiv_id":"2207.06695","repositories_listed":1,"syntology":null},{"url":"/paper/delivering-document-conversion-as-a-cloud","slug":"delivering-document-conversion-as-a-cloud","title":"Delivering Document Conversion as a Cloud Service with High Throughput and Responsiveness","date":"2022-06-01","arxiv_id":"2206.00785","repositories_listed":1,"syntology":null},{"url":"/paper/markuplm-pre-training-of-text-and-markup-1","slug":"markuplm-pre-training-of-text-and-markup-1","title":"MarkupLM: Pre-training of Text and Markup Language for Visually Rich Document Understanding","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/textrm-dureader-textrm-vis-a-chinese-dataset","slug":"textrm-dureader-textrm-vis-a-chinese-dataset","title":"\\textrm{DuReader}_{\\textrm{vis}}: A Chinese Dataset for Open-domain Document Visual Question Answering","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-pre-training-based-on-graph","slug":"multimodal-pre-training-based-on-graph","title":"Multimodal Pre-training Based on Graph Attention Network for Document Understanding","date":"2022-03-25","arxiv_id":"2203.13530","repositories_listed":1,"syntology":null},{"url":"/paper/xylayoutlm-towards-layout-aware-multimodal","slug":"xylayoutlm-towards-layout-aware-multimodal","title":"XYLayoutLM: Towards Layout-Aware Multimodal Networks For Visually-Rich Document Understanding","date":"2022-03-14","arxiv_id":"2203.06947","repositories_listed":1,"syntology":null},{"url":"/paper/ernie-layout-layout-knowledge-enhanced-multi","slug":"ernie-layout-layout-knowledge-enhanced-multi","title":"ERNIE-Layout: Layout-Knowledge Enhanced Multi-modal Pre-training for Document Understanding","date":"2022-01-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deeper-clinical-document-understanding-using","slug":"deeper-clinical-document-understanding-using","title":"Deeper Clinical Document Understanding Using Relation Extraction","date":"2021-12-25","arxiv_id":"2112.13259","repositories_listed":1,"syntology":null},{"url":"/paper/value-retrieval-with-arbitrary-queries-for","slug":"value-retrieval-with-arbitrary-queries-for","title":"Value Retrieval with Arbitrary Queries for Form-like Documents","date":"2021-12-15","arxiv_id":"2112.07820","repositories_listed":1,"syntology":null},{"url":"/paper/skim-attention-learning-to-focus-via-document","slug":"skim-attention-learning-to-focus-via-document","title":"Skim-Attention: Learning to Focus via Document Layout","date":"2021-09-02","arxiv_id":"2109.01078","repositories_listed":1,"syntology":null},{"url":"/paper/docformer-end-to-end-transformer-for-document","slug":"docformer-end-to-end-transformer-for-document","title":"DocFormer: End-to-End Transformer for Document Understanding","date":"2021-06-22","arxiv_id":"2106.11539","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/docformer-end-to-end-transformer-for-document#ran","syntology_url":"https://syntology.ai/paper/2106.11539","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.11539"}},"official":null}},{"url":"/paper/citeworth-cite-worthiness-detection-for","slug":"citeworth-cite-worthiness-detection-for","title":"CiteWorth: Cite-Worthiness Detection for Improved Scientific Document Understanding","date":"2021-05-23","arxiv_id":"2105.10912","repositories_listed":1,"syntology":null},{"url":"/paper/going-full-tilt-boogie-on-document","slug":"going-full-tilt-boogie-on-document","title":"Going Full-TILT Boogie on Document Understanding with Text-Image-Layout Transformer","date":"2021-02-18","arxiv_id":"2102.09550","repositories_listed":1,"syntology":null},{"url":"/paper/towards-robust-visual-information-extraction","slug":"towards-robust-visual-information-extraction","title":"Towards Robust Visual Information Extraction in Real World: New Dataset and Novel Solution","date":"2021-01-24","arxiv_id":"2102.06732","repositories_listed":1,"syntology":null},{"url":"/paper/understood-in-translation-transformers-for","slug":"understood-in-translation-transformers-for","title":"Understood in Translation, Transformers for Domain Understanding","date":"2020-12-18","arxiv_id":"2012.10271","repositories_listed":1,"syntology":null},{"url":"/paper/primer-ai-s-systems-for-acronym","slug":"primer-ai-s-systems-for-acronym","title":"Primer AI's Systems for Acronym Identification and Disambiguation","date":"2020-12-14","arxiv_id":"2012.08013","repositories_listed":1,"syntology":null},{"url":"/paper/evalda-efficient-evasion-attacks-towards","slug":"evalda-efficient-evasion-attacks-towards","title":"EvaLDA: Efficient Evasion Attacks Towards Latent Dirichlet Allocation","date":"2020-12-09","arxiv_id":"2012.04864","repositories_listed":1,"syntology":null},{"url":"/paper/improving-clinical-document-understanding-on","slug":"improving-clinical-document-understanding-on","title":"Improving Clinical Document Understanding on COVID-19 Research with Spark NLP","date":"2020-12-07","arxiv_id":"2012.04005","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-of-deep-learning-approaches-for-ocr","slug":"a-survey-of-deep-learning-approaches-for-ocr","title":"A Survey of Deep Learning Approaches for OCR and Document Understanding","date":"2020-11-27","arxiv_id":"2011.13534","repositories_listed":1,"syntology":null},{"url":"/paper/wsl-ds-weakly-supervised-learning-with","slug":"wsl-ds-weakly-supervised-learning-with","title":"WSL-DS: Weakly Supervised Learning with Distant Supervision for Query Focused Multi-Document Abstractive Summarization","date":"2020-11-03","arxiv_id":"2011.01421","repositories_listed":1,"syntology":null},{"url":"/paper/a-discrete-variational-recurrent-topic-model","slug":"a-discrete-variational-recurrent-topic-model","title":"A Discrete Variational Recurrent Topic Model without the Reparametrization Trick","date":"2020-10-22","arxiv_id":"2010.12055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-discrete-variational-recurrent-topic-model#ran","syntology_url":"https://syntology.ai/paper/2010.12055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.12055"}},"official":{"repos":["mmrezaee/VRTM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/medicat-a-dataset-of-medical-images-captions","slug":"medicat-a-dataset-of-medical-images-captions","title":"MedICaT: A Dataset of Medical Images, Captions, and Textual References","date":"2020-10-12","arxiv_id":"2010.06000","repositories_listed":1,"syntology":null},{"url":"/paper/trie-end-to-end-text-reading-and-information","slug":"trie-end-to-end-text-reading-and-information","title":"TRIE: End-to-End Text Reading and Information Extraction for Document Understanding","date":"2020-05-27","arxiv_id":"2005.13118","repositories_listed":1,"syntology":null},{"url":"/paper/blockwise-self-attention-for-long-document","slug":"blockwise-self-attention-for-long-document","title":"Blockwise Self-Attention for Long Document Understanding","date":"2019-11-07","arxiv_id":"1911.02972","repositories_listed":1,"syntology":null},{"url":"/paper/fast-and-accurate-knowledge-aware-document","slug":"fast-and-accurate-knowledge-aware-document","title":"KRED: Knowledge-Aware Document Representation for News Recommendations","date":"2019-10-25","arxiv_id":"1910.11494","repositories_listed":1,"syntology":null},{"url":"/paper/bidirectional-context-aware-hierarchical","slug":"bidirectional-context-aware-hierarchical","title":"Bidirectional Context-Aware Hierarchical Attention Network for Document Understanding","date":"2019-08-16","arxiv_id":"1908.06006","repositories_listed":1,"syntology":null},{"url":"/paper/matching-long-text-documents-via-graph","slug":"matching-long-text-documents-via-graph","title":"Matching Article Pairs with Graphical Decomposition and Convolutions","date":"2018-02-21","arxiv_id":"1802.07459","repositories_listed":1,"syntology":null},{"url":null,"slug":"a-survey-on-mllm-based-visually-rich-document","title":"A Survey on MLLM-based Visually Rich Document Understanding: Methods, Challenges, and Emerging Trends","date":"2025-07-14","arxiv_id":"2507.09861","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-is-believing-mitigating-ocr","title":"Seeing is Believing? Mitigating OCR Hallucinations in Multimodal Large Language Models","date":"2025-06-25","arxiv_id":"2506.20168","repositories_listed":0,"syntology":null},{"url":null,"slug":"wikimixqa-a-multimodal-benchmark-for-question","title":"WikiMixQA: A Multimodal Benchmark for Question Answering over Tables and Charts","date":"2025-06-18","arxiv_id":"2506.15594","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-vietnamese-document-analysis-and","title":"A Survey on Vietnamese Document Analysis and Recognition: Challenges and Future Directions","date":"2025-06-05","arxiv_id":"2506.05061","repositories_listed":0,"syntology":null},{"url":null,"slug":"dicore-enhancing-zero-shot-event-detection","title":"DiCoRe: Enhancing Zero-shot Event Detection via Divergent-Convergent LLM Reasoning","date":"2025-06-05","arxiv_id":"2506.05128","repositories_listed":0,"syntology":null},{"url":null,"slug":"mt-3-scaling-mllm-based-text-image-machine","title":"MT$^{3}$: Scaling MLLM-based Text Image Machine Translation via Multi-Task Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.19714","repositories_listed":0,"syntology":null},{"url":null,"slug":"point-rft-improving-multimodal-reasoning-with","title":"Point-RFT: Improving Multimodal Reasoning with Visually Grounded Reinforcement Finetuning","date":"2025-05-26","arxiv_id":"2505.19702","repositories_listed":0,"syntology":null},{"url":null,"slug":"doc-cob-enhancing-multi-modal-document","title":"Doc-CoB: Enhancing Multi-Modal Document Understanding with Visual Chain-of-Boxes Reasoning","date":"2025-05-24","arxiv_id":"2505.18603","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-hidden-structure-improving-legal-document","title":"The Hidden Structure -- Improving Legal Document Understanding Through Explicit Text Formatting","date":"2025-05-19","arxiv_id":"2505.12837","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11015","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","date":"2025-05-16","arxiv_id":"2505.11015","repositories_listed":0,"syntology":null},{"url":null,"slug":"document-image-rectification-bases-on-self","title":"Document Image Rectification Bases on Self-Adaptive Multitask Fusion","date":"2025-05-09","arxiv_id":"2505.06038","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-parsing-of-engineering-drawings-for","title":"Automated Parsing of Engineering Drawings for Structured Information Extraction Using a Fine-tuned Document Understanding Transformer","date":"2025-05-02","arxiv_id":"2505.01530","repositories_listed":0,"syntology":null},{"url":null,"slug":"notes-bank-benchmarking-neural-transcription","title":"NoTeS-Bank: Benchmarking Neural Transcription and Search for Scientific Notes Understanding","date":"2025-04-12","arxiv_id":"2504.09249","repositories_listed":0,"syntology":null},{"url":null,"slug":"qid-efficient-query-informed-vits-in-data","title":"QID: Efficient Query-Informed ViTs in Data-Scarce Regimes for OCR-free Visual Document Understanding","date":"2025-04-03","arxiv_id":"2504.02971","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-does-watermarking-affect-visual-language","title":"How does Watermarking Affect Visual Language Models in Document Understanding?","date":"2025-04-01","arxiv_id":"2504.01048","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-applicability-of-deep-learning","title":"Improving Applicability of Deep Learning based Token Classification models during Training","date":"2025-03-28","arxiv_id":"2504.01028","repositories_listed":0,"syntology":null},{"url":null,"slug":"m-docsum-do-lvlms-genuinely-comprehend","title":"M-DocSum: Do LVLMs Genuinely Comprehend Interleaved Image-Text in Document Summarization?","date":"2025-03-27","arxiv_id":"2503.21839","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simple-yet-effective-layout-token-in-large","title":"A Simple yet Effective Layout Token in Large Language Models for Document Understanding","date":"2025-03-24","arxiv_id":"2503.18434","repositories_listed":0,"syntology":null},{"url":"/paper/marten-visual-question-answering-with-mask","slug":"marten-visual-question-answering-with-mask","title":"Marten: Visual Question Answering with Mask Generation for Multi-modal Document Understanding","date":"2025-03-18","arxiv_id":"2503.14140","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/marten-visual-question-answering-with-mask#ran","syntology_url":"https://syntology.ai/paper/2503.14140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.14140"}},"official":null}},{"url":"/paper/a-token-level-text-image-foundation-model-for","slug":"a-token-level-text-image-foundation-model-for","title":"A Token-level Text Image Foundation Model for Document Understanding","date":"2025-03-04","arxiv_id":"2503.02304","repositories_listed":0,"syntology":{"n":6,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/a-token-level-text-image-foundation-model-for#ran","syntology_url":"https://syntology.ai/paper/2503.02304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.02304"}},"official":null}},{"url":null,"slug":"shakti-vlms-scalable-vision-language-models","title":"Shakti-VLMs: Scalable Vision-Language Models for Enterprise AI","date":"2025-02-24","arxiv_id":"2502.17092","repositories_listed":0,"syntology":null},{"url":null,"slug":"kitab-bench-a-comprehensive-multi-domain","title":"KITAB-Bench: A Comprehensive Multi-Domain Benchmark for Arabic OCR and Document Understanding","date":"2025-02-20","arxiv_id":"2502.14949","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-generative-ai-value-in-a-public","title":"Assessing Generative AI value in a public sector context: evidence from a field experiment","date":"2025-02-13","arxiv_id":"2502.09479","repositories_listed":0,"syntology":null},{"url":null,"slug":"boundingdocs-a-unified-dataset-for-document","title":"BoundingDocs: a Unified Dataset for Document Question Answering with Spatial Annotations","date":"2025-01-06","arxiv_id":"2501.03403","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-on-question-answering-over-visually","title":"Survey on Question Answering over Visually Rich Documents: Methods, Challenges, and Trends","date":"2025-01-04","arxiv_id":"2501.02235","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-natural-language-based-document-image","title":"Towards Natural Language-Based Document Image Retrieval: New Dataset and Benchmark","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-prompting-and-few-shot-fine-tuning","title":"Zero-Shot Prompting and Few-Shot Fine-Tuning: Revisiting Document Image Classification Using Large Language Models","date":"2024-12-18","arxiv_id":"2412.13859","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-augmented-agent-training-for-business","title":"Memory-Augmented Agent Training for Business Document Understanding","date":"2024-12-17","arxiv_id":"2412.15274","repositories_listed":0,"syntology":null},{"url":null,"slug":"docvlm-make-your-vlm-an-efficient-reader","title":"DocVLM: Make Your VLM an Efficient Reader","date":"2024-12-11","arxiv_id":"2412.08746","repositories_listed":0,"syntology":null},{"url":null,"slug":"bigdocs-an-open-and-permissively-licensed","title":"BigDocs: An Open and Permissively-Licensed Dataset for Training Multimodal Models on Document and Code Tasks","date":"2024-12-05","arxiv_id":"2412.04626","repositories_listed":0,"syntology":null},{"url":null,"slug":"matata-a-weak-supervised-mathematical-tool","title":"MATATA: Weakly Supervised End-to-End MAthematical Tool-Augmented Reasoning for Tabular Applications","date":"2024-11-28","arxiv_id":"2411.18915","repositories_listed":0,"syntology":null},{"url":null,"slug":"doge-towards-versatile-visual-document","title":"DOGE: Towards Versatile Visual Document Grounding and Referring","date":"2024-11-26","arxiv_id":"2411.17125","repositories_listed":0,"syntology":null},{"url":null,"slug":"structformer-document-structure-based-masked","title":"StructFormer: Document Structure-based Masked Attention and its Impact on Language Model Pre-Training","date":"2024-11-25","arxiv_id":"2411.16618","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-extraction-from-heterogenous","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","date":"2024-11-22","arxiv_id":"2411.14957","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-cognition-consistent-with-perception","title":"Is Cognition consistent with Perception? Assessing and Mitigating Multimodal Knowledge Conflicts in Document Understanding","date":"2024-11-12","arxiv_id":"2411.07722","repositories_listed":0,"syntology":null},{"url":null,"slug":"m-longdoc-a-benchmark-for-multimodal-super","title":"M-Longdoc: A Benchmark For Multimodal Super-Long Document Understanding And A Retrieval-Aware Tuning Framework","date":"2024-11-09","arxiv_id":"2411.06176","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-visual-feature-aggregation-for","title":"Hierarchical Visual Feature Aggregation for OCR-Free Document Understanding","date":"2024-11-08","arxiv_id":"2411.05254","repositories_listed":0,"syntology":null},{"url":null,"slug":"m3docrag-multi-modal-retrieval-is-what-you","title":"M3DocRAG: Multi-modal Retrieval is What You Need for Multi-page Multi-document Understanding","date":"2024-11-07","arxiv_id":"2411.04952","repositories_listed":0,"syntology":null},{"url":null,"slug":"tokenselect-efficient-long-context-inference","title":"TokenSelect: Efficient Long-Context Inference and Length Extrapolation for LLMs via Dynamic Token-Level KV Cache Selection","date":"2024-11-05","arxiv_id":"2411.02886","repositories_listed":0,"syntology":null},{"url":null,"slug":"lora-contextualizing-adaptation-of-large","title":"LoRA-Contextualizing Adaptation of Large Multimodal Models for Long Document Understanding","date":"2024-11-02","arxiv_id":"2411.01106","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmdocbench-benchmarking-large-vision-language","title":"MMDocBench: Benchmarking Large Vision-Language Models for Fine-Grained Visual Document Understanding","date":"2024-10-25","arxiv_id":"2410.21311","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-is-the-value-of-templates-rethinking","title":"\"What is the value of {templates}?\" Rethinking Document Information Extraction Datasets for LLMs","date":"2024-10-20","arxiv_id":"2410.15484","repositories_listed":0,"syntology":null},{"url":null,"slug":"harnessing-webpage-uis-for-text-rich-visual","title":"Harnessing Webpage UIs for Text-Rich Visual Understanding","date":"2024-10-17","arxiv_id":"2410.13824","repositories_listed":0,"syntology":null},{"url":null,"slug":"relayout-towards-real-world-document","title":"ReLayout: Towards Real-World Document Understanding via Layout-enhanced Pre-training","date":"2024-10-14","arxiv_id":"2410.10471","repositories_listed":0,"syntology":null},{"url":null,"slug":"dockd-knowledge-distillation-from-llms-for","title":"DocKD: Knowledge Distillation from LLMs for Open-World Document Understanding Models","date":"2024-10-04","arxiv_id":"2410.03061","repositories_listed":0,"syntology":null},{"url":null,"slug":"david-domain-adaptive-visually-rich-document","title":"DAViD: Domain Adaptive Visually-Rich Document Understanding with Synthetic Insights","date":"2024-10-02","arxiv_id":"2410.01609","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-long-context-large-language-models","title":"Leveraging Long-Context Large Language Models for Multi-Document Understanding and Summarization in Enterprise Applications","date":"2024-09-27","arxiv_id":"2409.18454","repositories_listed":0,"syntology":null},{"url":null,"slug":"docmamba-efficient-document-pre-training-with","title":"DocMamba: Efficient Document Pre-training with State Space Model","date":"2024-09-18","arxiv_id":"2409.11887","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-distillation-techniques-for","title":"Leveraging Distillation Techniques for Document Understanding: A Case Study with FLAN-T5","date":"2024-09-17","arxiv_id":"2409.11282","repositories_listed":0,"syntology":null},{"url":null,"slug":"vired-prediction-of-visual-relations-in","title":"ViRED: Prediction of Visual Relations in Engineering Drawings","date":"2024-09-02","arxiv_id":"2409.00909","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-merit-dataset-modelling-and-efficiently","title":"The MERIT Dataset: Modelling and Efficiently Rendering Interpretable Transcripts","date":"2024-08-31","arxiv_id":"2409.00447","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthdoc-bilingual-documents-synthesis-for","title":"SynthDoc: Bilingual Documents Synthesis for Visual Document Understanding","date":"2024-08-27","arxiv_id":"2408.14764","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-and-better-understanding-vision","title":"Building and better understanding vision-language models: insights and future directions","date":"2024-08-22","arxiv_id":"2408.12637","repositories_listed":0,"syntology":null},{"url":null,"slug":"arctic-tilt-business-document-understanding","title":"Arctic-TILT. Business Document Understanding at Sub-Billion Scale","date":"2024-08-08","arxiv_id":"2408.04632","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01287","title":"Deep Learning based Visually Rich Document Content Understanding: A Survey","date":"2024-08-02","arxiv_id":"2408.01287","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-based-key-information","title":"Deep Learning based Key Information Extraction from Business Documents: Systematic Literature Review","date":"2024-07-23","arxiv_id":"2408.06345","repositories_listed":0,"syntology":null},{"url":"/paper/namer-non-autoregressive-modeling-for","slug":"namer-non-autoregressive-modeling-for","title":"NAMER: Non-Autoregressive Modeling for Handwritten Mathematical Expression Recognition","date":"2024-07-16","arxiv_id":"2407.11380","repositories_listed":0,"syntology":null},{"url":null,"slug":"dockylin-a-large-multimodal-model-for-visual","title":"DocKylin: A Large Multimodal Model for Visual Document Understanding with Efficient Visual Slimming","date":"2024-06-27","arxiv_id":"2406.19101","repositories_listed":0,"syntology":null},{"url":null,"slug":"drvideo-document-retrieval-based-long-video","title":"DrVideo: Document Retrieval Based Long Video Understanding","date":"2024-06-18","arxiv_id":"2406.12846","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-question-answering-on-charts","title":"Enhancing Question Answering on Charts Through Effective Pre-training Tasks","date":"2024-06-14","arxiv_id":"2406.10085","repositories_listed":0,"syntology":null}],"record_sha256":"c16dd620ec716c5cc3524bd3834da2fd2c15b78e158e885d152e6f967c3fb954","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}