{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/optical-character-recognition/papers/2","list_of":"/task/optical-character-recognition","task":"Optical Character Recognition (OCR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":13,"rows_per_page":100,"rows":[101,200],"of":1243,"counts":{"archive_papers_tagged":1243,"with_a_code_link":462,"where_syntology_ran_a_sample":76,"not_listed_spam_title":0,"listed":1243,"listed_where_code_ran":76,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/optical-character-recognition","prev":"/task/optical-character-recognition","next":"/task/optical-character-recognition/papers/3","papers":[{"url":"/paper/infinity-parser-layout-aware-reinforcement","slug":"infinity-parser-layout-aware-reinforcement","title":"Infinity Parser: Layout Aware Reinforcement Learning for Scanned Document Parsing","date":"2025-06-01","arxiv_id":"2506.03197","repositories_listed":1,"syntology":null},{"url":"/paper/predicting-the-past-estimating-historical","slug":"predicting-the-past-estimating-historical","title":"Predicting the Past: Estimating Historical Appraisals with OCR and Machine Learning","date":"2025-05-30","arxiv_id":"2505.24676","repositories_listed":1,"syntology":null},{"url":"/paper/synthetic-document-question-answering-in","slug":"synthetic-document-question-answering-in","title":"Synthetic Document Question Answering in Hungarian","date":"2025-05-29","arxiv_id":"2505.23008","repositories_listed":1,"syntology":null},{"url":"/paper/uni-mumer-unified-multi-task-fine-tuning-of","slug":"uni-mumer-unified-multi-task-fine-tuning-of","title":"Uni-MuMER: Unified Multi-Task Fine-Tuning of Vision-Language Model for Handwritten Mathematical Expression Recognition","date":"2025-05-29","arxiv_id":"2505.23566","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":1,"n_ran_checked":3,"n_instrument":4,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uni-mumer-unified-multi-task-fine-tuning-of#ran","syntology_url":"https://syntology.ai/paper/2505.23566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23566"}},"official":{"repos":["bflameswift/uni-mumer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vidtext-towards-comprehensive-evaluation-for","slug":"vidtext-towards-comprehensive-evaluation-for","title":"VidText: Towards Comprehensive Evaluation for Video Text Understanding","date":"2025-05-28","arxiv_id":"2505.22810","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vidtext-towards-comprehensive-evaluation-for#ran","syntology_url":"https://syntology.ai/paper/2505.22810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22810"}},"official":{"repos":["shuyansy/vidtext"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-action-model-with-open-world","slug":"vision-language-action-model-with-open-world","title":"ChatVLA-2: Vision-Language-Action Model with Open-World Embodied Reasoning from Pretrained Knowledge","date":"2025-05-28","arxiv_id":"2505.21906","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vision-language-action-model-with-open-world#ran","syntology_url":"https://syntology.ai/paper/2505.21906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21906"}},"official":null}},{"url":"/paper/unifying-multimodal-large-language-model","slug":"unifying-multimodal-large-language-model","title":"Unifying Multimodal Large Language Model Capabilities and Modalities via Model Merging","date":"2025-05-26","arxiv_id":"2505.19892","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2505.19892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19892"}},"official":{"repos":["walkerworldpeace/mllmerging"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/readbench-measuring-the-dense-text-visual","slug":"readbench-measuring-the-dense-text-visual","title":"ReadBench: Measuring the Dense Text Visual Reading Ability of Vision-Language Models","date":"2025-05-25","arxiv_id":"2505.19091","repositories_listed":1,"syntology":null},{"url":"/paper/arb-a-comprehensive-arabic-multimodal","slug":"arb-a-comprehensive-arabic-multimodal","title":"ARB: A Comprehensive Arabic Multimodal Reasoning Benchmark","date":"2025-05-22","arxiv_id":"2505.17021","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-ocr-can-large-multimodal-models","slug":"reasoning-ocr-can-large-multimodal-models","title":"Reasoning-OCR: Can Large Multimodal Models Solve Complex Logical Reasoning Problems from OCR Cues?","date":"2025-05-19","arxiv_id":"2505.12766","repositories_listed":1,"syntology":null},{"url":"/paper/logicocr-do-your-large-multimodal-models","slug":"logicocr-do-your-large-multimodal-models","title":"LogicOCR: Do Your Large Multimodal Models Excel at Logical Reasoning on Text-Rich Images?","date":"2025-05-18","arxiv_id":"2505.12307","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11177","slug":"2505-11177","title":"Low-Resource Language Processing: An OCR-Driven Summarization and Translation Pipeline","date":"2025-05-16","arxiv_id":"2505.11177","repositories_listed":1,"syntology":null},{"url":"/paper/an-agentic-system-with-reinforcement-learned","slug":"an-agentic-system-with-reinforcement-learned","title":"An agentic system with reinforcement-learned subsystem improvements for parsing form-like documents","date":"2025-05-16","arxiv_id":"2505.13504","repositories_listed":1,"syntology":null},{"url":"/paper/psocr-benchmarking-large-multimodal-models","slug":"psocr-benchmarking-large-multimodal-models","title":"PsOCR: Benchmarking Large Multimodal Models for Optical Character Recognition in Low-resource Pashto Language","date":"2025-05-15","arxiv_id":"2505.10055","repositories_listed":1,"syntology":null},{"url":"/paper/reproducibility-replicability-and-insights","slug":"reproducibility-replicability-and-insights","title":"Reproducibility, Replicability, and Insights into Visual Document Retrieval with Late Interaction","date":"2025-05-12","arxiv_id":"2505.07730","repositories_listed":1,"syntology":null},{"url":"/paper/arrow-guided-vlm-enhancing-flowchart","slug":"arrow-guided-vlm-enhancing-flowchart","title":"Arrow-Guided VLM: Enhancing Flowchart Understanding via Arrow Direction Encoding","date":"2025-05-09","arxiv_id":"2505.07864","repositories_listed":1,"syntology":null},{"url":"/paper/toward-advancing-license-plate-super","slug":"toward-advancing-license-plate-super","title":"Toward Advancing License Plate Super-Resolution in Real-World Scenarios: A Dataset and Benchmark","date":"2025-05-09","arxiv_id":"2505.06393","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-menu-ocr-and-translation-a","slug":"evaluating-menu-ocr-and-translation-a","title":"Evaluating Menu OCR and Translation: A Benchmark for Aligning Human and Automated Evaluations in Large Vision-Language Models","date":"2025-04-16","arxiv_id":"2504.13945","repositories_listed":1,"syntology":null},{"url":"/paper/relation-rich-visual-document-generator-for","slug":"relation-rich-visual-document-generator-for","title":"Relation-Rich Visual Document Generator for Visual Information Extraction","date":"2025-04-14","arxiv_id":"2504.10659","repositories_listed":1,"syntology":null},{"url":"/paper/kimi-vl-technical-report","slug":"kimi-vl-technical-report","title":"Kimi-VL Technical Report","date":"2025-04-10","arxiv_id":"2504.07491","repositories_listed":1,"syntology":null},{"url":"/paper/playing-non-embedded-card-based-games-with","slug":"playing-non-embedded-card-based-games-with","title":"Playing Non-Embedded Card-Based Games with Reinforcement Learning","date":"2025-04-07","arxiv_id":"2504.04783","repositories_listed":1,"syntology":null},{"url":"/paper/from-panels-to-prose-generating-literary","slug":"from-panels-to-prose-generating-literary","title":"From Panels to Prose: Generating Literary Narratives from Comics","date":"2025-03-30","arxiv_id":"2503.23344","repositories_listed":1,"syntology":null},{"url":"/paper/bibliopage-a-dataset-of-scanned-title-pages","slug":"bibliopage-a-dataset-of-scanned-title-pages","title":"BiblioPage: A Dataset of Scanned Title Pages for Bibliographic Metadata Extraction","date":"2025-03-25","arxiv_id":"2503.19658","repositories_listed":1,"syntology":null},{"url":"/paper/pm4bench-a-parallel-multilingual-multi-modal","slug":"pm4bench-a-parallel-multilingual-multi-modal","title":"PM4Bench: A Parallel Multilingual Multi-Modal Multi-task Benchmark for Large Vision Language Model","date":"2025-03-24","arxiv_id":"2503.18484","repositories_listed":1,"syntology":null},{"url":"/paper/kl3m-tokenizers-a-family-of-domain-specific","slug":"kl3m-tokenizers-a-family-of-domain-specific","title":"KL3M Tokenizers: A Family of Domain-Specific and Character-Level Tokenizers for Legal, Financial, and Preprocessing Applications","date":"2025-03-21","arxiv_id":"2503.17247","repositories_listed":1,"syntology":null},{"url":"/paper/a-data-driven-investigation-of-euphemistic","slug":"a-data-driven-investigation-of-euphemistic","title":"A Data-driven Investigation of Euphemistic Language: Comparing the usage of \"slave\" and \"servant\" in 19th century US newspapers","date":"2025-03-19","arxiv_id":"2503.15057","repositories_listed":1,"syntology":null},{"url":"/paper/kap-mllm-assisted-ocr-text-enhancement-for","slug":"kap-mllm-assisted-ocr-text-enhancement-for","title":"KAP: MLLM-assisted OCR Text Enhancement for Hybrid Retrieval in Chinese Non-Narrative Documents","date":"2025-03-11","arxiv_id":"2503.08452","repositories_listed":1,"syntology":null},{"url":"/paper/pp-docbee-improving-multimodal-document","slug":"pp-docbee-improving-multimodal-document","title":"PP-DocBee: Improving Multimodal Document Understanding Through a Bag of Tricks","date":"2025-03-06","arxiv_id":"2503.04065","repositories_listed":1,"syntology":null},{"url":"/paper/an-approach-for-air-drawing-using-background","slug":"an-approach-for-air-drawing-using-background","title":"An Approach for Air Drawing Using Background Subtraction and Contour Extraction","date":"2025-03-03","arxiv_id":"2503.01497","repositories_listed":1,"syntology":null},{"url":"/paper/judge-a-book-by-its-cover-investigating-multi","slug":"judge-a-book-by-its-cover-investigating-multi","title":"Judge a Book by its Cover: Investigating Multi-Modal LLMs for Multi-Page Handwritten Document Transcription","date":"2025-02-27","arxiv_id":"2502.20295","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-offensive-memes-with-social-biases","slug":"detecting-offensive-memes-with-social-biases","title":"Detecting Offensive Memes with Social Biases in Singapore Context Using Multimodal Large Language Models","date":"2025-02-25","arxiv_id":"2502.18101","repositories_listed":1,"syntology":null},{"url":"/paper/multiocr-qa-dataset-for-evaluating-robustness","slug":"multiocr-qa-dataset-for-evaluating-robustness","title":"MultiOCR-QA: Dataset for Evaluating Robustness of LLMs in Question Answering on Multilingual OCR Texts","date":"2025-02-24","arxiv_id":"2502.16781","repositories_listed":1,"syntology":null},{"url":"/paper/reading-the-unreadable-creating-a-dataset-of","slug":"reading-the-unreadable-creating-a-dataset-of","title":"Reading the unreadable: Creating a dataset of 19th century English newspapers using image-to-text language models","date":"2025-02-18","arxiv_id":"2502.14901","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-vision-language-models-on","slug":"benchmarking-vision-language-models-on","title":"Benchmarking Vision-Language Models on Optical Character Recognition in Dynamic Video Environments","date":"2025-02-10","arxiv_id":"2502.06445","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-vision-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2502.06445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06445"}},"official":{"repos":["video-db/ocr-benchmark"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-making-flowchart-images-machine-1","slug":"towards-making-flowchart-images-machine-1","title":"Towards Making Flowchart Images Machine Interpretable","date":"2025-01-29","arxiv_id":"2501.17441","repositories_listed":1,"syntology":null},{"url":"/paper/ocean-ocr-towards-general-ocr-application-via","slug":"ocean-ocr-towards-general-ocr-application-via","title":"Ocean-OCR: Towards General OCR Application via a Vision-Language Model","date":"2025-01-26","arxiv_id":"2501.15558","repositories_listed":1,"syntology":null},{"url":"/paper/early-evidence-of-how-llms-outperform","slug":"early-evidence-of-how-llms-outperform","title":"Early evidence of how LLMs outperform traditional systems on OCR/HTR tasks for historical records","date":"2025-01-20","arxiv_id":"2501.11623","repositories_listed":1,"syntology":null},{"url":"/paper/mathreader-text-to-speech-for-mathematical","slug":"mathreader-text-to-speech-for-mathematical","title":"MathReader : Text-to-Speech for Mathematical Documents","date":"2025-01-13","arxiv_id":"2501.07088","repositories_listed":1,"syntology":null},{"url":"/paper/centurio-on-drivers-of-multilingual-ability","slug":"centurio-on-drivers-of-multilingual-ability","title":"Centurio: On Drivers of Multilingual Ability of Large Vision-Language Model","date":"2025-01-09","arxiv_id":"2501.05122","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-license-plate-recognition-in-videos","slug":"efficient-license-plate-recognition-in-videos","title":"Efficient License Plate Recognition in Videos Using Visual Rhythm and Accumulative Line Analysis","date":"2025-01-08","arxiv_id":"2501.04750","repositories_listed":1,"syntology":null},{"url":"/paper/geometry-restoration-and-dewarping-of-camera","slug":"geometry-restoration-and-dewarping-of-camera","title":"Geometry Restoration and Dewarping of Camera-Captured Document Images","date":"2025-01-06","arxiv_id":"2501.03145","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-video-based-alpr-system-using-yolo","slug":"efficient-video-based-alpr-system-using-yolo","title":"Efficient Video-Based ALPR System Using YOLO and Visual Rhythm","date":"2025-01-04","arxiv_id":"2501.02270","repositories_listed":1,"syntology":null},{"url":"/paper/crossing-language-borders-a-pipeline-for","slug":"crossing-language-borders-a-pipeline-for","title":"Crossing Language Borders: A Pipeline for Indonesian Manhwa Translation","date":"2025-01-03","arxiv_id":"2501.01629","repositories_listed":1,"syntology":null},{"url":"/paper/2-5-years-in-class-a-multimodal-textbook-for","slug":"2-5-years-in-class-a-multimodal-textbook-for","title":"2.5 Years in Class: A Multimodal Textbook for Vision-Language Pretraining","date":"2025-01-01","arxiv_id":"2501.00958","repositories_listed":1,"syntology":null},{"url":"/paper/doclayllm-an-efficient-multi-modal-extension","slug":"doclayllm-an-efficient-multi-modal-extension","title":"DocLayLLM: An Efficient Multi-modal Extension of Large Language Models for Text-rich Document Understanding","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ocrbench-v2-an-improved-benchmark-for","slug":"ocrbench-v2-an-improved-benchmark-for","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","date":"2024-12-31","arxiv_id":"2501.00321","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ocrbench-v2-an-improved-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2501.00321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.00321"}},"official":{"repos":["yuliang-liu/multimodalocr"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/do-current-video-llms-have-strong-ocr","slug":"do-current-video-llms-have-strong-ocr","title":"Do Current Video LLMs Have Strong OCR Abilities? A Preliminary Study","date":"2024-12-29","arxiv_id":"2412.20613","repositories_listed":1,"syntology":null},{"url":"/paper/lmv-rpa-large-model-voting-based-robotic","slug":"lmv-rpa-large-model-voting-based-robotic","title":"LMV-RPA: Large Model Voting-based Robotic Process Automation","date":"2024-12-23","arxiv_id":"2412.17965","repositories_listed":1,"syntology":null},{"url":"/paper/deciphering-the-underserved-benchmarking-llm","slug":"deciphering-the-underserved-benchmarking-llm","title":"Deciphering the Underserved: Benchmarking LLM OCR for Low-Resource Scripts","date":"2024-12-20","arxiv_id":"2412.16119","repositories_listed":1,"syntology":null},{"url":"/paper/instructocr-instruction-boosting-scene-text","slug":"instructocr-instruction-boosting-scene-text","title":"InstructOCR: Instruction Boosting Scene Text Spotting","date":"2024-12-20","arxiv_id":"2412.15523","repositories_listed":1,"syntology":null},{"url":"/paper/track-the-answer-extending-textvqa-from-image","slug":"track-the-answer-extending-textvqa-from-image","title":"Track the Answer: Extending TextVQA from Image to Video with Spatio-Temporal Clues","date":"2024-12-17","arxiv_id":"2412.12502","repositories_listed":1,"syntology":null},{"url":"/paper/roundtripocr-a-data-generation-technique-for","slug":"roundtripocr-a-data-generation-technique-for","title":"RoundTripOCR: A Data Generation Technique for Enhancing Post-OCR Error Correction in Low-Resource Devanagari Languages","date":"2024-12-14","arxiv_id":"2412.15248","repositories_listed":1,"syntology":null},{"url":"/paper/deepseek-vl2-mixture-of-experts-vision","slug":"deepseek-vl2-mixture-of-experts-vision","title":"DeepSeek-VL2: Mixture-of-Experts Vision-Language Models for Advanced Multimodal Understanding","date":"2024-12-13","arxiv_id":"2412.10302","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deepseek-vl2-mixture-of-experts-vision#ran","syntology_url":"https://syntology.ai/paper/2412.10302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.10302"}},"official":{"repos":["deepseek-ai/deepseek-vl2"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/taco-learning-multi-modal-action-models-with","slug":"taco-learning-multi-modal-action-models-with","title":"TACO: Learning Multi-modal Action Models with Synthetic Chains-of-Thought-and-Action","date":"2024-12-07","arxiv_id":"2412.05479","repositories_listed":1,"syntology":null},{"url":"/paper/aligned-music-notation-and-lyrics","slug":"aligned-music-notation-and-lyrics","title":"Aligned Music Notation and Lyrics Transcription","date":"2024-12-05","arxiv_id":"2412.04217","repositories_listed":1,"syntology":null},{"url":"/paper/florence-vl-enhancing-vision-language-models","slug":"florence-vl-enhancing-vision-language-models","title":"Florence-VL: Enhancing Vision-Language Models with Generative Vision Encoder and Depth-Breadth Fusion","date":"2024-12-05","arxiv_id":"2412.04424","repositories_listed":1,"syntology":null},{"url":"/paper/synfintabs-a-dataset-of-synthetic-financial","slug":"synfintabs-a-dataset-of-synthetic-financial","title":"SynFinTabs: A Dataset of Synthetic Financial Tables for Information and Table Extraction","date":"2024-12-05","arxiv_id":"2412.04262","repositories_listed":1,"syntology":null},{"url":"/paper/paligemma-2-a-family-of-versatile-vlms-for","slug":"paligemma-2-a-family-of-versatile-vlms-for","title":"PaliGemma 2: A Family of Versatile VLMs for Transfer","date":"2024-12-04","arxiv_id":"2412.03555","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/paligemma-2-a-family-of-versatile-vlms-for#ran","syntology_url":"https://syntology.ai/paper/2412.03555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03555"}},"official":null}},{"url":"/paper/ocr-hinders-rag-evaluating-the-cascading","slug":"ocr-hinders-rag-evaluating-the-cascading","title":"OCR Hinders RAG: Evaluating the Cascading Impact of OCR on Retrieval-Augmented Generation","date":"2024-12-03","arxiv_id":"2412.02592","repositories_listed":1,"syntology":null},{"url":"/paper/textssr-diffusion-based-data-synthesis-for","slug":"textssr-diffusion-based-data-synthesis-for","title":"TextSSR: Diffusion-based Data Synthesis for Scene Text Recognition","date":"2024-12-02","arxiv_id":"2412.01137","repositories_listed":1,"syntology":null},{"url":"/paper/dlava-document-language-and-vision-assistant","slug":"dlava-document-language-and-vision-assistant","title":"DLaVA: Document Language and Vision Assistant for Answer Localization with Enhanced Interpretability and Trustworthiness","date":"2024-11-29","arxiv_id":"2412.00151","repositories_listed":1,"syntology":null},{"url":"/paper/svtrv2-ctc-beats-encoder-decoder-models-in","slug":"svtrv2-ctc-beats-encoder-decoder-models-in","title":"SVTRv2: CTC Beats Encoder-Decoder Models in Scene Text Recognition","date":"2024-11-24","arxiv_id":"2411.15858","repositories_listed":1,"syntology":null},{"url":"/paper/arabic-nougat-fine-tuning-vision-transformers","slug":"arabic-nougat-fine-tuning-vision-transformers","title":"Arabic-Nougat: Fine-Tuning Vision Transformers for Arabic OCR and Markdown Extraction","date":"2024-11-19","arxiv_id":"2411.17835","repositories_listed":1,"syntology":null},{"url":"/paper/awaker2-5-vl-stably-scaling-mllms-with","slug":"awaker2-5-vl-stably-scaling-mllms-with","title":"Awaker2.5-VL: Stably Scaling MLLMs with Parameter-Efficient Mixture of Experts","date":"2024-11-16","arxiv_id":"2411.10669","repositories_listed":1,"syntology":null},{"url":"/paper/drivethru-a-document-extraction-platform-and","slug":"drivethru-a-document-extraction-platform-and","title":"DriveThru: a Document Extraction Platform and Benchmark Datasets for Indonesian Local Language Archives","date":"2024-11-14","arxiv_id":"2411.09318","repositories_listed":1,"syntology":null},{"url":"/paper/are-vlms-really-blind","slug":"are-vlms-really-blind","title":"Are VLMs Really Blind","date":"2024-10-29","arxiv_id":"2410.22029","repositories_listed":1,"syntology":null},{"url":"/paper/toxicity-of-the-commons-curating-open-source","slug":"toxicity-of-the-commons-curating-open-source","title":"Toxicity of the Commons: Curating Open-Source Pre-Training Data","date":"2024-10-29","arxiv_id":"2410.22587","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/toxicity-of-the-commons-curating-open-source#ran","syntology_url":"https://syntology.ai/paper/2410.22587","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22587"}},"official":{"repos":["Pleias/toxic-commons"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/legal-uqa-a-low-resource-urdu-english-dataset","slug":"legal-uqa-a-low-resource-urdu-english-dataset","title":"LEGAL-UQA: A Low-Resource Urdu-English Dataset for Legal Question Answering","date":"2024-10-16","arxiv_id":"2410.13013","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-assamese-nlp-capabilities","slug":"enhancing-assamese-nlp-capabilities","title":"Enhancing Assamese NLP Capabilities: Introducing a Centralized Dataset Repository","date":"2024-10-15","arxiv_id":"2410.11291","repositories_listed":1,"syntology":null},{"url":"/paper/stratified-domain-adaptation-a-progressive","slug":"stratified-domain-adaptation-a-progressive","title":"Stratified Domain Adaptation: A Progressive Self-Training Approach for Scene Text Recognition","date":"2024-10-13","arxiv_id":"2410.09913","repositories_listed":1,"syntology":null},{"url":"/paper/hespi-a-pipeline-for-automatically-detecting","slug":"hespi-a-pipeline-for-automatically-detecting","title":"Hespi: A pipeline for automatically detecting information from hebarium specimen sheets","date":"2024-10-11","arxiv_id":"2410.08740","repositories_listed":1,"syntology":null},{"url":"/paper/texthawk2-a-large-vision-language-model","slug":"texthawk2-a-large-vision-language-model","title":"TextHawk2: A Large Vision-Language Model Excels in Bilingual OCR and Grounding with 16x Fewer Tokens","date":"2024-10-07","arxiv_id":"2410.05261","repositories_listed":1,"syntology":null},{"url":"/paper/world-to-code-multi-modal-data-generation-via","slug":"world-to-code-multi-modal-data-generation-via","title":"World to Code: Multi-modal Data Generation via Self-Instructed Compositional Captioning and Filtering","date":"2024-09-30","arxiv_id":"2409.20424","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/world-to-code-multi-modal-data-generation-via#ran","syntology_url":"https://syntology.ai/paper/2409.20424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.20424"}},"official":{"repos":["foundation-multimodal-models/world2code"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/general-detection-based-text-line-recognition","slug":"general-detection-based-text-line-recognition","title":"General Detection-based Text Line Recognition","date":"2024-09-25","arxiv_id":"2409.17095","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-ocr-sensitive-neurons-to","slug":"investigating-ocr-sensitive-neurons-to","title":"Investigating OCR-Sensitive Neurons to Improve Entity Recognition in Historical Documents","date":"2024-09-25","arxiv_id":"2409.16934","repositories_listed":1,"syntology":null},{"url":"/paper/mavils-a-benchmark-dataset-for-video-to-slide","slug":"mavils-a-benchmark-dataset-for-video-to-slide","title":"MaViLS, a Benchmark Dataset for Video-to-Slide Alignment, Assessing Baseline Accuracy with a Multimodal Alignment Algorithm Leveraging Speech, OCR, and Visual Features","date":"2024-09-25","arxiv_id":"2409.16765","repositories_listed":1,"syntology":null},{"url":"/paper/2409-13920","slug":"2409-13920","title":"One Model is All You Need: ByT5-Sanskrit, a Unified Model for Sanskrit NLP Tasks","date":"2024-09-20","arxiv_id":"2409.13920","repositories_listed":1,"syntology":null},{"url":"/paper/pdftable-a-unified-toolkit-for-deep-learning","slug":"pdftable-a-unified-toolkit-for-deep-learning","title":"PdfTable: A Unified Toolkit for Deep Learning-Based Table Extraction","date":"2024-09-08","arxiv_id":"2409.05125","repositories_listed":1,"syntology":null},{"url":"/paper/mplug-docowl2-high-resolution-compressing-for","slug":"mplug-docowl2-high-resolution-compressing-for","title":"mPLUG-DocOwl2: High-resolution Compressing for OCR-free Multi-page Document Understanding","date":"2024-09-05","arxiv_id":"2409.03420","repositories_listed":1,"syntology":null},{"url":"/paper/general-ocr-theory-towards-ocr-2-0-via-a","slug":"general-ocr-theory-towards-ocr-2-0-via-a","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","date":"2024-09-03","arxiv_id":"2409.01704","repositories_listed":1,"syntology":null},{"url":"/paper/post-ocr-text-correction-for-bulgarian","slug":"post-ocr-text-correction-for-bulgarian","title":"Post-OCR Text Correction for Bulgarian Historical Documents","date":"2024-08-31","arxiv_id":"2409.00527","repositories_listed":1,"syntology":null},{"url":"/paper/clocr-c-context-leveraging-ocr-correction","slug":"clocr-c-context-leveraging-ocr-correction","title":"CLOCR-C: Context Leveraging OCR Correction with Pre-trained Language Models","date":"2024-08-30","arxiv_id":"2408.17428","repositories_listed":1,"syntology":null},{"url":"/paper/doclayllm-an-efficient-and-effective-multi","slug":"doclayllm-an-efficient-and-effective-multi","title":"DocLayLLM: An Efficient and Effective Multi-modal Extension of Large Language Models for Text-rich Document Understanding","date":"2024-08-27","arxiv_id":"2408.15045","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/doclayllm-an-efficient-and-effective-multi#ran","syntology_url":"https://syntology.ai/paper/2408.15045","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.15045"}},"official":{"repos":["whlscut/DocLayLLM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-license-plate-super-resolution-a","slug":"enhancing-license-plate-super-resolution-a","title":"Enhancing License Plate Super-Resolution: A Layout-Aware and Character-Driven Approach","date":"2024-08-27","arxiv_id":"2408.15103","repositories_listed":1,"syntology":null},{"url":"/paper/fasttextspotter-a-high-efficiency-transformer","slug":"fasttextspotter-a-high-efficiency-transformer","title":"FastTextSpotter: A High-Efficiency Transformer for Multilingual Scene Text Spotting","date":"2024-08-27","arxiv_id":"2408.14998","repositories_listed":1,"syntology":null},{"url":"/paper/platypus-a-generalized-specialist-model-for","slug":"platypus-a-generalized-specialist-model-for","title":"Platypus: A Generalized Specialist Model for Reading Text in Various Forms","date":"2024-08-27","arxiv_id":"2408.14805","repositories_listed":1,"syntology":null},{"url":"/paper/seeing-and-understanding-bridging-vision-with","slug":"seeing-and-understanding-bridging-vision-with","title":"ChemVLM: Exploring the Power of Multimodal Large Language Models in Chemistry Area","date":"2024-08-14","arxiv_id":"2408.07246","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/seeing-and-understanding-bridging-vision-with#ran","syntology_url":"https://syntology.ai/paper/2408.07246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07246"}},"official":{"repos":["AI4Chem/ChemVlm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/handwritten-code-recognition-for-pen-and","slug":"handwritten-code-recognition-for-pen-and","title":"Handwritten Code Recognition for Pen-and-Paper CS Education","date":"2024-08-07","arxiv_id":"2408.07220","repositories_listed":1,"syntology":null},{"url":"/paper/2408-02253","slug":"2408-02253","title":"Advancing Post-OCR Correction: A Comparative Study of Synthetic Data","date":"2024-08-05","arxiv_id":"2408.02253","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00765","slug":"2408-00765","title":"MM-Vet v2: A Challenging Benchmark to Evaluate Large Multimodal Models for Integrated Capabilities","date":"2024-08-01","arxiv_id":"2408.00765","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2408-00765#ran","syntology_url":"https://syntology.ai/paper/2408.00765","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00765"}},"official":{"repos":["yuweihao/mm-vet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/focus-distinguish-and-prompt-unleashing-clip","slug":"focus-distinguish-and-prompt-unleashing-clip","title":"Focus, Distinguish, and Prompt: Unleashing CLIP for Efficient and Flexible Scene Text Retrieval","date":"2024-08-01","arxiv_id":"2408.00441","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/focus-distinguish-and-prompt-unleashing-clip#ran","syntology_url":"https://syntology.ai/paper/2408.00441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00441"}},"official":{"repos":["gyann-z/fdp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pixelmod-improving-soft-moderation-of-visual","slug":"pixelmod-improving-soft-moderation-of-visual","title":"PIXELMOD: Improving Soft Moderation of Visual Misleading Information on Twitter","date":"2024-07-30","arxiv_id":"2407.20987","repositories_listed":1,"syntology":null},{"url":"/paper/image-text-matching-for-large-scale-book","slug":"image-text-matching-for-large-scale-book","title":"Image-text matching for large-scale book collections","date":"2024-07-29","arxiv_id":"2407.19812","repositories_listed":1,"syntology":null},{"url":"/paper/visfocus-prompt-guided-vision-encoders-for","slug":"visfocus-prompt-guided-vision-encoders-for","title":"VisFocus: Prompt-Guided Vision Encoders for OCR-Free Dense Document Understanding","date":"2024-07-17","arxiv_id":"2407.12594","repositories_listed":1,"syntology":null},{"url":"/paper/spanish-trocr-leveraging-transfer-learning","slug":"spanish-trocr-leveraging-transfer-learning","title":"Spanish TrOCR: Leveraging Transfer Learning for Language Adaptation","date":"2024-07-09","arxiv_id":"2407.06950","repositories_listed":1,"syntology":null},{"url":"/paper/high-throughput-phenotyping-using-computer","slug":"high-throughput-phenotyping-using-computer","title":"High-Throughput Phenotyping using Computer Vision and Machine Learning","date":"2024-07-08","arxiv_id":"2407.06354","repositories_listed":1,"syntology":null},{"url":"/paper/flowlearn-evaluating-large-vision-language","slug":"flowlearn-evaluating-large-vision-language","title":"FlowLearn: Evaluating Large Vision-Language Models on Flowchart Understanding","date":"2024-07-06","arxiv_id":"2407.05183","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-nepali-pdf-extraction-a","slug":"optimizing-nepali-pdf-extraction-a","title":"Optimizing Nepali PDF Extraction: A Comparative Study of Parser and OCR Technologies","date":"2024-07-05","arxiv_id":"2407.04577","repositories_listed":1,"syntology":null},{"url":"/paper/historical-ink-19th-century-latin-american","slug":"historical-ink-19th-century-latin-american","title":"Historical Ink: 19th Century Latin American Spanish Newspaper Corpus with LLM OCR Correction","date":"2024-07-04","arxiv_id":"2407.12838","repositories_listed":1,"syntology":null},{"url":"/paper/a-bounding-box-is-worth-one-token","slug":"a-bounding-box-is-worth-one-token","title":"A Bounding Box is Worth One Token: Interleaving Layout and Text in a Large Language Model for Document Understanding","date":"2024-07-02","arxiv_id":"2407.01976","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":4,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-bounding-box-is-worth-one-token#ran","syntology_url":"https://syntology.ai/paper/2407.01976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01976"}},"official":{"repos":["laytextllm/laytextllm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"59ab3f0f4fbecae672580557c04d1ba3a319f20bd0492ce2eb48fbc064a2f596","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}