{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/optical-character-recognition/papers/3","list_of":"/task/optical-character-recognition","task":"Optical Character Recognition (OCR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":13,"rows_per_page":100,"rows":[201,300],"of":1243,"counts":{"archive_papers_tagged":1243,"with_a_code_link":462,"where_syntology_ran_a_sample":76,"not_listed_spam_title":0,"listed":1243,"listed_where_code_ran":76,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/optical-character-recognition","prev":"/task/optical-character-recognition/papers/2","next":"/task/optical-character-recognition/papers/4","papers":[{"url":"/paper/mmlongbench-doc-benchmarking-long-context","slug":"mmlongbench-doc-benchmarking-long-context","title":"MMLongBench-Doc: Benchmarking Long-context Document Understanding with Visualizations","date":"2024-07-01","arxiv_id":"2407.01523","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mmlongbench-doc-benchmarking-long-context#ran","syntology_url":"https://syntology.ai/paper/2407.01523","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01523"}},"official":null}},{"url":"/paper/docparsenet-advanced-semantic-segmentation","slug":"docparsenet-advanced-semantic-segmentation","title":"DocParseNet: Advanced Semantic Segmentation and OCR Embeddings for Efficient Scanned Document Annotation","date":"2024-06-25","arxiv_id":"2406.17591","repositories_listed":1,"syntology":null},{"url":"/paper/unambiguous-recognition-should-not-rely","slug":"unambiguous-recognition-should-not-rely","title":"MixTex: Unambiguous Recognition Should Not Rely Solely on Real Data","date":"2024-06-24","arxiv_id":"2406.17148","repositories_listed":1,"syntology":null},{"url":"/paper/guicourse-from-general-vision-language-models","slug":"guicourse-from-general-vision-language-models","title":"GUICourse: From General Vision Language Models to Versatile GUI Agents","date":"2024-06-17","arxiv_id":"2406.11317","repositories_listed":1,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":16,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/guicourse-from-general-vision-language-models#ran","syntology_url":"https://syntology.ai/paper/2406.11317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11317"}},"official":{"repos":["yiye3/guicourse"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/m3t-a-new-benchmark-dataset-for-multi-modal","slug":"m3t-a-new-benchmark-dataset-for-multi-modal","title":"M3T: A New Benchmark Dataset for Multi-Modal Document-Level Machine Translation","date":"2024-06-12","arxiv_id":"2406.08255","repositories_listed":1,"syntology":null},{"url":"/paper/vcr-visual-caption-restoration","slug":"vcr-visual-caption-restoration","title":"VCR: A Task for Pixel-Level Complex Reasoning in Vision Language Models via Restoring Occluded Text","date":"2024-06-10","arxiv_id":"2406.06462","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vcr-visual-caption-restoration#ran","syntology_url":"https://syntology.ai/paper/2406.06462","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06462"}},"official":{"repos":["tianyu-z/vcr"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/coru-comprehensive-post-ocr-parsing-and","slug":"coru-comprehensive-post-ocr-parsing-and","title":"CORU: Comprehensive Post-OCR Parsing and Receipt Understanding Dataset","date":"2024-06-06","arxiv_id":"2406.04493","repositories_listed":1,"syntology":null},{"url":"/paper/focus-anywhere-for-fine-grained-multi-page","slug":"focus-anywhere-for-fine-grained-multi-page","title":"Focus Anywhere for Fine-grained Multi-page Document Understanding","date":"2024-05-23","arxiv_id":"2405.14295","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/focus-anywhere-for-fine-grained-multi-page#ran","syntology_url":"https://syntology.ai/paper/2405.14295","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14295"}},"official":null}},{"url":"/paper/from-text-to-pixel-advancing-long-context","slug":"from-text-to-pixel-advancing-long-context","title":"From Text to Pixel: Advancing Long-Context Understanding in MLLMs","date":"2024-05-23","arxiv_id":"2405.14213","repositories_listed":1,"syntology":null},{"url":"/paper/let-s-fuse-step-by-step-a-generative-fusion","slug":"let-s-fuse-step-by-step-a-generative-fusion","title":"Let's Fuse Step by Step: A Generative Fusion Decoding Algorithm with LLMs for Multi-modal Text Recognition","date":"2024-05-23","arxiv_id":"2405.14259","repositories_listed":1,"syntology":null},{"url":"/paper/geocontrastnet-contrastive-key-value-edge","slug":"geocontrastnet-contrastive-key-value-edge","title":"GeoContrastNet: Contrastive Key-Value Edge Learning for Language-Agnostic Document Understanding","date":"2024-05-06","arxiv_id":"2405.03104","repositories_listed":1,"syntology":null},{"url":"/paper/multi-page-document-visual-question-answering","slug":"multi-page-document-visual-question-answering","title":"Multi-Page Document Visual Question Answering using Self-Attention Scoring Mechanism","date":"2024-04-29","arxiv_id":"2404.19024","repositories_listed":1,"syntology":null},{"url":"/paper/how-far-are-we-to-gpt-4v-closing-the-gap-to","slug":"how-far-are-we-to-gpt-4v-closing-the-gap-to","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","date":"2024-04-25","arxiv_id":"2404.16821","repositories_listed":1,"syntology":null},{"url":"/paper/convolution-based-probability-gradient-loss","slug":"convolution-based-probability-gradient-loss","title":"Convolution-based Probability Gradient Loss for Semantic Segmentation","date":"2024-04-10","arxiv_id":"2404.06704","repositories_listed":1,"syntology":null},{"url":"/paper/visualwebbench-how-far-have-multimodal-llms","slug":"visualwebbench-how-far-have-multimodal-llms","title":"VisualWebBench: How Far Have Multimodal LLMs Evolved in Web Page Understanding and Grounding?","date":"2024-04-09","arxiv_id":"2404.05955","repositories_listed":1,"syntology":null},{"url":"/paper/cmulab-an-open-source-framework-for-training","slug":"cmulab-an-open-source-framework-for-training","title":"CMULAB: An Open-Source Framework for Training and Deployment of Natural Language Processing Models","date":"2024-04-03","arxiv_id":"2404.02408","repositories_listed":1,"syntology":null},{"url":"/paper/draw-and-understand-leveraging-visual-prompts","slug":"draw-and-understand-leveraging-visual-prompts","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","date":"2024-03-29","arxiv_id":"2403.20271","repositories_listed":1,"syntology":null},{"url":"/paper/chroniclingamericaqa-a-large-scale-question","slug":"chroniclingamericaqa-a-large-scale-question","title":"ChroniclingAmericaQA: A Large-scale Question Answering Dataset based on Historical American Newspaper Pages","date":"2024-03-26","arxiv_id":"2403.17859","repositories_listed":1,"syntology":null},{"url":"/paper/visually-guided-generative-text-layout-pre","slug":"visually-guided-generative-text-layout-pre","title":"Visually Guided Generative Text-Layout Pre-training for Document Intelligence","date":"2024-03-25","arxiv_id":"2403.16516","repositories_listed":1,"syntology":null},{"url":"/paper/peace-a-chemistry-oriented-dataset-for","slug":"peace-a-chemistry-oriented-dataset-for","title":"PEaCE: A Chemistry-Oriented Dataset for Optical Character Recognition on Scientific Documents","date":"2024-03-23","arxiv_id":"2403.15724","repositories_listed":1,"syntology":null},{"url":"/paper/mplug-docowl-1-5-unified-structure-learning","slug":"mplug-docowl-1-5-unified-structure-learning","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","date":"2024-03-19","arxiv_id":"2403.12895","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mplug-docowl-1-5-unified-structure-learning#ran","syntology_url":"https://syntology.ai/paper/2403.12895","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12895"}},"official":{"repos":["x-plug/mplug-docowl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/advancing-multilingual-handwritten-numeral","slug":"advancing-multilingual-handwritten-numeral","title":"Advancing Multilingual Handwritten Numeral Recognition with Attention-driven Transfer Learning","date":"2024-03-18","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-training-with-ocr-modality","slug":"adversarial-training-with-ocr-modality","title":"Adversarial Training with OCR Modality Perturbation for Scene-Text Visual Question Answering","date":"2024-03-14","arxiv_id":"2403.09288","repositories_listed":1,"syntology":null},{"url":"/paper/deepseek-vl-towards-real-world-vision","slug":"deepseek-vl-towards-real-world-vision","title":"DeepSeek-VL: Towards Real-World Vision-Language Understanding","date":"2024-03-08","arxiv_id":"2403.05525","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":7,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepseek-vl-towards-real-world-vision#ran","syntology_url":"https://syntology.ai/paper/2403.05525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05525"}},"official":{"repos":["deepseek-ai/deepseek-vl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/textmonkey-an-ocr-free-large-multimodal-model","slug":"textmonkey-an-ocr-free-large-multimodal-model","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","date":"2024-03-07","arxiv_id":"2403.04473","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textmonkey-an-ocr-free-large-multimodal-model#ran","syntology_url":"https://syntology.ai/paper/2403.04473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04473"}},"official":{"repos":["yuliang-liu/monkey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/odm-a-text-image-further-alignment-pre","slug":"odm-a-text-image-further-alignment-pre","title":"ODM: A Text-Image Further Alignment Pre-training Approach for Scene Text Detection and Spotting","date":"2024-03-01","arxiv_id":"2403.00303","repositories_listed":1,"syntology":null},{"url":"/paper/syntactic-language-change-in-english-and","slug":"syntactic-language-change-in-english-and","title":"Syntactic Language Change in English and German: Metrics, Parsers, and Convergences","date":"2024-02-18","arxiv_id":"2402.11549","repositories_listed":1,"syntology":null},{"url":"/paper/textron-weakly-supervised-multilingual-text-1","slug":"textron-weakly-supervised-multilingual-text-1","title":"TEXTRON: Weakly Supervised Multilingual Text Detection through Data Programming","date":"2024-02-15","arxiv_id":"2402.09811","repositories_listed":1,"syntology":null},{"url":"/paper/clustertabnet-supervised-clustering-method","slug":"clustertabnet-supervised-clustering-method","title":"ClusterTabNet: Supervised clustering method for table detection and table structure recognition","date":"2024-02-12","arxiv_id":"2402.07502","repositories_listed":1,"syntology":{"n":16,"n_ran":16,"n_constructed":0,"n_ran_checked":13,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":12,"n_pointer_only":4,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clustertabnet-supervised-clustering-method#ran","syntology_url":"https://syntology.ai/paper/2402.07502","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07502"}},"official":{"repos":["sap-samples/clustertabnet"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sphinx-x-scaling-data-and-parameters-for-a","slug":"sphinx-x-scaling-data-and-parameters-for-a","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","date":"2024-02-08","arxiv_id":"2402.05935","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sphinx-x-scaling-data-and-parameters-for-a#ran","syntology_url":"https://syntology.ai/paper/2402.05935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05935"}},"official":{"repos":["alpha-vllm/llama2-accessory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mousi-poly-visual-expert-vision-language","slug":"mousi-poly-visual-expert-vision-language","title":"MouSi: Poly-Visual-Expert Vision-Language Models","date":"2024-01-30","arxiv_id":"2401.17221","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-scaling-law-for-scene","slug":"an-empirical-study-of-scaling-law-for-scene","title":"An Empirical Study of Scaling Law for Scene Text Recognition","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/efficient-multi-domain-text-recognition-deep","slug":"efficient-multi-domain-text-recognition-deep","title":"Efficient Multi-domain Text Recognition Deep Neural Network Parameterization with Residual Adapters","date":"2024-01-01","arxiv_id":"2401.00971","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-scaling-law-for-ocr","slug":"an-empirical-study-of-scaling-law-for-ocr","title":"An Empirical Study of Scaling Law for OCR","date":"2023-12-29","arxiv_id":"2401.00028","repositories_listed":1,"syntology":null},{"url":"/paper/when-graph-data-meets-multimodal-a-new","slug":"when-graph-data-meets-multimodal-a-new","title":"When Graph Data Meets Multimodal: A New Paradigm for Graph Understanding and Reasoning","date":"2023-12-16","arxiv_id":"2312.10372","repositories_listed":1,"syntology":null},{"url":"/paper/privacy-aware-document-visual-question","slug":"privacy-aware-document-visual-question","title":"Privacy-Aware Document Visual Question Answering","date":"2023-12-15","arxiv_id":"2312.10108","repositories_listed":1,"syntology":null},{"url":"/paper/vary-scaling-up-the-vision-vocabulary-for","slug":"vary-scaling-up-the-vision-vocabulary-for","title":"Vary: Scaling up the Vision Vocabulary for Large Vision-Language Models","date":"2023-12-11","arxiv_id":"2312.06109","repositories_listed":1,"syntology":null},{"url":"/paper/idpl-pfod2-a-new-large-scale-dataset-for","slug":"idpl-pfod2-a-new-large-scale-dataset-for","title":"IDPL-PFOD2: A New Large-Scale Dataset for Printed Farsi Optical Character Recognition","date":"2023-12-02","arxiv_id":"2312.01177","repositories_listed":1,"syntology":null},{"url":"/paper/sut-a-new-multi-purpose-synthetic-dataset-for","slug":"sut-a-new-multi-purpose-synthetic-dataset-for","title":"SUT: a new multi-purpose synthetic dataset for Farsi document image analysis","date":"2023-11-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reading-between-the-mud-a-challenging","slug":"reading-between-the-mud-a-challenging","title":"Reading Between the Mud: A Challenging Motorcycle Racer Number Dataset","date":"2023-11-14","arxiv_id":"2311.09256","repositories_listed":1,"syntology":null},{"url":"/paper/anytext-multilingual-visual-text-generation","slug":"anytext-multilingual-visual-text-generation","title":"AnyText: Multilingual Visual Text Generation And Editing","date":"2023-11-06","arxiv_id":"2311.03054","repositories_listed":1,"syntology":{"n":19,"n_ran":14,"n_constructed":0,"n_ran_checked":12,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/anytext-multilingual-visual-text-generation#ran","syntology_url":"https://syntology.ai/paper/2311.03054","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.03054"}},"official":{"repos":["tyxsspa/anytext"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/on-manipulating-scene-text-in-the-wild-with","slug":"on-manipulating-scene-text-in-the-wild-with","title":"On Manipulating Scene Text in the Wild with Diffusion Models","date":"2023-11-01","arxiv_id":"2311.00734","repositories_listed":1,"syntology":null},{"url":"/paper/dcqa-document-level-chart-question-answering","slug":"dcqa-document-level-chart-question-answering","title":"DCQA: Document-Level Chart Question Answering towards Complex Reasoning and Common-Sense Understanding","date":"2023-10-29","arxiv_id":"2310.18983","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-ocr-capabilities-of-gpt-4v-ision-a","slug":"exploring-ocr-capabilities-of-gpt-4v-ision-a","title":"Exploring OCR Capabilities of GPT-4V(ision) : A Quantitative and In-depth Evaluation","date":"2023-10-25","arxiv_id":"2310.16809","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploring-ocr-capabilities-of-gpt-4v-ision-a#ran","syntology_url":"https://syntology.ai/paper/2310.16809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16809"}},"official":{"repos":["scut-dlvclab/gpt-4v_ocr"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/genkie-robust-generative-multimodal-document","slug":"genkie-robust-generative-multimodal-document","title":"GenKIE: Robust Generative Multimodal Document Key Information Extraction","date":"2023-10-24","arxiv_id":"2310.16131","repositories_listed":1,"syntology":null},{"url":"/paper/phd-pixel-based-language-modeling-of","slug":"phd-pixel-based-language-modeling-of","title":"PHD: Pixel-Based Language Modeling of Historical Documents","date":"2023-10-22","arxiv_id":"2310.18343","repositories_listed":1,"syntology":null},{"url":"/paper/docxchain-a-powerful-open-source-toolchain","slug":"docxchain-a-powerful-open-source-toolchain","title":"DocXChain: A Powerful Open-Source Toolchain for Document Parsing and Beyond","date":"2023-10-19","arxiv_id":"2310.12430","repositories_listed":1,"syntology":null},{"url":"/paper/dsg-an-end-to-end-document-structure","slug":"dsg-an-end-to-end-document-structure","title":"DSG: An End-to-End Document Structure Generator","date":"2023-10-13","arxiv_id":"2310.09118","repositories_listed":1,"syntology":null},{"url":"/paper/persis-a-persian-font-recognition-pipeline-1","slug":"persis-a-persian-font-recognition-pipeline-1","title":"Persis: A Persian Font Recognition Pipeline Using Convolutional Neural Networks","date":"2023-10-08","arxiv_id":"2310.05255","repositories_listed":1,"syntology":null},{"url":"/paper/symmetrical-linguistic-feature-distillation","slug":"symmetrical-linguistic-feature-distillation","title":"Symmetrical Linguistic Feature Distillation with CLIP for Scene Text Recognition","date":"2023-10-08","arxiv_id":"2310.04999","repositories_listed":1,"syntology":null},{"url":"/paper/ureader-universal-ocr-free-visually-situated","slug":"ureader-universal-ocr-free-visually-situated","title":"UReader: Universal OCR-free Visually-situated Language Understanding with Multimodal Large Language Model","date":"2023-10-08","arxiv_id":"2310.05126","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ureader-universal-ocr-free-visually-situated#ran","syntology_url":"https://syntology.ai/paper/2310.05126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05126"}},"official":{"repos":["lukeforeveryoung/ureader"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mathvista-evaluating-mathematical-reasoning","slug":"mathvista-evaluating-mathematical-reasoning","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","date":"2023-10-03","arxiv_id":"2310.02255","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathvista-evaluating-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2310.02255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02255"}},"official":null}},{"url":"/paper/order-preserving-consistency-regularization","slug":"order-preserving-consistency-regularization","title":"Order-preserving Consistency Regularization for Domain Adaptation and Generalization","date":"2023-09-23","arxiv_id":"2309.13258","repositories_listed":1,"syntology":null},{"url":"/paper/step-towards-structured-scene-text-spotting","slug":"step-towards-structured-scene-text-spotting","title":"STEP -- Towards Structured Scene-Text Spotting","date":"2023-09-05","arxiv_id":"2309.02356","repositories_listed":1,"syntology":null},{"url":"/paper/separate-and-locate-rethink-the-text-in-text","slug":"separate-and-locate-rethink-the-text-in-text","title":"Separate and Locate: Rethink the Text in Text-based Visual Question Answering","date":"2023-08-31","arxiv_id":"2308.16383","repositories_listed":1,"syntology":null},{"url":"/paper/dtrocr-decoder-only-transformer-for-optical","slug":"dtrocr-decoder-only-transformer-for-optical","title":"DTrOCR: Decoder-only Transformer for Optical Character Recognition","date":"2023-08-30","arxiv_id":"2308.15996","repositories_listed":1,"syntology":null},{"url":"/paper/vision-grid-transformer-for-document-layout","slug":"vision-grid-transformer-for-document-layout","title":"Vision Grid Transformer for Document Layout Analysis","date":"2023-08-29","arxiv_id":"2308.14978","repositories_listed":1,"syntology":null},{"url":"/paper/bbocr-an-open-source-multi-domain-ocr","slug":"bbocr-an-open-source-multi-domain-ocr","title":"bbOCR: An Open-source Multi-domain OCR Pipeline for Bengali Documents","date":"2023-08-21","arxiv_id":"2308.10647","repositories_listed":1,"syntology":null},{"url":"/paper/bliva-a-simple-multimodal-llm-for-better","slug":"bliva-a-simple-multimodal-llm-for-better","title":"BLIVA: A Simple Multimodal LLM for Better Handling of Text-Rich Visual Questions","date":"2023-08-19","arxiv_id":"2308.09936","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bliva-a-simple-multimodal-llm-for-better#ran","syntology_url":"https://syntology.ai/paper/2308.09936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09936"}},"official":{"repos":["mlpc-ucsd/bliva"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/fashionlogo-prompting-multimodal-large","slug":"fashionlogo-prompting-multimodal-large","title":"FashionLOGO: Prompting Multimodal Large Language Models for Fashion Logo Embeddings","date":"2023-08-17","arxiv_id":"2308.09012","repositories_listed":1,"syntology":null},{"url":"/paper/omnidatacomposer-a-unified-data-structure-for","slug":"omnidatacomposer-a-unified-data-structure-for","title":"OmniDataComposer: A Unified Data Structure for Multimodal Data Fusion and Infinite Data Generation","date":"2023-08-08","arxiv_id":"2308.04126","repositories_listed":1,"syntology":null},{"url":"/paper/universal-defensive-underpainting-patch","slug":"universal-defensive-underpainting-patch","title":"Universal Defensive Underpainting Patch: Making Your Text Invisible to Optical Character Recognition","date":"2023-08-04","arxiv_id":"2308.02369","repositories_listed":1,"syntology":null},{"url":"/paper/augmented-math-authoring-ar-based-explorable","slug":"augmented-math-authoring-ar-based-explorable","title":"Augmented Math: Authoring AR-Based Explorable Explanations by Augmenting Static Math Textbooks","date":"2023-07-30","arxiv_id":"2307.16112","repositories_listed":1,"syntology":null},{"url":"/paper/validation-of-a-zero-shot-learning-natural","slug":"validation-of-a-zero-shot-learning-natural","title":"Validation of a Zero-Shot Learning Natural Language Processing Tool for Data Abstraction from Unstructured Healthcare Data","date":"2023-07-23","arxiv_id":"2308.00107","repositories_listed":1,"syntology":null},{"url":"/paper/multiqg-ti-towards-question-generation-from","slug":"multiqg-ti-towards-question-generation-from","title":"MultiQG-TI: Towards Question Generation from Multi-modal Sources","date":"2023-07-07","arxiv_id":"2307.04643","repositories_listed":1,"syntology":null},{"url":"/paper/t-mars-improving-visual-representations-by","slug":"t-mars-improving-visual-representations-by","title":"T-MARS: Improving Visual Representations by Circumventing Text Feature Learning","date":"2023-07-06","arxiv_id":"2307.03132","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/t-mars-improving-visual-representations-by#ran","syntology_url":"https://syntology.ai/paper/2307.03132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.03132"}},"official":{"repos":["locuslab/t-mars"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mplug-docowl-modularized-multimodal-large","slug":"mplug-docowl-modularized-multimodal-large","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","date":"2023-07-04","arxiv_id":"2307.02499","repositories_listed":1,"syntology":null},{"url":"/paper/utrnet-high-resolution-urdu-text-recognition","slug":"utrnet-high-resolution-urdu-text-recognition","title":"UTRNet: High-Resolution Urdu Text Recognition In Printed Documents","date":"2023-06-27","arxiv_id":"2306.15782","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-multimodal-large-language-models","slug":"a-survey-on-multimodal-large-language-models","title":"A Survey on Multimodal Large Language Models","date":"2023-06-23","arxiv_id":"2306.13549","repositories_listed":1,"syntology":null},{"url":"/paper/document-image-cleaning-using-budget-aware","slug":"document-image-cleaning-using-budget-aware","title":"Document Image Cleaning using Budget-Aware Black-Box Approximation","date":"2023-06-22","arxiv_id":"2306.13236","repositories_listed":1,"syntology":null},{"url":"/paper/genplot-increasing-the-scale-and-diversity-of","slug":"genplot-increasing-the-scale-and-diversity-of","title":"GenPlot: Increasing the Scale and Diversity of Chart Derendering Data","date":"2023-06-20","arxiv_id":"2306.11699","repositories_listed":1,"syntology":null},{"url":"/paper/when-vision-fails-text-attacks-against-vit","slug":"when-vision-fails-text-attacks-against-vit","title":"When Vision Fails: Text Attacks Against ViT and OCR","date":"2023-06-12","arxiv_id":"2306.07033","repositories_listed":1,"syntology":null},{"url":"/paper/scicap-a-knowledge-augmented-dataset-to-study","slug":"scicap-a-knowledge-augmented-dataset-to-study","title":"SciCap+: A Knowledge Augmented Dataset to Study the Challenges of Scientific Figure Captioning","date":"2023-06-06","arxiv_id":"2306.03491","repositories_listed":1,"syntology":null},{"url":"/paper/transdocanalyser-a-framework-for-offline-semi","slug":"transdocanalyser-a-framework-for-offline-semi","title":"TransDocAnalyser: A Framework for Offline Semi-structured Handwritten Document Analysis in the Legal Domain","date":"2023-06-03","arxiv_id":"2306.02142","repositories_listed":1,"syntology":null},{"url":"/paper/docformerv2-local-features-for-document","slug":"docformerv2-local-features-for-document","title":"DocFormerv2: Local Features for Document Understanding","date":"2023-06-02","arxiv_id":"2306.01733","repositories_listed":1,"syntology":null},{"url":"/paper/a-template-independent-approach-for","slug":"a-template-independent-approach-for","title":"A template-independent approach for information extraction in real estate documents","date":"2023-05-30","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/duosearch-a-novel-search-engine-for-bulgarian","slug":"duosearch-a-novel-search-engine-for-bulgarian","title":"DuoSearch: A Novel Search Engine for Bulgarian Historical Documents","date":"2023-05-30","arxiv_id":"2305.19392","repositories_listed":1,"syntology":null},{"url":"/paper/glyphcontrol-glyph-conditional-control-for-1","slug":"glyphcontrol-glyph-conditional-control-for-1","title":"GlyphControl: Glyph Conditional Control for Visual Text Generation","date":"2023-05-29","arxiv_id":"2305.18259","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/glyphcontrol-glyph-conditional-control-for-1#ran","syntology_url":"https://syntology.ai/paper/2305.18259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18259"}},"official":{"repos":["aigtext/glyphcontrol-release"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/fusecap-leveraging-large-language-models-to","slug":"fusecap-leveraging-large-language-models-to","title":"FuseCap: Leveraging Large Language Models for Enriched Fused Image Captions","date":"2023-05-28","arxiv_id":"2305.17718","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-better-text-image-translation-with","slug":"exploring-better-text-image-translation-with","title":"Exploring Better Text Image Translation with Multimodal Codebook","date":"2023-05-27","arxiv_id":"2305.17415","repositories_listed":1,"syntology":null},{"url":"/paper/mrn-multiplexed-routing-network-for","slug":"mrn-multiplexed-routing-network-for","title":"MRN: Multiplexed Routing Network for Incremental Multilingual Text Recognition","date":"2023-05-24","arxiv_id":"2305.14758","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-character-similarity-with-vision","slug":"quantifying-character-similarity-with-vision","title":"Quantifying Character Similarity with Vision Transformers","date":"2023-05-24","arxiv_id":"2305.14672","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-intersectional-biases-in-historical","slug":"measuring-intersectional-biases-in-historical","title":"Measuring Intersectional Biases in Historical Documents","date":"2023-05-21","arxiv_id":"2305.12376","repositories_listed":1,"syntology":null},{"url":"/paper/xtreme-up-a-user-centric-scarce-data","slug":"xtreme-up-a-user-centric-scarce-data","title":"XTREME-UP: A User-Centric Scarce-Data Benchmark for Under-Represented Languages","date":"2023-05-19","arxiv_id":"2305.11938","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/xtreme-up-a-user-centric-scarce-data#ran","syntology_url":"https://syntology.ai/paper/2305.11938","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11938"}},"official":{"repos":["google-research/xtreme-up"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mobile-user-interface-element-detection-via-1","slug":"mobile-user-interface-element-detection-via-1","title":"Mobile User Interface Element Detection Via Adaptively Prompt Tuning","date":"2023-05-16","arxiv_id":"2305.09699","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-hidden-mystery-of-ocr-in-large","slug":"on-the-hidden-mystery-of-ocr-in-large","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","date":"2023-05-13","arxiv_id":"2305.07895","repositories_listed":1,"syntology":null},{"url":"/paper/visual-information-extraction-in-the-wild","slug":"visual-information-extraction-in-the-wild","title":"Visual Information Extraction in the Wild: Practical Dataset and End-to-end Solution","date":"2023-05-12","arxiv_id":"2305.07498","repositories_listed":1,"syntology":null},{"url":"/paper/combining-ocr-models-for-reading-early-modern","slug":"combining-ocr-models-for-reading-early-modern","title":"Combining OCR Models for Reading Early Modern Printed Books","date":"2023-05-11","arxiv_id":"2305.07131","repositories_listed":1,"syntology":null},{"url":"/paper/e2timt-efficient-and-effective-modal-adapter","slug":"e2timt-efficient-and-effective-modal-adapter","title":"E2TIMT: Efficient and Effective Modal Adapter for Text Image Machine Translation","date":"2023-05-09","arxiv_id":"2305.05166","repositories_listed":1,"syntology":null},{"url":"/paper/tps-attention-enhanced-thin-plate-spline-for","slug":"tps-attention-enhanced-thin-plate-spline-for","title":"TPS++: Attention-Enhanced Thin-Plate Spline for Scene Text Recognition","date":"2023-05-09","arxiv_id":"2305.05322","repositories_listed":1,"syntology":null},{"url":"/paper/docparser-end-to-end-ocr-free-information","slug":"docparser-end-to-end-ocr-free-information","title":"DocParser: End-to-end OCR-free Information Extraction from Visually Rich Documents","date":"2023-04-24","arxiv_id":"2304.12484","repositories_listed":1,"syntology":null},{"url":"/paper/transdocs-optical-character-recognition-with","slug":"transdocs-optical-character-recognition-with","title":"TransDocs: Optical Character Recognition with word to word translation","date":"2023-04-15","arxiv_id":"2304.07637","repositories_listed":1,"syntology":null},{"url":"/paper/taggpt-large-language-models-are-zero-shot","slug":"taggpt-large-language-models-are-zero-shot","title":"TagGPT: Large Language Models are Zero-shot Multimodal Taggers","date":"2023-04-06","arxiv_id":"2304.03022","repositories_listed":1,"syntology":null},{"url":"/paper/chartreader-a-unified-framework-for-chart","slug":"chartreader-a-unified-framework-for-chart","title":"ChartReader: A Unified Framework for Chart Derendering and Comprehension without Heuristic Rules","date":"2023-04-05","arxiv_id":"2304.02173","repositories_listed":1,"syntology":{"n":19,"n_ran":16,"n_constructed":3,"n_ran_checked":6,"n_instrument":10,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":19,"phrase":"16 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 10 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/chartreader-a-unified-framework-for-chart#ran","syntology_url":"https://syntology.ai/paper/2304.02173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.02173"}},"official":{"repos":["zhiqic/chartreader"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":3,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-ocr-for-building-a-diverse-digital","slug":"efficient-ocr-for-building-a-diverse-digital","title":"Efficient OCR for Building a Diverse Digital History","date":"2023-04-05","arxiv_id":"2304.02737","repositories_listed":1,"syntology":null},{"url":"/paper/a-study-of-autoregressive-decoders-for-multi","slug":"a-study-of-autoregressive-decoders-for-multi","title":"A Study of Autoregressive Decoders for Multi-Tasking in Computer Vision","date":"2023-03-30","arxiv_id":"2303.17376","repositories_listed":1,"syntology":null},{"url":"/paper/ovenet-offset-vector-network-for-semantic","slug":"ovenet-offset-vector-network-for-semantic","title":"OVeNet: Offset Vector Network for Semantic Segmentation","date":"2023-03-25","arxiv_id":"2303.14516","repositories_listed":1,"syntology":null},{"url":"/paper/badlad-a-large-multi-domain-bengali-document","slug":"badlad-a-large-multi-domain-bengali-document","title":"BaDLAD: A Large Multi-Domain Bengali Document Layout Analysis Dataset","date":"2023-03-09","arxiv_id":"2303.05325","repositories_listed":1,"syntology":null},{"url":"/paper/structextv2-masked-visual-textual-prediction","slug":"structextv2-masked-visual-textual-prediction","title":"StrucTexTv2: Masked Visual-Textual Prediction for Document Image Pre-training","date":"2023-03-01","arxiv_id":"2303.00289","repositories_listed":1,"syntology":null},{"url":"/paper/language-is-not-all-you-need-aligning-1","slug":"language-is-not-all-you-need-aligning-1","title":"Language Is Not All You Need: Aligning Perception with Language Models","date":"2023-02-27","arxiv_id":"2302.14045","repositories_listed":1,"syntology":null}],"record_sha256":"ac8e5f883890ea7b47c2bcd9edfb5411fcfcfb985c58282c56d23ab5efb777c8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}