{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/optical-character-recognition/papers/ran/1","list_of":"/task/optical-character-recognition","task":"Optical Character Recognition (OCR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,76],"of":76,"counts":{"archive_papers_tagged":1243,"with_a_code_link":462,"where_syntology_ran_a_sample":76,"not_listed_spam_title":0,"listed":1243,"listed_where_code_ran":76,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/optical-character-recognition/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/reviving-cultural-heritage-a-novel-approach","slug":"reviving-cultural-heritage-a-novel-approach","title":"Reviving Cultural Heritage: A Novel Approach for Comprehensive Historical Document Restoration","date":"2025-07-07","arxiv_id":"2507.05108","repositories_listed":1,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":12,"n_pointer_only":17,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/reviving-cultural-heritage-a-novel-approach#ran","syntology_url":"https://syntology.ai/paper/2507.05108","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.05108"}},"official":null}},{"url":"/paper/uni-mumer-unified-multi-task-fine-tuning-of","slug":"uni-mumer-unified-multi-task-fine-tuning-of","title":"Uni-MuMER: Unified Multi-Task Fine-Tuning of Vision-Language Model for Handwritten Mathematical Expression Recognition","date":"2025-05-29","arxiv_id":"2505.23566","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":1,"n_ran_checked":3,"n_instrument":4,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uni-mumer-unified-multi-task-fine-tuning-of#ran","syntology_url":"https://syntology.ai/paper/2505.23566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23566"}},"official":{"repos":["bflameswift/uni-mumer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-action-model-with-open-world","slug":"vision-language-action-model-with-open-world","title":"ChatVLA-2: Vision-Language-Action Model with Open-World Embodied Reasoning from Pretrained Knowledge","date":"2025-05-28","arxiv_id":"2505.21906","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vision-language-action-model-with-open-world#ran","syntology_url":"https://syntology.ai/paper/2505.21906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21906"}},"official":null}},{"url":"/paper/vidtext-towards-comprehensive-evaluation-for","slug":"vidtext-towards-comprehensive-evaluation-for","title":"VidText: Towards Comprehensive Evaluation for Video Text Understanding","date":"2025-05-28","arxiv_id":"2505.22810","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vidtext-towards-comprehensive-evaluation-for#ran","syntology_url":"https://syntology.ai/paper/2505.22810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22810"}},"official":{"repos":["shuyansy/vidtext"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-multimodal-large-language-model","slug":"unifying-multimodal-large-language-model","title":"Unifying Multimodal Large Language Model Capabilities and Modalities via Model Merging","date":"2025-05-26","arxiv_id":"2505.19892","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2505.19892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19892"}},"official":{"repos":["walkerworldpeace/mllmerging"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ocr-reasoning-benchmark-unveiling-the-true","slug":"ocr-reasoning-benchmark-unveiling-the-true","title":"OCR-Reasoning Benchmark: Unveiling the True Capabilities of MLLMs in Complex Text-Rich Image Reasoning","date":"2025-05-22","arxiv_id":"2505.17163","repositories_listed":0,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ocr-reasoning-benchmark-unveiling-the-true#ran","syntology_url":"https://syntology.ai/paper/2505.17163","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17163"}},"official":null}},{"url":"/paper/scalable-video-to-dataset-generation-for","slug":"scalable-video-to-dataset-generation-for","title":"Scalable Video-to-Dataset Generation for Cross-Platform Mobile Agents","date":"2025-05-19","arxiv_id":"2505.12632","repositories_listed":0,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scalable-video-to-dataset-generation-for#ran","syntology_url":"https://syntology.ai/paper/2505.12632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12632"}},"official":null}},{"url":"/paper/lvagent-long-video-understanding-by-multi","slug":"lvagent-long-video-understanding-by-multi","title":"LVAgent: Long Video Understanding by Multi-Round Dynamical Collaboration of MLLM Agents","date":"2025-03-13","arxiv_id":"2503.10200","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lvagent-long-video-understanding-by-multi#ran","syntology_url":"https://syntology.ai/paper/2503.10200","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10200"}},"official":null}},{"url":"/paper/benchmarking-vision-language-models-on","slug":"benchmarking-vision-language-models-on","title":"Benchmarking Vision-Language Models on Optical Character Recognition in Dynamic Video Environments","date":"2025-02-10","arxiv_id":"2502.06445","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-vision-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2502.06445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06445"}},"official":{"repos":["video-db/ocr-benchmark"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ocrbench-v2-an-improved-benchmark-for","slug":"ocrbench-v2-an-improved-benchmark-for","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","date":"2024-12-31","arxiv_id":"2501.00321","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ocrbench-v2-an-improved-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2501.00321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.00321"}},"official":{"repos":["yuliang-liu/multimodalocr"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deepseek-vl2-mixture-of-experts-vision","slug":"deepseek-vl2-mixture-of-experts-vision","title":"DeepSeek-VL2: Mixture-of-Experts Vision-Language Models for Advanced Multimodal Understanding","date":"2024-12-13","arxiv_id":"2412.10302","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deepseek-vl2-mixture-of-experts-vision#ran","syntology_url":"https://syntology.ai/paper/2412.10302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.10302"}},"official":{"repos":["deepseek-ai/deepseek-vl2"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/paligemma-2-a-family-of-versatile-vlms-for","slug":"paligemma-2-a-family-of-versatile-vlms-for","title":"PaliGemma 2: A Family of Versatile VLMs for Transfer","date":"2024-12-04","arxiv_id":"2412.03555","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/paligemma-2-a-family-of-versatile-vlms-for#ran","syntology_url":"https://syntology.ai/paper/2412.03555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03555"}},"official":null}},{"url":"/paper/toxicity-of-the-commons-curating-open-source","slug":"toxicity-of-the-commons-curating-open-source","title":"Toxicity of the Commons: Curating Open-Source Pre-Training Data","date":"2024-10-29","arxiv_id":"2410.22587","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/toxicity-of-the-commons-curating-open-source#ran","syntology_url":"https://syntology.ai/paper/2410.22587","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22587"}},"official":{"repos":["Pleias/toxic-commons"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/world-to-code-multi-modal-data-generation-via","slug":"world-to-code-multi-modal-data-generation-via","title":"World to Code: Multi-modal Data Generation via Self-Instructed Compositional Captioning and Filtering","date":"2024-09-30","arxiv_id":"2409.20424","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/world-to-code-multi-modal-data-generation-via#ran","syntology_url":"https://syntology.ai/paper/2409.20424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.20424"}},"official":{"repos":["foundation-multimodal-models/world2code"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mineru-an-open-source-solution-for-precise","slug":"mineru-an-open-source-solution-for-precise","title":"MinerU: An Open-Source Solution for Precise Document Content Extraction","date":"2024-09-27","arxiv_id":"2409.18839","repositories_listed":2,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mineru-an-open-source-solution-for-precise#ran","syntology_url":"https://syntology.ai/paper/2409.18839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.18839"}},"official":{"repos":["opendatalab/mineru","opendatalab/PDF-Extract-Kit"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mmmu-pro-a-more-robust-multi-discipline","slug":"mmmu-pro-a-more-robust-multi-discipline","title":"MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark","date":"2024-09-04","arxiv_id":"2409.02813","repositories_listed":2,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mmmu-pro-a-more-robust-multi-discipline#ran","syntology_url":"https://syntology.ai/paper/2409.02813","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02813"}},"official":null}},{"url":"/paper/eagle-exploring-the-design-space-for","slug":"eagle-exploring-the-design-space-for","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","date":"2024-08-28","arxiv_id":"2408.15998","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eagle-exploring-the-design-space-for#ran","syntology_url":"https://syntology.ai/paper/2408.15998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.15998"}},"official":{"repos":["nvlabs/eagle"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/doclayllm-an-efficient-and-effective-multi","slug":"doclayllm-an-efficient-and-effective-multi","title":"DocLayLLM: An Efficient and Effective Multi-modal Extension of Large Language Models for Text-rich Document Understanding","date":"2024-08-27","arxiv_id":"2408.15045","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/doclayllm-an-efficient-and-effective-multi#ran","syntology_url":"https://syntology.ai/paper/2408.15045","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.15045"}},"official":{"repos":["whlscut/DocLayLLM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seeing-and-understanding-bridging-vision-with","slug":"seeing-and-understanding-bridging-vision-with","title":"ChemVLM: Exploring the Power of Multimodal Large Language Models in Chemistry Area","date":"2024-08-14","arxiv_id":"2408.07246","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/seeing-and-understanding-bridging-vision-with#ran","syntology_url":"https://syntology.ai/paper/2408.07246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07246"}},"official":{"repos":["AI4Chem/ChemVlm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-01800","slug":"2408-01800","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","date":"2024-08-03","arxiv_id":"2408.01800","repositories_listed":2,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":7,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/2408-01800#ran","syntology_url":"https://syntology.ai/paper/2408.01800","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.01800"}},"official":{"repos":["OpenBMB/MiniCPM-o"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/focus-distinguish-and-prompt-unleashing-clip","slug":"focus-distinguish-and-prompt-unleashing-clip","title":"Focus, Distinguish, and Prompt: Unleashing CLIP for Efficient and Flexible Scene Text Retrieval","date":"2024-08-01","arxiv_id":"2408.00441","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/focus-distinguish-and-prompt-unleashing-clip#ran","syntology_url":"https://syntology.ai/paper/2408.00441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00441"}},"official":{"repos":["gyann-z/fdp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-00765","slug":"2408-00765","title":"MM-Vet v2: A Challenging Benchmark to Evaluate Large Multimodal Models for Integrated Capabilities","date":"2024-08-01","arxiv_id":"2408.00765","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2408-00765#ran","syntology_url":"https://syntology.ai/paper/2408.00765","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00765"}},"official":{"repos":["yuweihao/mm-vet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-bounding-box-is-worth-one-token","slug":"a-bounding-box-is-worth-one-token","title":"A Bounding Box is Worth One Token: Interleaving Layout and Text in a Large Language Model for Document Understanding","date":"2024-07-02","arxiv_id":"2407.01976","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":4,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-bounding-box-is-worth-one-token#ran","syntology_url":"https://syntology.ai/paper/2407.01976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01976"}},"official":{"repos":["laytextllm/laytextllm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mmlongbench-doc-benchmarking-long-context","slug":"mmlongbench-doc-benchmarking-long-context","title":"MMLongBench-Doc: Benchmarking Long-context Document Understanding with Visualizations","date":"2024-07-01","arxiv_id":"2407.01523","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mmlongbench-doc-benchmarking-long-context#ran","syntology_url":"https://syntology.ai/paper/2407.01523","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01523"}},"official":null}},{"url":"/paper/guicourse-from-general-vision-language-models","slug":"guicourse-from-general-vision-language-models","title":"GUICourse: From General Vision Language Models to Versatile GUI Agents","date":"2024-06-17","arxiv_id":"2406.11317","repositories_listed":1,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":16,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/guicourse-from-general-vision-language-models#ran","syntology_url":"https://syntology.ai/paper/2406.11317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11317"}},"official":{"repos":["yiye3/guicourse"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vcr-visual-caption-restoration","slug":"vcr-visual-caption-restoration","title":"VCR: A Task for Pixel-Level Complex Reasoning in Vision Language Models via Restoring Occluded Text","date":"2024-06-10","arxiv_id":"2406.06462","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vcr-visual-caption-restoration#ran","syntology_url":"https://syntology.ai/paper/2406.06462","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06462"}},"official":{"repos":["tianyu-z/vcr"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/focus-anywhere-for-fine-grained-multi-page","slug":"focus-anywhere-for-fine-grained-multi-page","title":"Focus Anywhere for Fine-grained Multi-page Document Understanding","date":"2024-05-23","arxiv_id":"2405.14295","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/focus-anywhere-for-fine-grained-multi-page#ran","syntology_url":"https://syntology.ai/paper/2405.14295","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14295"}},"official":null}},{"url":"/paper/mplug-docowl-1-5-unified-structure-learning","slug":"mplug-docowl-1-5-unified-structure-learning","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","date":"2024-03-19","arxiv_id":"2403.12895","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mplug-docowl-1-5-unified-structure-learning#ran","syntology_url":"https://syntology.ai/paper/2403.12895","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12895"}},"official":{"repos":["x-plug/mplug-docowl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deepseek-vl-towards-real-world-vision","slug":"deepseek-vl-towards-real-world-vision","title":"DeepSeek-VL: Towards Real-World Vision-Language Understanding","date":"2024-03-08","arxiv_id":"2403.05525","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":7,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepseek-vl-towards-real-world-vision#ran","syntology_url":"https://syntology.ai/paper/2403.05525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05525"}},"official":{"repos":["deepseek-ai/deepseek-vl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/textmonkey-an-ocr-free-large-multimodal-model","slug":"textmonkey-an-ocr-free-large-multimodal-model","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","date":"2024-03-07","arxiv_id":"2403.04473","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textmonkey-an-ocr-free-large-multimodal-model#ran","syntology_url":"https://syntology.ai/paper/2403.04473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04473"}},"official":{"repos":["yuliang-liu/monkey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/clustertabnet-supervised-clustering-method","slug":"clustertabnet-supervised-clustering-method","title":"ClusterTabNet: Supervised clustering method for table detection and table structure recognition","date":"2024-02-12","arxiv_id":"2402.07502","repositories_listed":1,"syntology":{"n":16,"n_ran":16,"n_constructed":0,"n_ran_checked":13,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":12,"n_pointer_only":4,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clustertabnet-supervised-clustering-method#ran","syntology_url":"https://syntology.ai/paper/2402.07502","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07502"}},"official":{"repos":["sap-samples/clustertabnet"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sphinx-x-scaling-data-and-parameters-for-a","slug":"sphinx-x-scaling-data-and-parameters-for-a","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","date":"2024-02-08","arxiv_id":"2402.05935","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sphinx-x-scaling-data-and-parameters-for-a#ran","syntology_url":"https://syntology.ai/paper/2402.05935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05935"}},"official":{"repos":["alpha-vllm/llama2-accessory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/figstep-jailbreaking-large-vision-language","slug":"figstep-jailbreaking-large-vision-language","title":"FigStep: Jailbreaking Large Vision-Language Models via Typographic Visual Prompts","date":"2023-11-09","arxiv_id":"2311.05608","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/figstep-jailbreaking-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2311.05608","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.05608"}},"official":{"repos":["thuccslab/figstep"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/anytext-multilingual-visual-text-generation","slug":"anytext-multilingual-visual-text-generation","title":"AnyText: Multilingual Visual Text Generation And Editing","date":"2023-11-06","arxiv_id":"2311.03054","repositories_listed":1,"syntology":{"n":19,"n_ran":14,"n_constructed":0,"n_ran_checked":12,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/anytext-multilingual-visual-text-generation#ran","syntology_url":"https://syntology.ai/paper/2311.03054","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.03054"}},"official":{"repos":["tyxsspa/anytext"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-ocr-capabilities-of-gpt-4v-ision-a","slug":"exploring-ocr-capabilities-of-gpt-4v-ision-a","title":"Exploring OCR Capabilities of GPT-4V(ision) : A Quantitative and In-depth Evaluation","date":"2023-10-25","arxiv_id":"2310.16809","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploring-ocr-capabilities-of-gpt-4v-ision-a#ran","syntology_url":"https://syntology.ai/paper/2310.16809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16809"}},"official":{"repos":["scut-dlvclab/gpt-4v_ocr"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ureader-universal-ocr-free-visually-situated","slug":"ureader-universal-ocr-free-visually-situated","title":"UReader: Universal OCR-free Visually-situated Language Understanding with Multimodal Large Language Model","date":"2023-10-08","arxiv_id":"2310.05126","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ureader-universal-ocr-free-visually-situated#ran","syntology_url":"https://syntology.ai/paper/2310.05126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05126"}},"official":{"repos":["lukeforeveryoung/ureader"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mathvista-evaluating-mathematical-reasoning","slug":"mathvista-evaluating-mathematical-reasoning","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","date":"2023-10-03","arxiv_id":"2310.02255","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathvista-evaluating-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2310.02255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02255"}},"official":null}},{"url":"/paper/nougat-neural-optical-understanding-for","slug":"nougat-neural-optical-understanding-for","title":"Nougat: Neural Optical Understanding for Academic Documents","date":"2023-08-25","arxiv_id":"2308.13418","repositories_listed":3,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/nougat-neural-optical-understanding-for#ran","syntology_url":"https://syntology.ai/paper/2308.13418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.13418"}},"official":{"repos":["facebookresearch/nougat"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bliva-a-simple-multimodal-llm-for-better","slug":"bliva-a-simple-multimodal-llm-for-better","title":"BLIVA: A Simple Multimodal LLM for Better Handling of Text-Rich Visual Questions","date":"2023-08-19","arxiv_id":"2308.09936","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bliva-a-simple-multimodal-llm-for-better#ran","syntology_url":"https://syntology.ai/paper/2308.09936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09936"}},"official":{"repos":["mlpc-ucsd/bliva"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/t-mars-improving-visual-representations-by","slug":"t-mars-improving-visual-representations-by","title":"T-MARS: Improving Visual Representations by Circumventing Text Feature Learning","date":"2023-07-06","arxiv_id":"2307.03132","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/t-mars-improving-visual-representations-by#ran","syntology_url":"https://syntology.ai/paper/2307.03132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.03132"}},"official":{"repos":["locuslab/t-mars"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llavar-enhanced-visual-instruction-tuning-for","slug":"llavar-enhanced-visual-instruction-tuning-for","title":"LLaVAR: Enhanced Visual Instruction Tuning for Text-Rich Image Understanding","date":"2023-06-29","arxiv_id":"2306.17107","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llavar-enhanced-visual-instruction-tuning-for#ran","syntology_url":"https://syntology.ai/paper/2306.17107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.17107"}},"official":{"repos":["SALT-NLP/LLaVAR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/layout-and-task-aware-instruction-prompt-for","slug":"layout-and-task-aware-instruction-prompt-for","title":"Layout and Task Aware Instruction Prompt for Zero-shot Document Image Question Answering","date":"2023-06-01","arxiv_id":"2306.00526","repositories_listed":3,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/layout-and-task-aware-instruction-prompt-for#ran","syntology_url":"https://syntology.ai/paper/2306.00526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00526"}},"official":{"repos":["wenjinw/latin-prompt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/glyphcontrol-glyph-conditional-control-for-1","slug":"glyphcontrol-glyph-conditional-control-for-1","title":"GlyphControl: Glyph Conditional Control for Visual Text Generation","date":"2023-05-29","arxiv_id":"2305.18259","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/glyphcontrol-glyph-conditional-control-for-1#ran","syntology_url":"https://syntology.ai/paper/2305.18259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18259"}},"official":{"repos":["aigtext/glyphcontrol-release"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/xtreme-up-a-user-centric-scarce-data","slug":"xtreme-up-a-user-centric-scarce-data","title":"XTREME-UP: A User-Centric Scarce-Data Benchmark for Under-Represented Languages","date":"2023-05-19","arxiv_id":"2305.11938","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/xtreme-up-a-user-centric-scarce-data#ran","syntology_url":"https://syntology.ai/paper/2305.11938","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11938"}},"official":{"repos":["google-research/xtreme-up"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/chartreader-a-unified-framework-for-chart","slug":"chartreader-a-unified-framework-for-chart","title":"ChartReader: A Unified Framework for Chart Derendering and Comprehension without Heuristic Rules","date":"2023-04-05","arxiv_id":"2304.02173","repositories_listed":1,"syntology":{"n":19,"n_ran":16,"n_constructed":3,"n_ran_checked":6,"n_instrument":10,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":19,"phrase":"16 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 10 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/chartreader-a-unified-framework-for-chart#ran","syntology_url":"https://syntology.ai/paper/2304.02173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.02173"}},"official":{"repos":["zhiqic/chartreader"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":3,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/glyphdraw-learning-to-draw-chinese-characters","slug":"glyphdraw-learning-to-draw-chinese-characters","title":"GlyphDraw: Seamlessly Rendering Text with Intricate Spatial Structures in Text-to-Image Generation","date":"2023-03-31","arxiv_id":"2303.17870","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/glyphdraw-learning-to-draw-chinese-characters#ran","syntology_url":"https://syntology.ai/paper/2303.17870","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17870"}},"official":{"repos":["OPPO-Mente-Lab/GlyphDraw"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/meta-album-multi-domain-meta-dataset-for-few-1","slug":"meta-album-multi-domain-meta-dataset-for-few-1","title":"Meta-Album: Multi-domain Meta-Dataset for Few-Shot Image Classification","date":"2023-02-16","arxiv_id":"2302.08909","repositories_listed":3,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-album-multi-domain-meta-dataset-for-few-1#ran","syntology_url":"https://syntology.ai/paper/2302.08909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.08909"}},"official":{"repos":["dustincarrion/cd-metadl","ihsaan-ullah/meta-album","ihsanullah2131/meta-album"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pix2struct-screenshot-parsing-as-pretraining","slug":"pix2struct-screenshot-parsing-as-pretraining","title":"Pix2Struct: Screenshot Parsing as Pretraining for Visual Language Understanding","date":"2022-10-07","arxiv_id":"2210.03347","repositories_listed":4,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pix2struct-screenshot-parsing-as-pretraining#ran","syntology_url":"https://syntology.ai/paper/2210.03347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03347"}},"official":{"repos":["google-research/pix2struct"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-counting-meets-hmer-counting-aware","slug":"when-counting-meets-hmer-counting-aware","title":"When Counting Meets HMER: Counting-Aware Network for Handwritten Mathematical Expression Recognition","date":"2022-07-23","arxiv_id":"2207.11463","repositories_listed":3,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/when-counting-meets-hmer-counting-aware#ran","syntology_url":"https://syntology.ai/paper/2207.11463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.11463"}},"official":{"repos":["lbh1024/can"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/git-a-generative-image-to-text-transformer","slug":"git-a-generative-image-to-text-transformer","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","date":"2022-05-27","arxiv_id":"2205.14100","repositories_listed":1,"syntology":{"n":21,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/git-a-generative-image-to-text-transformer#ran","syntology_url":"https://syntology.ai/paper/2205.14100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14100"}},"official":{"repos":["microsoft/GenerativeImage2Text"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/document-dewarping-with-control-points","slug":"document-dewarping-with-control-points","title":"Document Dewarping with Control Points","date":"2022-03-20","arxiv_id":"2203.10543","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/document-dewarping-with-control-points#ran","syntology_url":"https://syntology.ai/paper/2203.10543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.10543"}},"official":{"repos":["gwxie/document-dewarping-with-control-points"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dit-self-supervised-pre-training-for-document","slug":"dit-self-supervised-pre-training-for-document","title":"DiT: Self-supervised Pre-training for Document Image Transformer","date":"2022-03-04","arxiv_id":"2203.02378","repositories_listed":4,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dit-self-supervised-pre-training-for-document#ran","syntology_url":"https://syntology.ai/paper/2203.02378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.02378"}},"official":{"repos":["microsoft/unilm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/latr-layout-aware-transformer-for-scene-text","slug":"latr-layout-aware-transformer-for-scene-text","title":"LaTr: Layout-Aware Transformer for Scene-Text VQA","date":"2021-12-23","arxiv_id":"2112.12494","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/latr-layout-aware-transformer-for-scene-text#ran","syntology_url":"https://syntology.ai/paper/2112.12494","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.12494"}},"official":null}},{"url":"/paper/donut-document-understanding-transformer","slug":"donut-document-understanding-transformer","title":"OCR-free Document Understanding Transformer","date":"2021-11-30","arxiv_id":"2111.15664","repositories_listed":5,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/donut-document-understanding-transformer#ran","syntology_url":"https://syntology.ai/paper/2111.15664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.15664"}},"official":{"repos":["clovaai/donut"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/wenetspeech-a-10000-hours-multi-domain","slug":"wenetspeech-a-10000-hours-multi-domain","title":"WenetSpeech: A 10000+ Hours Multi-domain Mandarin Corpus for Speech Recognition","date":"2021-10-07","arxiv_id":"2110.03370","repositories_listed":2,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wenetspeech-a-10000-hours-multi-domain#ran","syntology_url":"https://syntology.ai/paper/2110.03370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.03370"}},"official":{"repos":["wenet-e2e/wenetspeech"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/trocr-transformer-based-optical-character","slug":"trocr-transformer-based-optical-character","title":"TrOCR: Transformer-based Optical Character Recognition with Pre-trained Models","date":"2021-09-21","arxiv_id":"2109.10282","repositories_listed":8,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/trocr-transformer-based-optical-character#ran","syntology_url":"https://syntology.ai/paper/2109.10282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.10282"}},"official":{"repos":["microsoft/unilm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/pp-ocrv2-bag-of-tricks-for-ultra-lightweight","slug":"pp-ocrv2-bag-of-tricks-for-ultra-lightweight","title":"PP-OCRv2: Bag of Tricks for Ultra Lightweight OCR System","date":"2021-09-07","arxiv_id":"2109.03144","repositories_listed":3,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pp-ocrv2-bag-of-tricks-for-ultra-lightweight#ran","syntology_url":"https://syntology.ai/paper/2109.03144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.03144"}},"official":{"repos":["PaddlePaddle/PaddleOCR"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/bros-a-layout-aware-pre-trained-language","slug":"bros-a-layout-aware-pre-trained-language","title":"BROS: A Pre-trained Language Model Focusing on Text and Layout for Better Key Information Extraction from Documents","date":"2021-08-10","arxiv_id":"2108.04539","repositories_listed":2,"syntology":{"n":13,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bros-a-layout-aware-pre-trained-language#ran","syntology_url":"https://syntology.ai/paper/2108.04539","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.04539"}},"official":{"repos":["clovaai/bros"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lights-camera-action-a-framework-to-improve","slug":"lights-camera-action-a-framework-to-improve","title":"Lights, Camera, Action! A Framework to Improve NLP Accuracy over OCR documents","date":"2021-08-06","arxiv_id":"2108.02899","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lights-camera-action-a-framework-to-improve#ran","syntology_url":"https://syntology.ai/paper/2108.02899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.02899"}},"official":{"repos":["microsoft/genalog"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/implicit-feature-alignment-learn-to-convert","slug":"implicit-feature-alignment-learn-to-convert","title":"Implicit Feature Alignment: Learn to Convert Text Recognizer to Text Spotter","date":"2021-06-10","arxiv_id":"2106.05920","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/implicit-feature-alignment-learn-to-convert#ran","syntology_url":"https://syntology.ai/paper/2106.05920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05920"}},"official":{"repos":["Wang-Tianwei/Implicit-feature-alignment"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/video-aided-unsupervised-grammar-induction","slug":"video-aided-unsupervised-grammar-induction","title":"Video-aided Unsupervised Grammar Induction","date":"2021-04-09","arxiv_id":"2104.04369","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/video-aided-unsupervised-grammar-induction#ran","syntology_url":"https://syntology.ai/paper/2104.04369","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.04369"}},"official":{"repos":["Sy-Zhang/MMC-PCFG"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/pp-ocr-a-practical-ultra-lightweight-ocr","slug":"pp-ocr-a-practical-ultra-lightweight-ocr","title":"PP-OCR: A Practical Ultra Lightweight OCR System","date":"2020-09-21","arxiv_id":"2009.09941","repositories_listed":10,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/pp-ocr-a-practical-ultra-lightweight-ocr#ran","syntology_url":"https://syntology.ai/paper/2009.09941","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.09941"}},"official":{"repos":["PaddlePaddle/PaddleOCR"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/attack-of-the-tails-yes-you-really-can","slug":"attack-of-the-tails-yes-you-really-can","title":"Attack of the Tails: Yes, You Really Can Backdoor Federated Learning","date":"2020-07-09","arxiv_id":"2007.05084","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/attack-of-the-tails-yes-you-really-can#ran","syntology_url":"https://syntology.ai/paper/2007.05084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.05084"}},"official":{"repos":["ksreenivasan/OOD_Federated_Learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cleval-character-level-evaluation-for-text","slug":"cleval-character-level-evaluation-for-text","title":"CLEval: Character-Level Evaluation for Text Detection and Recognition Tasks","date":"2020-06-11","arxiv_id":"2006.06244","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cleval-character-level-evaluation-for-text#ran","syntology_url":"https://syntology.ai/paper/2006.06244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06244"}},"official":{"repos":["clovaai/CLEval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pick-processing-key-information-extraction","slug":"pick-processing-key-information-extraction","title":"PICK: Processing Key Information Extraction from Documents using Improved Graph Learning-Convolutional Networks","date":"2020-04-16","arxiv_id":"2004.07464","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pick-processing-key-information-extraction#ran","syntology_url":"https://syntology.ai/paper/2004.07464","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07464"}},"official":{"repos":["wenwenyu/PICK-pytorch"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/scrabblegan-semi-supervised-varying-length","slug":"scrabblegan-semi-supervised-varying-length","title":"ScrabbleGAN: Semi-Supervised Varying Length Handwritten Text Generation","date":"2020-03-23","arxiv_id":"2003.10557","repositories_listed":3,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":12,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":3,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scrabblegan-semi-supervised-varying-length#ran","syntology_url":"https://syntology.ai/paper/2003.10557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.10557"}},"official":null}},{"url":"/paper/deep-relational-reasoning-graph-network-for","slug":"deep-relational-reasoning-graph-network-for","title":"Deep Relational Reasoning Graph Network for Arbitrary Shape Text Detection","date":"2020-03-17","arxiv_id":"2003.07493","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-relational-reasoning-graph-network-for#ran","syntology_url":"https://syntology.ai/paper/2003.07493","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.07493"}},"official":{"repos":["GXYM/DRRG"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chemgrapher-optical-graph-recognition-of","slug":"chemgrapher-optical-graph-recognition-of","title":"ChemGrapher: Optical Graph Recognition of Chemical Compounds by Deep Learning","date":"2020-02-23","arxiv_id":"2002.09914","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/chemgrapher-optical-graph-recognition-of#ran","syntology_url":"https://syntology.ai/paper/2002.09914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.09914"}},"official":null}},{"url":"/paper/image-based-table-recognition-data-model-and","slug":"image-based-table-recognition-data-model-and","title":"Image-based table recognition: data, model, and evaluation","date":"2019-11-25","arxiv_id":"1911.10683","repositories_listed":6,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/image-based-table-recognition-data-model-and#ran","syntology_url":"https://syntology.ai/paper/1911.10683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.10683"}},"official":{"repos":["ibm-aur-nlp/PubTabNet"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/real-time-scene-text-detection-with","slug":"real-time-scene-text-detection-with","title":"Real-time Scene Text Detection with Differentiable Binarization","date":"2019-11-20","arxiv_id":"1911.08947","repositories_listed":15,"syntology":{"n":25,"n_ran":20,"n_constructed":0,"n_ran_checked":17,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":1,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/real-time-scene-text-detection-with#ran","syntology_url":"https://syntology.ai/paper/1911.08947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.08947"}},"official":{"repos":["MhLiao/DB"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/document-rectification-and-illumination","slug":"document-rectification-and-illumination","title":"Document Rectification and Illumination Correction using a Patch-based CNN","date":"2019-09-20","arxiv_id":"1909.09470","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/document-rectification-and-illumination#ran","syntology_url":"https://syntology.ai/paper/1909.09470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.09470"}},"official":null}},{"url":"/paper/multimodal-deep-networks-for-text-and-image","slug":"multimodal-deep-networks-for-text-and-image","title":"Multimodal deep networks for text and image-based document classification","date":"2019-07-15","arxiv_id":"1907.06370","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-deep-networks-for-text-and-image#ran","syntology_url":"https://syntology.ai/paper/1907.06370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.06370"}},"official":{"repos":["Quicksign/ocrized-text-dataset"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/show-attend-and-read-a-simple-and-strong","slug":"show-attend-and-read-a-simple-and-strong","title":"Show, Attend and Read: A Simple and Strong Baseline for Irregular Text Recognition","date":"2018-11-02","arxiv_id":"1811.00751","repositories_listed":8,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/show-attend-and-read-a-simple-and-strong#ran","syntology_url":"https://syntology.ai/paper/1811.00751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.00751"}},"official":null}},{"url":"/paper/east-an-efficient-and-accurate-scene-text","slug":"east-an-efficient-and-accurate-scene-text","title":"EAST: An Efficient and Accurate Scene Text Detector","date":"2017-04-11","arxiv_id":"1704.03155","repositories_listed":31,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/east-an-efficient-and-accurate-scene-text#ran","syntology_url":"https://syntology.ai/paper/1704.03155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1704.03155"}},"official":null}},{"url":"/paper/image-to-markup-generation-with-coarse-to","slug":"image-to-markup-generation-with-coarse-to","title":"Image-to-Markup Generation with Coarse-to-Fine Attention","date":"2016-09-16","arxiv_id":"1609.04938","repositories_listed":14,"syntology":{"n":13,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/image-to-markup-generation-with-coarse-to#ran","syntology_url":"https://syntology.ai/paper/1609.04938","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1609.04938"}},"official":{"repos":["harvardnlp/im2markup"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/an-end-to-end-trainable-neural-network-for","slug":"an-end-to-end-trainable-neural-network-for","title":"An End-to-End Trainable Neural Network for Image-based Sequence Recognition and Its Application to Scene Text Recognition","date":"2015-07-21","arxiv_id":"1507.05717","repositories_listed":85,"syntology":{"n":81,"n_ran":66,"n_constructed":0,"n_ran_checked":49,"n_instrument":17,"n_unverified":15,"n_honours":1,"n_violates":1,"n_no_contract":47,"n_pointer_only":17,"phrase":"66 ran (of which 0 constructed an object rather than computing a result; 49 with no instrument failure: 1 honoured, 1 violated, 47 with no contract checked; 17 where Syntology's instrument failed) · 15 unverified","sample_list":"/paper/an-end-to-end-trainable-neural-network-for#ran","syntology_url":"https://syntology.ai/paper/1507.05717","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1507.05717"}},"official":null}}],"record_sha256":"f0d4fec14f567b58bd0f9e4720bcc5646c477f8be225915c1bd9258d95b36bd9","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}