{"url":"/sota/scene-text-recognition-on-icdar2015","task":{"name":"Scene Text Recognition","url":"/task/scene-text-recognition","note":null},"dataset":{"name":"ICDAR2015","url":null},"category":"Computer Vision","categories":["Computer Vision"],"category_note":null,"description":"See [Scene Text Detection](https://paperswithcode.com/task/scene-text-detection) for leaderboards in this task.","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["Accuracy"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"Accuracy":"higher"}},"counts":{"rows":27,"rows_with_code":25,"rows_with_paper_page":27,"rows_dated":27,"rows_using_additional_data":7},"rows":[{"rank_in_archive_order":1,"model":"DTrOCR 105M","metrics":{"Accuracy":"93.5"},"uses_additional_data":false,"paper_date":"2023-08-30","paper":"/paper/dtrocr-decoder-only-transformer-for-optical","paper_url":"https://arxiv.org/abs/2308.15996v1","paper_title":"DTrOCR: Decoder-only Transformer for Optical Character Recognition","code":"https://github.com/arvindrajan92/DTrOCR","n_code_links":1,"syntology":null},{"rank_in_archive_order":2,"model":"CLIP4STR-L*","metrics":{"Accuracy":"92.6"},"uses_additional_data":true,"paper_date":"2023-12-29","paper":"/paper/an-empirical-study-of-scaling-law-for-ocr","paper_url":"https://arxiv.org/abs/2401.00028v3","paper_title":"An Empirical Study of Scaling Law for OCR","code":"https://github.com/large-ocr-model/large-ocr-model.github.io","n_code_links":1,"syntology":null},{"rank_in_archive_order":3,"model":"CPPD","metrics":{"Accuracy":"91.7"},"uses_additional_data":true,"paper_date":"2023-07-23","paper":"/paper/context-perception-parallel-decoder-for-scene","paper_url":"https://arxiv.org/abs/2307.12270v2","paper_title":"Context Perception Parallel Decoder for Scene Text Recognition","code":"https://github.com/PaddlePaddle/PaddleOCR","n_code_links":2,"syntology":null},{"rank_in_archive_order":4,"model":"CLIP4STR-L (DataComp-1B)","metrics":{"Accuracy":"91.4"},"uses_additional_data":false,"paper_date":"2023-05-23","paper":"/paper/clip4str-a-simple-baseline-for-scene-text-1","paper_url":"https://arxiv.org/abs/2305.14014v4","paper_title":"CLIP4STR: A Simple Baseline for Scene Text Recognition with Pre-trained Vision-Language Model","code":"https://github.com/VamosC/CLIP4STR","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":7,"n_samples":10,"n_pointer_only_licence":0}},{"rank_in_archive_order":5,"model":"MGP-STR","metrics":{"Accuracy":"90.9"},"uses_additional_data":true,"paper_date":"2022-09-08","paper":"/paper/multi-granularity-prediction-for-scene-text","paper_url":"https://arxiv.org/abs/2209.03592v2","paper_title":"Multi-Granularity Prediction for Scene Text Recognition","code":"https://github.com/alibabaresearch/advancedliteratemachinery","n_code_links":3,"syntology":null},{"rank_in_archive_order":6,"model":"CLIP4STR-L","metrics":{"Accuracy":"90.8"},"uses_additional_data":true,"paper_date":"2023-05-23","paper":"/paper/clip4str-a-simple-baseline-for-scene-text-1","paper_url":"https://arxiv.org/abs/2305.14014v4","paper_title":"CLIP4STR: A Simple Baseline for Scene Text Recognition with Pre-trained Vision-Language Model","code":"https://github.com/VamosC/CLIP4STR","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":7,"n_samples":10,"n_pointer_only_licence":0}},{"rank_in_archive_order":7,"model":"CLIP4STR-B","metrics":{"Accuracy":"90.6"},"uses_additional_data":true,"paper_date":"2023-05-23","paper":"/paper/clip4str-a-simple-baseline-for-scene-text-1","paper_url":"https://arxiv.org/abs/2305.14014v4","paper_title":"CLIP4STR: A Simple Baseline for Scene Text Recognition with Pre-trained Vision-Language Model","code":"https://github.com/VamosC/CLIP4STR","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":7,"n_samples":10,"n_pointer_only_licence":0}},{"rank_in_archive_order":8,"model":"PARSeq","metrics":{"Accuracy":"89.6±0.3"},"uses_additional_data":true,"paper_date":"2022-07-14","paper":"/paper/scene-text-recognition-with-permuted","paper_url":"https://arxiv.org/abs/2207.06966v1","paper_title":"Scene Text Recognition with Permuted Autoregressive Sequence Models","code":"https://github.com/topdu/openocr","n_code_links":2,"syntology":{"n_ran":5,"n_unverified":3,"n_samples":8,"n_pointer_only_licence":0}},{"rank_in_archive_order":9,"model":"SIGA_S","metrics":{"Accuracy":"87.6"},"uses_additional_data":false,"paper_date":"2022-03-07","paper":"/paper/a-glyph-driven-topology-enhancement-network","paper_url":"https://arxiv.org/abs/2203.03382v4","paper_title":"Self-supervised Implicit Glyph Attention for Text Recognition","code":"https://github.com/tongkunguan/siga","n_code_links":1,"syntology":null},{"rank_in_archive_order":10,"model":"S-GTR","metrics":{"Accuracy":"87.3"},"uses_additional_data":true,"paper_date":"2021-12-24","paper":"/paper/visual-semantics-allow-for-textual-reasoning-1","paper_url":"https://arxiv.org/abs/2112.12916v1","paper_title":"Visual Semantics Allow for Textual Reasoning Better in Scene Text Recognition","code":"https://github.com/adeline-cs/GTR","n_code_links":1,"syntology":null},{"rank_in_archive_order":11,"model":"MATRN","metrics":{"Accuracy":"86.6"},"uses_additional_data":false,"paper_date":"2021-11-30","paper":"/paper/multi-modal-text-recognition-networks","paper_url":"https://arxiv.org/abs/2111.15263v3","paper_title":"Multi-modal Text Recognition Networks: Interactive Enhancements between Visual and Semantic Features","code":"https://github.com/topdu/openocr","n_code_links":3,"syntology":null},{"rank_in_archive_order":12,"model":"CDistNet (Ours)","metrics":{"Accuracy":"86.25"},"uses_additional_data":false,"paper_date":"2021-11-22","paper":"/paper/cdistnet-perceiving-multi-domain-character","paper_url":"https://arxiv.org/abs/2111.11011v5","paper_title":"CDistNet: Perceiving Multi-Domain Character Distance for Robust Text Recognition","code":"https://github.com/topdu/openocr","n_code_links":3,"syntology":null},{"rank_in_archive_order":13,"model":"DiffusionSTR","metrics":{"Accuracy":"86"},"uses_additional_data":false,"paper_date":"2023-06-29","paper":"/paper/diffusionstr-diffusion-model-for-scene-text","paper_url":"https://arxiv.org/abs/2306.16707v1","paper_title":"DiffusionSTR: Diffusion Model for Scene Text Recognition","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":14,"model":"DPAN","metrics":{"Accuracy":"85.5"},"uses_additional_data":false,"paper_date":"2021-08-01","paper":"/paper/look-back-again-dual-parallel-attention","paper_url":"https://dl.acm.org/doi/10.1145/3460426.3463674","paper_title":"Look Back Again: Dual Parallel Attention Network for Accurate and Robust Scene Text Recognition","code":"https://github.com/siddagra/DPAN-look-back-Again-Dual-Parallel-Attention-Network-for-Accurate-and-Robust-Scene-Text-Recognition","n_code_links":2,"syntology":null},{"rank_in_archive_order":15,"model":"RCEED","metrics":{"Accuracy":"82.2"},"uses_additional_data":false,"paper_date":"2021-06-13","paper":"/paper/representation-and-correlation-enhanced","paper_url":"https://arxiv.org/abs/2106.06960v2","paper_title":"Representation and Correlation Enhanced Encoder-Decoder Framework for Scene Text Recognition","code":"https://github.com/Mona9955/RCEED-ICDAR2021","n_code_links":1,"syntology":null},{"rank_in_archive_order":16,"model":"CSTR","metrics":{"Accuracy":"81.6"},"uses_additional_data":false,"paper_date":"2021-02-22","paper":"/paper/cstr-a-classification-perspective-on-scene","paper_url":"https://arxiv.org/abs/2102.10884v3","paper_title":"Revisiting Classification Perspective on Scene Text Recognition","code":"https://github.com/Media-Smart/vedastr","n_code_links":1,"syntology":null},{"rank_in_archive_order":17,"model":"Yet Another Text Recognizer","metrics":{"Accuracy":"80.2"},"uses_additional_data":false,"paper_date":"2021-07-29","paper":"/paper/why-you-should-try-the-real-data-for-the","paper_url":"https://arxiv.org/abs/2107.13938v1","paper_title":"Why You Should Try the Real Data for the Scene Text Recognition","code":"https://github.com/openvinotoolkit/training_extensions","n_code_links":1,"syntology":null},{"rank_in_archive_order":18,"model":"SEED","metrics":{"Accuracy":"80"},"uses_additional_data":false,"paper_date":"2020-05-22","paper":"/paper/seed-semantics-enhanced-encoder-decoder","paper_url":"https://arxiv.org/abs/2005.10977v1","paper_title":"SEED: Semantics Enhanced Encoder-Decoder Framework for Scene Text Recognition","code":"https://github.com/PaddlePaddle/PaddleOCR","n_code_links":3,"syntology":null},{"rank_in_archive_order":19,"model":"TextScanner","metrics":{"Accuracy":"79.4"},"uses_additional_data":false,"paper_date":"2019-12-28","paper":"/paper/textscanner-reading-characters-in-order-for","paper_url":"https://arxiv.org/abs/1912.12422v2","paper_title":"TextScanner: Reading Characters in Order for Robust Scene Text Recognition","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":20,"model":"SATRN","metrics":{"Accuracy":"79.0"},"uses_additional_data":false,"paper_date":"2019-10-10","paper":"/paper/on-recognizing-texts-of-arbitrary-shapes-with","paper_url":"https://arxiv.org/abs/1910.04396v1","paper_title":"On Recognizing Texts of Arbitrary Shapes with 2D Self-Attention","code":"https://github.com/Media-Smart/vedastr","n_code_links":2,"syntology":null},{"rank_in_archive_order":21,"model":"SAFL","metrics":{"Accuracy":"77.5"},"uses_additional_data":false,"paper_date":"2022-01-01","paper":"/paper/safl-a-self-attention-scene-text-recognizer-1","paper_url":"https://arxiv.org/abs/2201.00132v1","paper_title":"SAFL: A Self-Attention Scene Text Recognizer with Focal Loss","code":"https://github.com/ICMLA-SAFL/SAFL_pytorch","n_code_links":1,"syntology":null},{"rank_in_archive_order":22,"model":"ASTER","metrics":{"Accuracy":"76.1"},"uses_additional_data":false,"paper_date":"2018-06-25","paper":"/paper/aster-an-attentional-scene-text-recognizer","paper_url":"http://122.205.5.5:8071/UpLoadFiles/Papers/ASTER_PAMI18.pdf","paper_title":"ASTER: An Attentional Scene Text Recognizer with Flexible Rectification","code":"https://github.com/bgshih/aster","n_code_links":4,"syntology":null},{"rank_in_archive_order":23,"model":"DAN","metrics":{"Accuracy":"74.5"},"uses_additional_data":false,"paper_date":"2019-12-21","paper":"/paper/decoupled-attention-network-for-text","paper_url":"https://arxiv.org/abs/1912.10205v1","paper_title":"Decoupled Attention Network for Text Recognition","code":"https://github.com/topdu/openocr","n_code_links":5,"syntology":null},{"rank_in_archive_order":24,"model":"AON","metrics":{"Accuracy":"73.0"},"uses_additional_data":false,"paper_date":"2017-11-12","paper":"/paper/aon-towards-arbitrarily-oriented-text","paper_url":"http://arxiv.org/abs/1711.04226v2","paper_title":"AON: Towards Arbitrarily-Oriented Text Recognition","code":"https://github.com/huizhang0110/AON","n_code_links":1,"syntology":null},{"rank_in_archive_order":25,"model":"ViTSTR","metrics":{"Accuracy":"72.6"},"uses_additional_data":false,"paper_date":"2021-05-18","paper":"/paper/vision-transformer-for-fast-and-efficient","paper_url":"https://arxiv.org/abs/2105.08582v1","paper_title":"Vision Transformer for Fast and Efficient Scene Text Recognition","code":"https://github.com/PaddlePaddle/PaddleOCR","n_code_links":3,"syntology":null},{"rank_in_archive_order":26,"model":"Baek et al.","metrics":{"Accuracy":"71.8"},"uses_additional_data":false,"paper_date":"2019-04-03","paper":"/paper/what-is-wrong-with-scene-text-recognition","paper_url":"https://arxiv.org/abs/1904.01906v4","paper_title":"What Is Wrong With Scene Text Recognition Model Comparisons? Dataset and Model Analysis","code":"https://github.com/clovaai/deep-text-recognition-benchmark","n_code_links":13,"syntology":{"n_ran":0,"n_unverified":19,"n_samples":19,"n_pointer_only_licence":0}},{"rank_in_archive_order":27,"model":"SAR","metrics":{"Accuracy":"69.2"},"uses_additional_data":false,"paper_date":"2018-11-02","paper":"/paper/show-attend-and-read-a-simple-and-strong","paper_url":"http://arxiv.org/abs/1811.00751v2","paper_title":"Show, Attend and Read: A Simple and Strong Baseline for Irregular Text Recognition","code":"https://github.com/PaddlePaddle/PaddleOCR","n_code_links":8,"syntology":{"n_ran":0,"n_unverified":8,"n_samples":8,"n_pointer_only_licence":0}}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":6,"rows_with_any_sample_ran":4,"distinct_papers_with_graph_line":4,"distinct_papers_with_any_sample_ran":2,"samples_over_distinct_papers":{"n_ran":8,"n_unverified":37,"n_samples":45,"n_pointer_only_licence":0,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":14,"n_unverified":51,"n_samples":65,"n_pointer_only_licence":0,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}