{"url":"/task/optical-character-recognition","name":"Optical Character Recognition (OCR)","slug":"optical-character-recognition","description_markdown":"**Optical Character Recognition** or **Optical Character Reader** (OCR) is the electronic or mechanical conversion of images of typed, handwritten or printed text into machine-encoded text, whether from a scanned document, a photo of a document, a scene-photo (for example the text on signs and billboards in a landscape photo, license plates in cars...) or from subtitle text superimposed on an image (for example: from a television broadcast)","categories":[{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Methodology","url":"/area/methodology"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":1243,"papers_with_code":462,"benchmarks":6,"benchmark_tables_in_archive":6,"benchmark_tables_shown":6,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":55,"subtasks":10,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/optical-character-recognition-on-benchmarking","slug":"optical-character-recognition-on-benchmarking","dataset":"Benchmarking Chinese Text Recognition: Datasets, Baselines, and an Empirical Study","dataset_url":null,"rows_in_archive":7,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"DTrOCR","paper_title":"DTrOCR: Decoder-only Transformer for Optical Character Recognition","paper_url":"/paper/dtrocr-decoder-only-transformer-for-optical","paper_date":"2023-08-30","arxiv_id":"2308.15996","code_links":[{"title":"arvindrajan92/DTrOCR","url":"https://github.com/arvindrajan92/DTrOCR"}],"syntology":null}},{"leaderboard":"/sota/optical-character-recognition-ocr-on-videodb","slug":"optical-character-recognition-ocr-on-videodb","dataset":"VideoDB's OCR Benchmark Public Collection","dataset_url":"/dataset/videodb-s-ocr-benchmark-public-collection","rows_in_archive":5,"metrics":["Average Accuracy","Character Error Rate (CER)","Word Error Rate (WER)"],"first_row_in_archive_order":{"model":"GPT-4o","paper_title":"Benchmarking Vision-Language Models on Optical Character Recognition in Dynamic Video Environments","paper_url":"/paper/benchmarking-vision-language-models-on","paper_date":"2025-02-10","arxiv_id":"2502.06445","code_links":[{"title":"video-db/ocr-benchmark","url":"https://github.com/video-db/ocr-benchmark"}],"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}}},{"leaderboard":"/sota/optical-character-recognition-on-fsns-test","slug":"optical-character-recognition-on-fsns-test","dataset":"FSNS - Test","dataset_url":"/dataset/fsns-test","rows_in_archive":3,"metrics":["Sequence error"],"first_row_in_archive_order":{"model":"AttentionOCR_Inception-resnet-v2_Location","paper_title":"Attention-based Extraction of Structured Information from Street View Imagery","paper_url":"/paper/attention-based-extraction-of-structured","paper_date":"2017-04-11","arxiv_id":"1704.03549","code_links":[{"title":"tensorflow/models","url":"https://github.com/tensorflow/models/tree/master/research/attention_ocr"},{"title":"tensorflow/models","url":"https://github.com/tensorflow/models"},{"title":"leandroschelb/attention_ocr","url":"https://github.com/leandroschelb/attention_ocr"}],"syntology":null}},{"leaderboard":"/sota/optical-character-recognition-ocr-on-sut","slug":"optical-character-recognition-ocr-on-sut","dataset":"SUT","dataset_url":"/dataset/sut","rows_in_archive":2,"metrics":["Character Error Rate (CER)"],"first_row_in_archive_order":{"model":"Tesseract","paper_title":"SUT: a new multi-purpose synthetic dataset for Farsi document image analysis","paper_url":"/paper/sut-a-new-multi-purpose-synthetic-dataset-for","paper_date":"2023-11-27","arxiv_id":null,"code_links":[{"title":"aliiafkari/SUT_Dataset","url":"https://github.com/aliiafkari/SUT_Dataset"}],"syntology":null}},{"leaderboard":"/sota/optical-character-recognition-on-i2l-140k","slug":"optical-character-recognition-on-i2l-140k","dataset":"I2L-140K","dataset_url":"/dataset/i2l-140k","rows_in_archive":2,"metrics":["BLEU"],"first_row_in_archive_order":{"model":"I2L-NOPOOL","paper_title":"Teaching Machines to Code: Neural Markup Generation with Visual Attention","paper_url":"/paper/teaching-machines-to-code-neural-markup","paper_date":"2018-02-15","arxiv_id":"1802.05415","code_links":[{"title":"untrix/im2latex","url":"https://github.com/untrix/im2latex"}],"syntology":null}},{"leaderboard":"/sota/optical-character-recognition-on-im2latex-1","slug":"optical-character-recognition-on-im2latex-1","dataset":"im2latex-100k","dataset_url":"/dataset/im2latex-100k","rows_in_archive":1,"metrics":["BLEU"],"first_row_in_archive_order":{"model":"I2L-STRIPS","paper_title":"Teaching Machines to Code: Neural Markup Generation with Visual Attention","paper_url":"/paper/teaching-machines-to-code-neural-markup","paper_date":"2018-02-15","arxiv_id":"1802.05415","code_links":[{"title":"untrix/im2latex","url":"https://github.com/untrix/im2latex"}],"syntology":null}}],"datasets":[{"url":"/dataset/iam","name":"IAM","full_name":"IAM Handwriting","num_papers_in_archive":198},{"url":"/dataset/funsd","name":"FUNSD","full_name":"Form Understanding in Noisy Scanned Documents","num_papers_in_archive":179},{"url":"/dataset/textcaps","name":"TextCaps","full_name":"","num_papers_in_archive":98},{"url":"/dataset/st-vqa","name":"ST-VQA","full_name":"Scene Text Visual Question Answering","num_papers_in_archive":90},{"url":"/dataset/icdar-2003","name":"ICDAR 2003","full_name":"ICDAR 2003","num_papers_in_archive":53},{"url":"/dataset/textocr","name":"TextOCR","full_name":"","num_papers_in_archive":37},{"url":"/dataset/scitsr","name":"SciTSR","full_name":null,"num_papers_in_archive":36},{"url":"/dataset/docbank","name":"DocBank","full_name":"","num_papers_in_archive":34},{"url":"/dataset/105941-images-natural-scenes-ocr-data-of-12","name":"105,941 Images Natural Scenes OCR Data of 12 Languages","full_name":"105,941 Images Natural Scenes OCR Data of 12 Languages","num_papers_in_archive":24},{"url":"/dataset/ufpr-alpr","name":"UFPR-ALPR","full_name":"","num_papers_in_archive":15},{"url":"/dataset/im2latex-100k","name":"im2latex-100k","full_name":"","num_papers_in_archive":12},{"url":"/dataset/mrr-benchmark","name":"MRR-Benchmark","full_name":"Multi-Modal Reading Benchmark","num_papers_in_archive":11},{"url":"/dataset/rodosol-alpr","name":"RodoSol-ALPR","full_name":"","num_papers_in_archive":8},{"url":"/dataset/ssig-segplate","name":"SSIG-SegPlate","full_name":"","num_papers_in_archive":8},{"url":"/dataset/kannada-mnist","name":"Kannada-MNIST","full_name":"","num_papers_in_archive":7},{"url":"/dataset/iiit-ar-13k","name":"IIIT-AR-13K","full_name":"","num_papers_in_archive":6},{"url":"/dataset/midv-2019","name":"MIDV-2019","full_name":"","num_papers_in_archive":6},{"url":"/dataset/mle2e","name":"MLe2e","full_name":"","num_papers_in_archive":6},{"url":"/dataset/chinese-text-in-the-wild","name":"Chinese Text in the Wild","full_name":"","num_papers_in_archive":5},{"url":"/dataset/infinity-mm","name":"Infinity-MM","full_name":"","num_papers_in_archive":5},{"url":"/dataset/twitter100k","name":"Twitter100k","full_name":"","num_papers_in_archive":5},{"url":"/dataset/chineselp","name":"ChineseLP","full_name":"","num_papers_in_archive":4},{"url":"/dataset/ddi-100","name":"DDI-100","full_name":"Distorted Document Images","num_papers_in_archive":4},{"url":"/dataset/banglalekha-isolated","name":"BanglaLekha-Isolated","full_name":"","num_papers_in_archive":3},{"url":"/dataset/fsns-test","name":"FSNS - Test","full_name":"","num_papers_in_archive":3},{"url":"/dataset/high-quality-invoice-images-for-ocr","name":"High-Quality Invoice Images for OCR","full_name":"","num_papers_in_archive":3},{"url":"/dataset/imgur5k","name":"Imgur5K","full_name":"","num_papers_in_archive":3},{"url":"/dataset/newspaper-navigator","name":"Newspaper Navigator","full_name":"","num_papers_in_archive":3},{"url":"/dataset/mcscset","name":"MCSCSet","full_name":"","num_papers_in_archive":2},{"url":"/dataset/molparser-7m","name":"MolParser-7M","full_name":"","num_papers_in_archive":2},{"url":"/dataset/msda","name":"MSDA","full_name":"Multi-source domain adaptation dataset for text recognition","num_papers_in_archive":2},{"url":"/dataset/ufpr-amr-dataset","name":"UFPR-AMR","full_name":"","num_papers_in_archive":2},{"url":"/dataset/aceparse","name":"AceParse","full_name":"","num_papers_in_archive":1},{"url":"/dataset/an-extensive-dataset-of-handwritten-central","name":"An extensive dataset of handwritten central Kurdish isolated characters","full_name":"Rebin M. Ahmed","num_papers_in_archive":1},{"url":"/dataset/arabic-img2md","name":"arabic-img2md","full_name":"Arabic Img2MD","num_papers_in_archive":1},{"url":"/dataset/bln600","name":"BLN600","full_name":"BLN600: A Parallel Corpus of Machine/Human Transcribed Nineteenth Century Newspaper Texts","num_papers_in_archive":1},{"url":"/dataset/copel-amr-dataset","name":"Copel-AMR","full_name":"","num_papers_in_archive":1},{"url":"/dataset/doc3dshade","name":"Doc3DShade","full_name":"","num_papers_in_archive":1},{"url":"/dataset/fics-pcb-image-collection-fpic","name":"FICS PCB Image Collection (FPIC)","full_name":"","num_papers_in_archive":1},{"url":"/dataset/i2l-140k","name":"I2L-140K","full_name":"","num_papers_in_archive":1},{"url":"/dataset/illusionchar-test","name":"IllusionChar_test","full_name":"","num_papers_in_archive":1},{"url":"/dataset/large-labelled-logo-dataset-l3d","name":"Large Labelled Logo Dataset (L3D)","full_name":"","num_papers_in_archive":1},{"url":"/dataset/matrivasha","name":"MatriVasha:","full_name":"MatriVasha: Compound Character atasetD","num_papers_in_archive":1},{"url":"/dataset/ncse-v2-0","name":"NCSE v2.0","full_name":"NCSE v2.0: A Dataset of OCR-Processed 19th Century English Newspapers","num_papers_in_archive":1},{"url":"/dataset/psocr","name":"PsOCR","full_name":"Pashto OCR Dataset","num_papers_in_archive":1},{"url":"/dataset/sut","name":"SUT","full_name":"SUT: a new multi-purpose synthetic dataset for Farsi document image analysis","num_papers_in_archive":1},{"url":"/dataset/utrset-real","name":"UTRSet-Real","full_name":"","num_papers_in_archive":1},{"url":"/dataset/utrset-synth","name":"UTRSet-Synth","full_name":"","num_papers_in_archive":1},{"url":"/dataset/videodb-s-ocr-benchmark-public-collection","name":"VideoDB's OCR Benchmark Public Collection","full_name":"VideoDB's OCR Benchmark Public Collection","num_papers_in_archive":1},{"url":"/dataset/webli","name":"WebLI","full_name":"Web Language Image","num_papers_in_archive":1},{"url":"/dataset/codescan","name":"CodeSCAN","full_name":"ScreenCast ANalysis for Video Programming Tutorials","num_papers_in_archive":0},{"url":"/dataset/hindi-text-image-dataset","name":"Hindi Text Image Dataset | Hindi in the wild","full_name":"datacluster.ai","num_papers_in_archive":0},{"url":"/dataset/indian-signboard-image-dataset-text-in-image","name":"Indian Signboard Image Dataset | Text in Image","full_name":"datacluster.ai","num_papers_in_archive":0},{"url":"/dataset/ts-tr","name":"TS-TR","full_name":"Turkish Scene Text Recognition Dataset","num_papers_in_archive":0},{"url":"/dataset/visiting-card-id-card-images-hindi-english","name":"Visiting Card | ID Card Images | Hindi-English","full_name":"datacluster.ai","num_papers_in_archive":0}],"subtasks":[{"url":"/task/active-learning","name":"Active Learning"},{"url":"/task/grapheme-detection","name":"Grapheme Detection"},{"url":"/task/handwriting-recognition","name":"Handwriting Recognition"},{"url":"/task/handwritten-chinese-text-recognition","name":"Handwritten Chinese Text Recognition"},{"url":"/task/handwritten-digit-image-synthesis","name":"Handwritten Digit Image Synthesis"},{"url":"/task/handwritten-digit-recognition","name":"Handwritten Digit Recognition"},{"url":"/task/handwritten-text-recognition","name":"Handwritten Text Recognition"},{"url":"/task/irregular-text-recognition","name":"Irregular Text Recognition"},{"url":"/task/offline-handwritten-chinese-character","name":"Offline Handwritten Chinese Character Recognition"},{"url":"/task/word-spotting-in-handwritten-documents","name":"Word Spotting In Handwritten Documents"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":462,"tagged_in_all":1243,"items":[{"url":"/paper/an-end-to-end-trainable-neural-network-for","title":"An End-to-End Trainable Neural Network for Image-based Sequence Recognition and Its Application to Scene Text Recognition","date":"2015-07-21","arxiv_id":"1507.05717","repositories_listed":85,"syntology":{"n":81,"n_ran":19,"n_unverified":62,"n_pointer_only":17}},{"url":"/paper/east-an-efficient-and-accurate-scene-text","title":"EAST: An Efficient and Accurate Scene Text Detector","date":"2017-04-11","arxiv_id":"1704.03155","repositories_listed":31,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/shape-robust-text-detection-with-progressive-1","title":"Shape Robust Text Detection with Progressive Scale Expansion Network","date":"2019-03-28","arxiv_id":"1903.12473","repositories_listed":19,"syntology":null},{"url":"/paper/real-time-scene-text-detection-with","title":"Real-time Scene Text Detection with Differentiable Binarization","date":"2019-11-20","arxiv_id":"1911.08947","repositories_listed":15,"syntology":{"n":25,"n_ran":3,"n_unverified":22,"n_pointer_only":0}},{"url":"/paper/image-to-markup-generation-with-coarse-to","title":"Image-to-Markup Generation with Coarse-to-Fine Attention","date":"2016-09-16","arxiv_id":"1609.04938","repositories_listed":14,"syntology":{"n":13,"n_ran":1,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/pp-ocr-a-practical-ultra-lightweight-ocr","title":"PP-OCR: A Practical Ultra Lightweight OCR System","date":"2020-09-21","arxiv_id":"2009.09941","repositories_listed":10,"syntology":{"n":12,"n_ran":0,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/pp-ocr-a-practical-ultra-lightweight-ocr","title":"PP-OCR: A Practical Ultra Lightweight OCR System","date":"2020-09-21","arxiv_id":"2009.09941","repositories_listed":10,"syntology":{"n":12,"n_ran":0,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/trocr-transformer-based-optical-character","title":"TrOCR: Transformer-based Optical Character Recognition with Pre-trained Models","date":"2021-09-21","arxiv_id":"2109.10282","repositories_listed":8,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/trocr-transformer-based-optical-character","title":"TrOCR: Transformer-based Optical Character Recognition with Pre-trained Models","date":"2021-09-21","arxiv_id":"2109.10282","repositories_listed":8,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/show-attend-and-read-a-simple-and-strong","title":"Show, Attend and Read: A Simple and Strong Baseline for Irregular Text Recognition","date":"2018-11-02","arxiv_id":"1811.00751","repositories_listed":8,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/a-multi-object-rectified-attention-network","title":"A Multi-Object Rectified Attention Network for Scene Text Recognition","date":"2019-01-10","arxiv_id":"1901.03003","repositories_listed":7,"syntology":null},{"url":"/paper/image-based-table-recognition-data-model-and","title":"Image-based table recognition: data, model, and evaluation","date":"2019-11-25","arxiv_id":"1911.10683","repositories_listed":6,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/donut-document-understanding-transformer","title":"OCR-free Document Understanding Transformer","date":"2021-11-30","arxiv_id":"2111.15664","repositories_listed":5,"syntology":{"n":9,"n_ran":0,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/donut-document-understanding-transformer","title":"OCR-free Document Understanding Transformer","date":"2021-11-30","arxiv_id":"2111.15664","repositories_listed":5,"syntology":{"n":9,"n_ran":0,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/exploring-cross-image-pixel-contrast-for","title":"Exploring Cross-Image Pixel Contrast for Semantic Segmentation","date":"2021-01-28","arxiv_id":"2101.11939","repositories_listed":5,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/chinese-text-in-the-wild","title":"Chinese Text in the Wild","date":"2018-02-28","arxiv_id":"1803.00085","repositories_listed":5,"syntology":null},{"url":"/paper/robust-scene-text-recognition-with-automatic","title":"Robust Scene Text Recognition with Automatic Rectification","date":"2016-03-12","arxiv_id":"1603.03915","repositories_listed":5,"syntology":null},{"url":"/paper/pix2struct-screenshot-parsing-as-pretraining","title":"Pix2Struct: Screenshot Parsing as Pretraining for Visual Language Understanding","date":"2022-10-07","arxiv_id":"2210.03347","repositories_listed":4,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/dit-self-supervised-pre-training-for-document","title":"DiT: Self-supervised Pre-training for Document Image Transformer","date":"2022-03-04","arxiv_id":"2203.02378","repositories_listed":4,"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/de-gan-a-conditional-generative-adversarial-1","title":"DE-GAN: A Conditional Generative Adversarial Network for Document Enhancement","date":"2020-10-17","arxiv_id":"2010.08764","repositories_listed":4,"syntology":null},{"url":"/paper/aster-an-attentional-scene-text-recognizer","title":"ASTER: An Attentional Scene Text Recognizer with Flexible Rectification","date":"2018-06-25","arxiv_id":null,"repositories_listed":4,"syntology":null},{"url":"/paper/aster-an-attentional-scene-text-recognizer","title":"ASTER: An Attentional Scene Text Recognizer with Flexible Rectification","date":"2018-06-25","arxiv_id":null,"repositories_listed":4,"syntology":null},{"url":"/paper/nrtr-a-no-recurrence-sequence-to-sequence","title":"NRTR: A No-Recurrence Sequence-to-Sequence Model For Scene Text Recognition","date":"2018-06-04","arxiv_id":"1806.00926","repositories_listed":4,"syntology":null},{"url":"/paper/end-to-end-interpretation-of-the-french","title":"End-to-End Interpretation of the French Street Name Signs Dataset","date":"2017-02-13","arxiv_id":"1702.03970","repositories_listed":4,"syntology":null},{"url":"/paper/coco-text-dataset-and-benchmark-for-text","title":"COCO-Text: Dataset and Benchmark for Text Detection and Recognition in Natural Images","date":"2016-01-26","arxiv_id":"1601.07140","repositories_listed":4,"syntology":null},{"url":"/paper/coco-text-dataset-and-benchmark-for-text","title":"COCO-Text: Dataset and Benchmark for Text Detection and Recognition in Natural Images","date":"2016-01-26","arxiv_id":"1601.07140","repositories_listed":4,"syntology":null},{"url":"/paper/scrambled-text-training-language-models-to","title":"Scrambled text: training Language Models to correct OCR errors using synthetic data","date":"2024-09-29","arxiv_id":"2409.19735","repositories_listed":3,"syntology":null},{"url":"/paper/swift-a-scalable-lightweight-infrastructure","title":"SWIFT:A Scalable lightWeight Infrastructure for Fine-Tuning","date":"2024-08-10","arxiv_id":"2408.05517","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/swift-a-scalable-lightweight-infrastructure","title":"SWIFT:A Scalable lightWeight Infrastructure for Fine-Tuning","date":"2024-08-10","arxiv_id":"2408.05517","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/docreal-robust-document-dewarping-of-real","title":"DocReal: Robust Document Dewarping of Real-Life Images via Attention-Enhanced Control Point Prediction","date":"2023-12-01","arxiv_id":null,"repositories_listed":3,"syntology":null}],"syntology_records":17,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}