{"url":"/method/trocr","slug":"trocr","name":"TrOCR","full_name":"TrOCR","full_name_withheld":false,"description_markdown":"**TrOCR** is an end-to-end [Transformer](https://paperswithcode.com/methods/category/transformers)-based OCR model for text recognition with pre-trained CV and NLP models. It leverages the [Transformer](https://paperswithcode.com/method/transformer) architecture for both image understanding and wordpiece-level text generation. It first resizes the input text image into $384 × 384$ and then the image is split into a sequence of 16 patches which are used as the input to image Transformers.  Standard Transformer architecture with the [self-attention mechanism](https://paperswithcode.com/method/scaled) is leveraged on both encoder and decoder parts, where wordpiece units are generated as the recognized text from the input image.","description_state":"present","introduced_year":null,"introduced_by":{"title":"TrOCR: Transformer-based Optical Character Recognition with Pre-trained Models","paper":"/paper/trocr-transformer-based-optical-character","first_author":"Minghao Li","n_authors":9,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/trocr-transformer-based-optical-character"},"source":{"url":"https://arxiv.org/abs/2109.10282v5","title":"TrOCR: Transformer-based Optical Character Recognition with Pre-trained Models","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"OCR Models","url":"/methods/category/ocr-models","pwc_aliases":[]}],"n_papers_tagged":11,"archive_num_papers":11,"papers_newest_first":[{"paper":null,"title":"TRIDIS: A Comprehensive Medieval and Early Modern Corpus for HTR and NER","date":"2025-03-25","arxiv_id":"2503.22714","n_code_links":0,"syntology":null},{"paper":"/paper/early-evidence-of-how-llms-outperform","title":"Early evidence of how LLMs outperform traditional systems on OCR/HTR tasks for historical records","date":"2025-01-20","arxiv_id":"2501.11623","n_code_links":1,"syntology":null},{"paper":"/paper/comparative-analysis-of-optical-character","title":"Comparative analysis of optical character recognition methods for Sámi texts from the National Library of Norway","date":"2025-01-13","arxiv_id":"2501.07300","n_code_links":2,"syntology":null},{"paper":null,"title":"Leveraging Deep Learning with Multi-Head Attention for Accurate Extraction of Medicine from Handwritten Prescriptions","date":"2024-12-24","arxiv_id":"2412.18199","n_code_links":0,"syntology":null},{"paper":"/paper/spanish-trocr-leveraging-transfer-learning","title":"Spanish TrOCR: Leveraging Transfer Learning for Language Adaptation","date":"2024-07-09","arxiv_id":"2407.06950","n_code_links":1,"syntology":null},{"paper":null,"title":"OSPC: Detecting Harmful Memes with Large Language Model as a Catalyst","date":"2024-06-14","arxiv_id":"2406.09779","n_code_links":0,"syntology":null},{"paper":"/paper/automatic-transcription-of-handwritten-old","title":"Automatic Transcription of Handwritten Old Occitan Language","date":"2023-12-06","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"title":"Vulnerability Analysis of Transformer-based Optical Character Recognition to Adversarial Attacks","date":"2023-11-28","arxiv_id":"2311.17128","n_code_links":0,"syntology":null},{"paper":null,"title":"Extending TrOCR for Text Localization-Free OCR of Full-Page Scanned Receipt Images","date":"2022-12-11","arxiv_id":"2212.05525","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-based-htr-for-historical","title":"Transformer-based HTR for Historical Documents","date":"2022-03-21","arxiv_id":"2203.11008","n_code_links":1,"syntology":null},{"paper":"/paper/trocr-transformer-based-optical-character","title":"TrOCR: Transformer-based Optical Character Recognition with Pre-trained Models","date":"2021-09-21","arxiv_id":"2109.10282","n_code_links":8,"syntology":{"ran":0,"of":6,"unverified":6,"pointer_only":0}}],"papers_shown":11,"tasks":[{"task":"/task/optical-character-recognition","name":"Optical Character Recognition","papers":7},{"task":"/task/optical-character-recognition","name":"Optical Character Recognition (OCR)","papers":7},{"task":"/task/htr","name":"HTR","papers":4},{"task":"/task/handwritten-text-recognition","name":"Handwritten Text Recognition","papers":3},{"task":"/task/decoder","name":"Decoder","papers":2},{"task":"/task/language-modeling","name":"Language Modeling","papers":2},{"task":"/task/language-modelling","name":"Language Modelling","papers":2},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":2},{"task":"/task/adversarial-attack","name":"Adversarial Attack","papers":1},{"task":"/task/data-augmentation","name":"Data Augmentation","papers":1},{"task":"/task/image-captioning","name":"Image Captioning","papers":1},{"task":"/task/image-generation","name":"Image Generation","papers":1},{"task":"/task/large-language-model","name":"Large Language Model","papers":1},{"task":"/task/cg","name":"NER","papers":1},{"task":"/task/named-entity-recognition-1","name":"Named Entity Recognition","papers":1},{"task":"/task/named-entity-recognition-ner","name":"Named Entity Recognition (NER)","papers":1},{"task":"/task/outlier-detection","name":"Outlier Detection","papers":1},{"task":"/task/scene-text-recognition","name":"Scene Text Recognition","papers":1},{"task":"/task/text-generation","name":"Text Generation","papers":1},{"task":"/task/named-entity-recognition","name":"named-entity-recognition","papers":1}],"tasks_shown":20,"n_tasks":20,"usage_by_year":[{"year":"2021","papers":1},{"year":"2022","papers":2},{"year":"2023","papers":2},{"year":"2024","papers":3},{"year":"2025","papers":3}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/trocr"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}