{"url":"/dataset/icdar-2003","name":"ICDAR 2003","full_name":"ICDAR 2003","description_markdown":"The ICDAR2003 dataset is a dataset for scene text recognition. It contains 507 natural scene images (including 258 training images and 249 test images) in total. The images are annotated at character level. Characters and words can be cropped from the images.\r\n\r\nSource: [Robust Scene Text Recognition Using Sparse Coding based Features](https://arxiv.org/abs/1512.08669)\r\nImage Source: [https://www.researchgate.net/figure/The-results-of-text-localization-and-extraction-on-ICDAR-2003-dataset_fig3_290070044](https://www.researchgate.net/figure/The-results-of-text-localization-and-extraction-on-ICDAR-2003-dataset_fig3_290070044)","description_withheld":null,"homepage":"http://www.imglab.org/db/index.html","introduced_date":"2003-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"ICDAR 2003 Robust Reading Competitions","first_author":null,"url":"https://doi.org/10.1109/ICDAR.2003.1227749"},"license":{"name":"Custom (research-only)","url":"http://www.imglab.org/db/index.html"},"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Optical Character Recognition (OCR)","url":"/task/optical-character-recognition","datasets_with_task":"/datasets/task/optical-character-recognition"},{"name":"Scene Text Recognition","url":"/task/scene-text-recognition","datasets_with_task":"/datasets/task/scene-text-recognition"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["ICDAR 2003"],"data_loaders":[{"repo":"https://github.com/mindee/doctr","url":"https://mindee.github.io/doctr/latest/datasets.html#doctr.datasets.IC03","frameworks":["tf","pytorch"]}],"num_papers_in_archive":53,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/scene-text-recognition-on-icdar-2003","task":"Scene Text Recognition","dataset_variant":"ICDAR 2003","rows":12,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Yet Another Text Recognizer","paper":"/paper/why-you-should-try-the-real-data-for-the","metrics":{"Accuracy":"97.1"},"code_links":[{"title":"openvinotoolkit/training_extensions","url":"https://github.com/openvinotoolkit/training_extensions"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/a-glyph-driven-topology-enhancement-network","title":"Self-supervised Implicit Glyph Attention for Text Recognition","date":"2022-03-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/safl-a-self-attention-scene-text-recognizer-1","title":"SAFL: A Self-Attention Scene Text Recognizer with Focal Loss","date":"2022-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/why-you-should-try-the-real-data-for-the","title":"Why You Should Try the Real Data for the Scene Text Recognition","date":"2021-07-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vision-transformer-for-fast-and-efficient","title":"Vision Transformer for Fast and Efficient Scene Text Recognition","date":"2021-05-18","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/cstr-a-classification-perspective-on-scene","title":"Revisiting Classification Perspective on Scene Text Recognition","date":"2021-02-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/decoupled-attention-network-for-text","title":"Decoupled Attention Network for Text Recognition","date":"2019-12-21","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/on-recognizing-texts-of-arbitrary-shapes-with","title":"On Recognizing Texts of Arbitrary Shapes with 2D Self-Attention","date":"2019-10-10","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/what-is-wrong-with-scene-text-recognition","title":"What Is Wrong With Scene Text Recognition Model Comparisons? Dataset and Model Analysis","date":"2019-04-03","rows_on_this_dataset":1,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":0,"samples_unverified":19,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/aon-towards-arbitrarily-oriented-text","title":"AON: Towards Arbitrarily-Oriented Text Recognition","date":"2017-11-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/star-net-a-spatial-attention-residue-network","title":"Star-net: A spatial attention residue network for scene text recognition.","date":"2016-09-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/robust-scene-text-recognition-with-automatic","title":"Robust Scene Text Recognition with Automatic Rectification","date":"2016-03-12","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/an-end-to-end-trainable-neural-network-for","title":"An End-to-End Trainable Neural Network for Image-based Sequence Recognition and Its Application to Scene Text Recognition","date":"2015-07-21","rows_on_this_dataset":1,"code_links":85,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":81,"samples_ran":19,"samples_unverified":62,"pointer_only_for_licence":17,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":100,"samples_ran":19,"samples_unverified":81,"pointer_only_for_licence":17,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}