{"url":"/dataset/coco-text","name":"COCO-Text","full_name":"COCO-Text","description_markdown":"The **COCO-Text** dataset is a dataset for text detection and recognition. It is based on the MS COCO dataset, which contains images of complex everyday scenes. The COCO-Text dataset contains non-text images, legible text images and illegible text images. In total there are 22184 training images and 7026 validation images with at least one instance of legible text.\r\n\r\nSource: [Improving Text Proposals for Scene Images with Fully Convolutional Networks](https://arxiv.org/abs/1702.05089)\r\nImage Source: [https://vision.cornell.edu/se3/coco-text-2/](https://vision.cornell.edu/se3/coco-text-2/)","description_withheld":null,"homepage":"https://bgshih.github.io/cocotext/","introduced_date":"2016-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/coco-text-dataset-and-benchmark-for-text","title":"COCO-Text: Dataset and Benchmark for Text Detection and Recognition in Natural Images","first_author":"Andreas Veit","url":null},"license":{"name":"Creative Commons Attribution 4.0 License","url":"https://bgshih.github.io/cocotext/#:~:Terms%20of%20Use"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Scene Text Recognition","url":"/task/scene-text-recognition","datasets_with_task":"/datasets/task/scene-text-recognition"},{"name":"Scene Text Detection","url":"/task/scene-text-detection","datasets_with_task":"/datasets/task/scene-text-detection"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["COCO-Text"],"data_loaders":[{"repo":"https://github.com/activeloopai/Hub","url":"https://docs.activeloop.ai/datasets/coco-text-dataset","frameworks":["tf","pytorch"]},{"repo":"https://github.com/open-mmlab/mmocr","url":"https://github.com/open-mmlab/mmocr/blob/main/docs/datasets.md","frameworks":["pytorch"]}],"num_papers_in_archive":89,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/scene-text-detection-on-coco-text","task":"Scene Text Detection","dataset_variant":"COCO-Text","rows":6,"metrics":["F-Measure","Precision","Recall"],"first_row_in_archive_order":{"model":"Corner-based Region Proposals","paper":"/paper/detecting-multi-oriented-text-with-corner","metrics":{"F-Measure":"59.1","Precision":"55.5","Recall":"63.3"},"code_links":[{"title":"xhzdeng/crpn","url":"https://github.com/xhzdeng/crpn"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/scene-text-recognition-on-coco-text","task":"Scene Text Recognition","dataset_variant":"COCO-Text","rows":4,"metrics":["1:1 Accuracy"],"first_row_in_archive_order":{"model":"CLIP4STR-L","paper":"/paper/clip4str-a-simple-baseline-for-scene-text-1","metrics":{"1:1 Accuracy":"81.9"},"code_links":[{"title":"VamosC/CLIP4STR","url":"https://github.com/VamosC/CLIP4STR"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/clip4str-a-simple-baseline-for-scene-text-1","title":"CLIP4STR: A Simple Baseline for Scene Text Recognition with Pre-trained Vision-Language Model","date":"2023-05-23","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":10,"samples_ran":6,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-granularity-prediction-for-scene-text","title":"Multi-Granularity Prediction for Scene Text Recognition","date":"2022-09-08","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/scene-text-recognition-with-permuted","title":"Scene Text Recognition with Permuted Autoregressive Sequence Models","date":"2022-07-14","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/detecting-multi-oriented-text-with-corner","title":"Detecting Multi-Oriented Text with Corner-based Region Proposals","date":"2018-04-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/textboxes-a-single-shot-oriented-scene-text","title":"TextBoxes++: A Single-Shot Oriented Scene Text Detector","date":"2018-01-09","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/single-shot-text-detector-with-regional","title":"Single Shot Text Detector with Regional Attention","date":"2017-09-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/wordsup-exploiting-word-annotations-for","title":"WordSup: Exploiting Word Annotations for Character based Text Detection","date":"2017-08-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/east-an-efficient-and-accurate-scene-text","title":"EAST: An Efficient and Accurate Scene Text Detector","date":"2017-04-11","rows_on_this_dataset":1,"code_links":31,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scene-text-detection-via-holistic-multi","title":"Scene Text Detection via Holistic, Multi-Channel Prediction","date":"2016-06-29","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":3,"samples_harvested":24,"samples_ran":16,"samples_unverified":8,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}