{"url":"/task/scene-text-detection","name":"Scene Text Detection","slug":"scene-text-detection","description_markdown":"**Scene Text Detection** is a computer vision task that involves automatically identifying and localizing text within natural images or videos. The goal of scene text detection is to develop algorithms that can robustly detect and and label text with bounding boxes in uncontrolled and complex environments, such as street signs, billboards, or license plates.\r\n\r\n\r\n<span class=\"description-source\">Source: [ContourNet: Taking a Further Step toward Accurate Arbitrary-shaped Scene Text Detection ](https://arxiv.org/abs/2004.04940)</span>","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":213,"papers_with_code":98,"benchmarks":9,"benchmark_tables_in_archive":9,"benchmark_tables_shown":9,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":17,"subtasks":2,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/scene-text-detection-on-icdar-2015","slug":"scene-text-detection-on-icdar-2015","dataset":"ICDAR 2015","dataset_url":"/dataset/icdar-2015","rows_in_archive":43,"metrics":["F-Measure","Precision","Recall","Accuracy","FPS"],"first_row_in_archive_order":{"model":"TextFuseNet (ResNeXt-101)","paper_title":"TextFuseNet: Scene Text Detection with Richer Fused Features","paper_url":"/paper/textfusenet-scene-text-detection-with-richer","paper_date":"2020-05-17","arxiv_id":null,"code_links":[{"title":"ying09/TextFuseNet","url":"https://github.com/ying09/TextFuseNet"},{"title":"mindspore-ai/models","url":"https://github.com/mindspore-ai/models/tree/master/research/cv/textfusenet"},{"title":"kingcong/textfusenet","url":"https://github.com/kingcong/textfusenet"},{"title":"2023-MindSpore-1/ms-code-217","url":"https://github.com/2023-MindSpore-1/ms-code-217/tree/main/textfusenet"},{"title":"2023-MindSpore-1/ms-code-7","url":"https://github.com/2023-MindSpore-1/ms-code-7/tree/main/textfusenet"},{"title":"MindSpore-paper-code-2/code3","url":"https://github.com/MindSpore-paper-code-2/code3/tree/main/textfusenet"}],"syntology":null}},{"leaderboard":"/sota/scene-text-detection-on-total-text","slug":"scene-text-detection-on-total-text","dataset":"Total-Text","dataset_url":"/dataset/total-text","rows_in_archive":27,"metrics":["F-Measure","Precision","Recall","FPS"],"first_row_in_archive_order":{"model":"MixNet","paper_title":"MixNet: Toward Accurate Detection of Challenging Scene Text in the Wild","paper_url":"/paper/mixnet-toward-accurate-detection-of","paper_date":"2023-08-23","arxiv_id":"2308.12817","code_links":[{"title":"D641593/MixNet","url":"https://github.com/D641593/MixNet"}],"syntology":null}},{"leaderboard":"/sota/scene-text-detection-on-msra-td500","slug":"scene-text-detection-on-msra-td500","dataset":"MSRA-TD500","dataset_url":"/dataset/msra-td500","rows_in_archive":18,"metrics":["F-Measure","Precision","Recall","FPS"],"first_row_in_archive_order":{"model":"MixNet","paper_title":"MixNet: Toward Accurate Detection of Challenging Scene Text in the Wild","paper_url":"/paper/mixnet-toward-accurate-detection-of","paper_date":"2023-08-23","arxiv_id":"2308.12817","code_links":[{"title":"D641593/MixNet","url":"https://github.com/D641593/MixNet"}],"syntology":null}},{"leaderboard":"/sota/scene-text-detection-on-scut-ctw1500","slug":"scene-text-detection-on-scut-ctw1500","dataset":"SCUT-CTW1500","dataset_url":"/dataset/scut-ctw1500","rows_in_archive":17,"metrics":["F-Measure","Precision","Recall","FPS"],"first_row_in_archive_order":{"model":"MixNet","paper_title":"MixNet: Toward Accurate Detection of Challenging Scene Text in the Wild","paper_url":"/paper/mixnet-toward-accurate-detection-of","paper_date":"2023-08-23","arxiv_id":"2308.12817","code_links":[{"title":"D641593/MixNet","url":"https://github.com/D641593/MixNet"}],"syntology":null}},{"leaderboard":"/sota/scene-text-detection-on-icdar-2013","slug":"scene-text-detection-on-icdar-2013","dataset":"ICDAR 2013","dataset_url":"/dataset/icdar-2013","rows_in_archive":16,"metrics":["F-Measure","Precision","Recall","H-Mean"],"first_row_in_archive_order":{"model":"TextFuseNet (ResNeXt-101)","paper_title":"TextFuseNet: Scene Text Detection with Richer Fused Features","paper_url":"/paper/textfusenet-scene-text-detection-with-richer","paper_date":"2020-05-17","arxiv_id":null,"code_links":[{"title":"ying09/TextFuseNet","url":"https://github.com/ying09/TextFuseNet"},{"title":"mindspore-ai/models","url":"https://github.com/mindspore-ai/models/tree/master/research/cv/textfusenet"},{"title":"kingcong/textfusenet","url":"https://github.com/kingcong/textfusenet"},{"title":"2023-MindSpore-1/ms-code-217","url":"https://github.com/2023-MindSpore-1/ms-code-217/tree/main/textfusenet"},{"title":"2023-MindSpore-1/ms-code-7","url":"https://github.com/2023-MindSpore-1/ms-code-7/tree/main/textfusenet"},{"title":"MindSpore-paper-code-2/code3","url":"https://github.com/MindSpore-paper-code-2/code3/tree/main/textfusenet"}],"syntology":null}},{"leaderboard":"/sota/scene-text-detection-on-icdar-2017-mlt-1","slug":"scene-text-detection-on-icdar-2017-mlt-1","dataset":"ICDAR 2017 MLT","dataset_url":"/dataset/icdar-2017","rows_in_archive":14,"metrics":["Precision","Recall","F-Measure","H-Mean"],"first_row_in_archive_order":{"model":"PMTD*","paper_title":"Pyramid Mask Text Detector","paper_url":"/paper/pyramid-mask-text-detector","paper_date":"2019-03-28","arxiv_id":"1903.11800","code_links":[{"title":"jjprincess/PMTD","url":"https://github.com/jjprincess/PMTD"},{"title":"anhnguyen9a7/pyramid-mask-and-plane-clustering-visualization","url":"https://github.com/anhnguyen9a7/pyramid-mask-and-plane-clustering-visualization"}],"syntology":null}},{"leaderboard":"/sota/scene-text-detection-on-coco-text","slug":"scene-text-detection-on-coco-text","dataset":"COCO-Text","dataset_url":"/dataset/coco-text","rows_in_archive":6,"metrics":["F-Measure","Precision","Recall"],"first_row_in_archive_order":{"model":"Corner-based Region Proposals","paper_title":"Detecting Multi-Oriented Text with Corner-based Region Proposals","paper_url":"/paper/detecting-multi-oriented-text-with-corner","paper_date":"2018-04-08","arxiv_id":"1804.02690","code_links":[{"title":"xhzdeng/crpn","url":"https://github.com/xhzdeng/crpn"}],"syntology":null}},{"leaderboard":"/sota/scene-text-detection-on-ic19-art","slug":"scene-text-detection-on-ic19-art","dataset":"IC19-Art","dataset_url":null,"rows_in_archive":4,"metrics":["H-Mean"],"first_row_in_archive_order":{"model":"MixNet","paper_title":"MixNet: Toward Accurate Detection of Challenging Scene Text in the Wild","paper_url":"/paper/mixnet-toward-accurate-detection-of","paper_date":"2023-08-23","arxiv_id":"2308.12817","code_links":[{"title":"D641593/MixNet","url":"https://github.com/D641593/MixNet"}],"syntology":null}},{"leaderboard":"/sota/scene-text-detection-on-ic19-rects","slug":"scene-text-detection-on-ic19-rects","dataset":"IC19-ReCTs","dataset_url":null,"rows_in_archive":1,"metrics":["F-Measure"],"first_row_in_archive_order":{"model":"BDN","paper_title":"Omnidirectional Scene Text Detection with Sequential-free Box Discretization","paper_url":"/paper/omnidirectional-scene-text-detection-with","paper_date":"2019-06-06","arxiv_id":"1906.02371","code_links":[{"title":"Yuliang-Liu/Box_Discretization_Network","url":"https://github.com/Yuliang-Liu/Box_Discretization_Network"}],"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":5}}}],"datasets":[{"url":"/dataset/icdar-2013","name":"ICDAR 2013","full_name":"","num_papers_in_archive":246},{"url":"/dataset/total-text","name":"Total-Text","full_name":"","num_papers_in_archive":156},{"url":"/dataset/msra-td500","name":"MSRA-TD500","full_name":"MSRA Text Detection 500 Database","num_papers_in_archive":124},{"url":"/dataset/coco-text","name":"COCO-Text","full_name":"COCO-Text","num_papers_in_archive":89},{"url":"/dataset/icdar-2015","name":"ICDAR 2015","full_name":"","num_papers_in_archive":52},{"url":"/dataset/scut-ctw1500","name":"SCUT-CTW1500","full_name":"","num_papers_in_archive":44},{"url":"/dataset/textocr","name":"TextOCR","full_name":"","num_papers_in_archive":37},{"url":"/dataset/rctw-17","name":"RCTW-17","full_name":"Reading Chinese Text in the Wild","num_papers_in_archive":20},{"url":"/dataset/icdar-2017","name":"ICDAR 2017","full_name":"ICDAR 2017","num_papers_in_archive":18},{"url":"/dataset/chinese-text-in-the-wild","name":"Chinese Text in the Wild","full_name":"","num_papers_in_archive":5},{"url":"/dataset/pku-license-plate-detection","name":"PKU (License Plate Detection)","full_name":"","num_papers_in_archive":2},{"url":"/dataset/shopsign","name":"ShopSign","full_name":"","num_papers_in_archive":2},{"url":"/dataset/urdudoc","name":"UrduDoc","full_name":"","num_papers_in_archive":1},{"url":"/dataset/cntd","name":"CNTD","full_name":"Chinese and Naxi text detection","num_papers_in_archive":0},{"url":"/dataset/indian-number-plates-dataset-vehicle-number","name":"Indian Number Plates Dataset | Vehicle Number Plates | English OCR Detection","full_name":"","num_papers_in_archive":0},{"url":"/dataset/media-text","name":"Media-Text","full_name":"MediaText: a media industry-based dataset for scene text detetcion","num_papers_in_archive":0},{"url":"/dataset/thanh","name":"SignboardText","full_name":"","num_papers_in_archive":0}],"subtasks":[{"url":"/task/curved-text-detection","name":"Curved Text Detection"},{"url":"/task/multi-oriented-scene-text-detection","name":"Multi-Oriented Scene Text Detection"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":98,"tagged_in_all":213,"items":[{"url":"/paper/east-an-efficient-and-accurate-scene-text","title":"EAST: An Efficient and Accurate Scene Text Detector","date":"2017-04-11","arxiv_id":"1704.03155","repositories_listed":31,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/detecting-text-in-natural-image-with","title":"Detecting Text in Natural Image with Connectionist Text Proposal Network","date":"2016-09-12","arxiv_id":"1609.03605","repositories_listed":27,"syntology":{"n":35,"n_ran":3,"n_unverified":32,"n_pointer_only":1}},{"url":"/paper/shape-robust-text-detection-with-progressive-1","title":"Shape Robust Text Detection with Progressive Scale Expansion Network","date":"2019-03-28","arxiv_id":"1903.12473","repositories_listed":19,"syntology":null},{"url":"/paper/character-region-awareness-for-text-detection","title":"Character Region Awareness for Text Detection","date":"2019-04-03","arxiv_id":"1904.01941","repositories_listed":18,"syntology":{"n":40,"n_ran":6,"n_unverified":34,"n_pointer_only":4}},{"url":"/paper/abcnet-real-time-scene-text-spotting-with","title":"ABCNet: Real-time Scene Text Spotting with Adaptive Bezier-Curve Network","date":"2020-02-24","arxiv_id":"2002.10200","repositories_listed":16,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/real-time-scene-text-detection-with","title":"Real-time Scene Text Detection with Differentiable Binarization","date":"2019-11-20","arxiv_id":"1911.08947","repositories_listed":15,"syntology":{"n":25,"n_ran":3,"n_unverified":22,"n_pointer_only":0}},{"url":"/paper/fourier-contour-embedding-for-arbitrary","title":"Fourier Contour Embedding for Arbitrary-Shaped Text Detection","date":"2021-04-21","arxiv_id":"2104.10442","repositories_listed":12,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/shape-robust-text-detection-with-progressive","title":"Shape Robust Text Detection with Progressive Scale Expansion Network","date":"2018-06-07","arxiv_id":"1806.02559","repositories_listed":9,"syntology":{"n":24,"n_ran":5,"n_unverified":19,"n_pointer_only":2}},{"url":"/paper/fots-fast-oriented-text-spotting-with-a","title":"FOTS: Fast Oriented Text Spotting with a Unified Network","date":"2018-01-05","arxiv_id":"1801.01671","repositories_listed":7,"syntology":null},{"url":"/paper/textfusenet-scene-text-detection-with-richer","title":"TextFuseNet: Scene Text Detection with Richer Fused Features","date":"2020-05-17","arxiv_id":null,"repositories_listed":6,"syntology":null},{"url":"/paper/efficient-and-accurate-arbitrary-shaped-text","title":"Efficient and Accurate Arbitrary-Shaped Text Detection with Pixel Aggregation Network","date":"2019-08-16","arxiv_id":"1908.05900","repositories_listed":6,"syntology":{"n":28,"n_ran":4,"n_unverified":24,"n_pointer_only":0}},{"url":"/paper/detecting-oriented-text-in-natural-images-by","title":"Detecting Oriented Text in Natural Images by Linking Segments","date":"2017-03-19","arxiv_id":"1703.06520","repositories_listed":6,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/real-time-scene-text-detection-with-1","title":"Real-Time Scene Text Detection with Differentiable Binarization and Adaptive Scale Fusion","date":"2022-02-21","arxiv_id":"2202.10304","repositories_listed":5,"syntology":null},{"url":"/paper/pixellink-detecting-scene-text-via-instance","title":"PixelLink: Detecting Scene Text via Instance Segmentation","date":"2018-01-04","arxiv_id":"1801.01315","repositories_listed":5,"syntology":null},{"url":"/paper/robust-scene-text-recognition-with-automatic","title":"Robust Scene Text Recognition with Automatic Rectification","date":"2016-03-12","arxiv_id":"1603.03915","repositories_listed":5,"syntology":null},{"url":"/paper/arbitrary-oriented-scene-text-detection-via","title":"Arbitrary-Oriented Scene Text Detection via Rotation Proposals","date":"2017-03-03","arxiv_id":"1703.01086","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/coco-text-dataset-and-benchmark-for-text","title":"COCO-Text: Dataset and Benchmark for Text Detection and Recognition in Natural Images","date":"2016-01-26","arxiv_id":"1601.07140","repositories_listed":4,"syntology":null},{"url":"/paper/towards-end-to-end-unified-scene-text","title":"Towards End-to-End Unified Scene Text Detection and Layout Analysis","date":"2022-03-28","arxiv_id":"2203.15143","repositories_listed":3,"syntology":{"n":15,"n_ran":8,"n_unverified":7,"n_pointer_only":15}},{"url":"/paper/unrealtext-synthesizing-realistic-scene-text","title":"UnrealText: Synthesizing Realistic Scene Text Images from the Unreal World","date":"2020-03-24","arxiv_id":"2003.10608","repositories_listed":3,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/textsnake-a-flexible-representation-for","title":"TextSnake: A Flexible Representation for Detecting Text of Arbitrary Shapes","date":"2018-07-04","arxiv_id":"1807.01544","repositories_listed":3,"syntology":{"n":6,"n_ran":1,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/textboxes-a-single-shot-oriented-scene-text","title":"TextBoxes++: A Single-Shot Oriented Scene Text Detector","date":"2018-01-09","arxiv_id":"1801.02765","repositories_listed":3,"syntology":null},{"url":"/paper/stn-ocr-a-single-neural-network-for-text","title":"STN-OCR: A single Neural Network for Text Detection and Text Recognition","date":"2017-07-27","arxiv_id":"1707.08831","repositories_listed":3,"syntology":null},{"url":"/paper/srformer-empowering-regression-based-text","title":"SRFormer: Text Detection Transformer with Incorporated Segmentation and Regression","date":"2023-08-21","arxiv_id":"2308.10531","repositories_listed":2,"syntology":null},{"url":"/paper/efficient-and-accurate-scene-text-detection","title":"LRANet: Towards Accurate and Efficient Scene Text Detection with Low-Rank Approximation Network","date":"2023-06-27","arxiv_id":"2306.15142","repositories_listed":2,"syntology":{"n":19,"n_ran":1,"n_unverified":18,"n_pointer_only":0}},{"url":"/paper/vision-language-pre-training-for-boosting","title":"Vision-Language Pre-Training for Boosting Scene Text Detectors","date":"2022-04-29","arxiv_id":"2204.13867","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/swintextspotter-scene-text-spotting-via","title":"SwinTextSpotter: Scene Text Spotting via Better Synergy between Text Detection and Text Recognition","date":"2022-03-19","arxiv_id":"2203.10209","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":5}},{"url":"/paper/fast-searching-for-a-faster-arbitrarily","title":"FAST: Faster Arbitrarily-Shaped Text Detector with Minimalist Kernel Representation","date":"2021-11-03","arxiv_id":"2111.02394","repositories_listed":2,"syntology":null},{"url":"/paper/pgnet-real-time-arbitrarily-shaped-text","title":"PGNet: Real-time Arbitrarily-Shaped Text Spotting with Point Gathering Network","date":"2021-04-12","arxiv_id":"2104.05458","repositories_listed":2,"syntology":null},{"url":"/paper/pyramid-mask-text-detector","title":"Pyramid Mask Text Detector","date":"2019-03-28","arxiv_id":"1903.11800","repositories_listed":2,"syntology":null},{"url":"/paper/shopsign-a-diverse-scene-text-dataset-of","title":"ShopSign: a Diverse Scene Text Dataset of Chinese Shop Signs in Street Views","date":"2019-03-25","arxiv_id":"1903.10412","repositories_listed":2,"syntology":null}],"syntology_records":16,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}