{"url":"/method/r-cnn","slug":"r-cnn","name":"R-CNN","full_name":"R-CNN","full_name_withheld":false,"description_markdown":"**R-CNN**, or **Regions with CNN Features**, is an object detection model that uses high-capacity CNNs to bottom-up region proposals in order to localize and segment objects. It uses [selective search](https://paperswithcode.com/method/selective-search) to identify a number of bounding-box object region candidates (“regions of interest”), and then extracts features from each region independently for classification.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Rich feature hierarchies for accurate object detection and semantic segmentation","paper":"/paper/rich-feature-hierarchies-for-accurate-object","first_author":"Ross Girshick","n_authors":4,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/rich-feature-hierarchies-for-accurate-object"},"source":{"url":"http://arxiv.org/abs/1311.2524v5","title":"Rich feature hierarchies for accurate object detection and semantic segmentation","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Object Detection Models","url":"/methods/category/object-detection-models","pwc_aliases":[]}],"n_papers_tagged":34,"archive_num_papers":34,"papers_newest_first":[{"paper":null,"title":"Image quality enhancement of embedded holograms in holographic information hiding using deep neural networks","date":"2021-12-20","arxiv_id":"2112.11246","n_code_links":0,"syntology":null},{"paper":"/paper/kohtd-kazakh-offline-handwritten-text-dataset","title":"KOHTD: Kazakh Offline Handwritten Text Dataset","date":"2021-09-22","arxiv_id":"2110.04075","n_code_links":1,"syntology":null},{"paper":"/paper/toward-improving-confidence-in-autonomous","title":"Toward Improving Confidence in Autonomous Vehicle Software: A Study on Traffic Sign Recognition Systems","date":"2021-08-03","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/sentiment-analysis-for-urdu-online-reviews","title":"Sentiment analysis for Urdu online reviews using deep learning models","date":"2021-06-28","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/cholecseg8k-a-semantic-segmentation-dataset","title":"CholecSeg8k: A Semantic Segmentation Dataset for Laparoscopic Cholecystectomy Based on Cholec80","date":"2020-12-23","arxiv_id":"2012.12453","n_code_links":2,"syntology":null},{"paper":null,"title":"Improvement in Land Cover and Crop Classification based on Temporal Features Learning from Sentinel-2 Data Using Recurrent-Convolutional Neural Network (R-CNN)","date":"2020-04-27","arxiv_id":"2004.12880","n_code_links":0,"syntology":null},{"paper":null,"title":"Facial Action Unit Detection on ICU Data for Pain Assessment","date":"2020-04-24","arxiv_id":"2005.02121","n_code_links":0,"syntology":null},{"paper":"/paper/ps-rcnn-detecting-secondary-human-instances","title":"PS-RCNN: Detecting Secondary Human Instances in a Crowd via Primary Object Suppression","date":"2020-03-16","arxiv_id":"2003.07080","n_code_links":0,"syntology":null},{"paper":"/paper/transferring-dense-pose-to-proximal-animal","title":"Transferring Dense Pose to Proximal Animal Classes","date":"2020-02-28","arxiv_id":"2003.00080","n_code_links":1,"syntology":null},{"paper":"/paper/smoke-single-stage-monocular-3d-object","title":"SMOKE: Single-Stage Monocular 3D Object Detection via Keypoint Estimation","date":"2020-02-24","arxiv_id":"2002.10111","n_code_links":3,"syntology":{"ran":2,"of":9,"unverified":7,"pointer_only":0}},{"paper":null,"title":"Fine-Grained Object Detection over Scientific Document Images with Region Embeddings","date":"2019-10-28","arxiv_id":"1910.12462","n_code_links":0,"syntology":null},{"paper":"/paper/stela-a-real-time-scene-text-detector-with","title":"STELA: A Real-Time Scene Text Detector with Learned Anchor","date":"2019-09-17","arxiv_id":"1909.07549","n_code_links":1,"syntology":null},{"paper":null,"title":"Multi-scale Aggregation R-CNN for 2D Multi-person Pose Estimation","date":"2019-05-10","arxiv_id":"1905.03912","n_code_links":0,"syntology":null},{"paper":null,"title":"A Review of Object Detection Models based on Convolutional Neural Network","date":"2019-05-05","arxiv_id":"1905.01614","n_code_links":0,"syntology":null},{"paper":null,"title":"BoLTVOS: Box-Level Tracking for Video Object Segmentation","date":"2019-04-09","arxiv_id":"1904.04552","n_code_links":0,"syntology":null},{"paper":"/paper/deeplung-deep-3d-dual-path-nets-for-automated","title":"DeepLung: Deep 3D Dual Path Nets for Automated Pulmonary Nodule Detection and Classification","date":"2018-01-25","arxiv_id":"1801.09555","n_code_links":2,"syntology":null},{"paper":"/paper/light-head-r-cnn-in-defense-of-two-stage","title":"Light-Head R-CNN: In Defense of Two-Stage Object Detector","date":"2017-11-20","arxiv_id":"1711.07264","n_code_links":5,"syntology":null},{"paper":null,"title":"The Use of Object Labels and Spatial Prepositions as Keywords in a Web-Retrieval-Based Image Caption Generation System","date":"2017-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Segmentation of Instances by Hashing","date":"2017-02-27","arxiv_id":"1702.08160","n_code_links":0,"syntology":null},{"paper":null,"title":"Learning Semantic Part-Based Models from Google Images","date":"2016-09-11","arxiv_id":"1609.03140","n_code_links":0,"syntology":null},{"paper":null,"title":"Shallow Networks for High-Accuracy Road Object-Detection","date":"2016-06-05","arxiv_id":"1606.01561","n_code_links":0,"syntology":null},{"paper":"/paper/craft-objects-from-images","title":"CRAFT Objects from Images","date":"2016-04-12","arxiv_id":"1604.03239","n_code_links":1,"syntology":null},{"paper":"/paper/relief-r-cnn-utilizing-convolutional-features","title":"Relief R-CNN : Utilizing Convolutional Features for Fast Object Detection","date":"2016-01-25","arxiv_id":"1601.06719","n_code_links":1,"syntology":null},{"paper":"/paper/context-aware-cnns-for-person-head-detection","title":"Context-aware CNNs for person head detection","date":"2015-11-24","arxiv_id":"1511.07917","n_code_links":1,"syntology":null},{"paper":null,"title":"DeePM: A Deep Part-Based Model for Object Detection and Semantic Part Localization","date":"2015-11-23","arxiv_id":"1511.07131","n_code_links":0,"syntology":null},{"paper":null,"title":"What is Holding Back Convnets for Detection?","date":"2015-08-12","arxiv_id":"1508.02844","n_code_links":0,"syntology":null},{"paper":null,"title":"Deep CNN Ensemble with Data Augmentation for Object Detection","date":"2015-06-24","arxiv_id":"1506.07224","n_code_links":0,"syntology":null},{"paper":null,"title":"R-CNN minus R","date":"2015-06-23","arxiv_id":"1506.06981","n_code_links":0,"syntology":null},{"paper":null,"title":"What makes for effective detection proposals?","date":"2015-02-17","arxiv_id":"1502.05082","n_code_links":0,"syntology":null},{"paper":null,"title":"segDeepM: Exploiting Segmentation and Context in Deep Neural Networks for Object Detection","date":"2015-02-15","arxiv_id":"1502.04275","n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/object-detection","name":"Object Detection","papers":16},{"task":"/task/object-detection-1","name":"object-detection","papers":14},{"task":"/task/object","name":"Object","papers":13},{"task":"/task/semantic-segmentation","name":"Semantic Segmentation","papers":4},{"task":"/task/classification","name":"General Classification","papers":3},{"task":"/task/human-detection","name":"Human Detection","papers":3},{"task":"/task/region-proposal","name":"Region Proposal","papers":3},{"task":"/task/segmentation","name":"Segmentation","papers":3},{"task":"/task/autonomous-vehicles","name":"Autonomous Vehicles","papers":2},{"task":"/task/classification-1","name":"Classification","papers":2},{"task":"/task/data-augmentation","name":"Data Augmentation","papers":2},{"task":"/task/image-classification","name":"Image Classification","papers":2},{"task":"/task/pose-estimation","name":"Pose Estimation","papers":2},{"task":"/task/3d-object-detection","name":"3D Object Detection","papers":1},{"task":"/task/action-classification","name":"Action Classification","papers":1},{"task":"/task/action-detection","name":"Action Detection","papers":1},{"task":"/task/action-unit-detection","name":"Action Unit Detection","papers":1},{"task":"/task/autonomous-driving","name":"Autonomous Driving","papers":1},{"task":"/task/autonomous-navigation","name":"Autonomous Navigation","papers":1},{"task":"/task/machine-learning","name":"BIG-bench Machine Learning","papers":1}],"tasks_shown":20,"n_tasks":62,"usage_by_year":[{"year":"2013","papers":1},{"year":"2014","papers":3},{"year":"2015","papers":7},{"year":"2016","papers":4},{"year":"2017","papers":3},{"year":"2018","papers":1},{"year":"2019","papers":5},{"year":"2020","papers":6},{"year":"2021","papers":4}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/r-cnn"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}