{"url":"/method/roipool","slug":"roipool","name":"RoIPool","full_name":"RoIPool","full_name_withheld":false,"description_markdown":"**Region of Interest Pooling**, or **RoIPool**, is an operation for extracting a small feature map (e.g., $7×7$) from each RoI in detection and segmentation based tasks. Features are extracted from each candidate box, and thereafter in models like [Fast R-CNN](https://paperswithcode.com/method/fast-r-cnn), are then classified and bounding box regression performed.\r\n\r\nThe actual scaling to, e.g., $7×7$, occurs by dividing the region proposal into equally sized sections, finding the largest value in each section, and then copying these max values to the output buffer. In essence, **RoIPool** is [max pooling](https://paperswithcode.com/method/max-pooling) on a discrete grid based on a box.\r\n\r\nImage Source: [Joyce Xu](https://towardsdatascience.com/deep-learning-for-object-detection-a-comprehensive-review-73930816d8d9)","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":"http://arxiv.org/abs/1311.2524v5","title":"Rich feature hierarchies for accurate object detection and semantic segmentation","url_on_a_paper_host":true},"code_snippet_url":"https://github.com/pytorch/vision/blob/5e9ebe8dadc0ea2841a46cfcd82a93b4ce0d4519/torchvision/ops/roi_pool.py#L10","code_snippet_url_on_a_code_host":true,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"RoI Feature Extractors","url":"/methods/category/roi-feature-extractors","pwc_aliases":[]}],"n_papers_tagged":620,"archive_num_papers":null,"papers_newest_first":[{"paper":"/paper/3d-gaussian-splat-vulnerabilities","title":"3D Gaussian Splat Vulnerabilities","date":"2025-05-30","arxiv_id":"2506.00280","n_code_links":1,"syntology":null},{"paper":"/paper/zero-to-hero-zero-shot-initialization","title":"Zero-to-Hero: Zero-Shot Initialization Empowering Reference-Based Video Appearance Editing","date":"2025-05-29","arxiv_id":"2505.23134","n_code_links":1,"syntology":null},{"paper":null,"title":"Force Prompting: Video Generation Models Can Learn and Generalize Physics-based Control Signals","date":"2025-05-26","arxiv_id":"2505.19386","n_code_links":0,"syntology":null},{"paper":"/paper/ob3d-a-new-dataset-for-benchmarking","title":"OB3D: A New Dataset for Benchmarking Omnidirectional 3D Reconstruction Using Blender","date":"2025-05-26","arxiv_id":"2505.20126","n_code_links":1,"syntology":null},{"paper":null,"title":"AppleGrowthVision: A large-scale stereo dataset for phenological analysis, fruit detection, and 3D reconstruction in apple orchards","date":"2025-05-20","arxiv_id":"2505.14029","n_code_links":0,"syntology":null},{"paper":null,"title":"Object detection in adverse weather conditions for autonomous vehicles using Instruct Pix2Pix","date":"2025-05-13","arxiv_id":"2505.08228","n_code_links":0,"syntology":null},{"paper":null,"title":"DARTer: Dynamic Adaptive Representation Tracker for Nighttime UAV Tracking","date":"2025-05-01","arxiv_id":"2505.00752","n_code_links":0,"syntology":null},{"paper":null,"title":"Learning Underwater Active Perception in Simulation","date":"2025-04-23","arxiv_id":"2504.17817","n_code_links":0,"syntology":null},{"paper":null,"title":"FMNV: A Dataset of Media-Published News Videos for Fake News Detection","date":"2025-04-10","arxiv_id":"2504.07687","n_code_links":0,"syntology":null},{"paper":null,"title":"From Specificity to Generality: Revisiting Generalizable Artifacts in Detecting Face Deepfakes","date":"2025-04-07","arxiv_id":"2504.04827","n_code_links":0,"syntology":null},{"paper":"/paper/blendergym-benchmarking-foundational-model","title":"BlenderGym: Benchmarking Foundational Model Systems for Graphics Editing","date":"2025-04-02","arxiv_id":"2504.01786","n_code_links":1,"syntology":null},{"paper":null,"title":"BBoxCut: A Targeted Data Augmentation Technique for Enhancing Wheat Head Detection Under Occlusions","date":"2025-03-31","arxiv_id":"2503.24032","n_code_links":0,"syntology":null},{"paper":"/paper/a-gan-enhanced-deep-learning-framework-for","title":"A GAN-Enhanced Deep Learning Framework for Rooftop Detection from Historical Aerial Imagery","date":"2025-03-29","arxiv_id":"2503.23200","n_code_links":1,"syntology":null},{"paper":null,"title":"Autonomous AI for Multi-Pathology Detection in Chest X-Rays: A Multi-Site Study in the Indian Healthcare System","date":"2025-03-28","arxiv_id":"2504.00022","n_code_links":0,"syntology":null},{"paper":null,"title":"Exploring Few-Shot Object Detection on Blood Smear Images: A Case Study of Leukocytes and Schistocytes","date":"2025-03-21","arxiv_id":"2503.17107","n_code_links":0,"syntology":null},{"paper":"/paper/tgbformer-transformer-graphformer-blender","title":"TGBFormer: Transformer-GraphFormer Blender Network for Video Object Detection","date":"2025-03-18","arxiv_id":"2503.13903","n_code_links":0,"syntology":null},{"paper":null,"title":"DivCon-NeRF: Generating Augmented Rays with Diversity and Consistency for Few-shot View Synthesis","date":"2025-03-17","arxiv_id":"2503.12947","n_code_links":0,"syntology":null},{"paper":"/paper/motion-blender-gaussian-splatting-for-dynamic","title":"Motion Blender Gaussian Splatting for Dynamic Scene Reconstruction","date":"2025-03-12","arxiv_id":"2503.09040","n_code_links":1,"syntology":null},{"paper":"/paper/walnutdata-a-uav-remote-sensing-dataset-of","title":"WalnutData: A UAV Remote Sensing Dataset of Green Walnuts and Model Evaluation","date":"2025-02-27","arxiv_id":"2502.20092","n_code_links":1,"syntology":null},{"paper":null,"title":"Automatic Vehicle Detection using DETR: A Transformer-Based Approach for Navigating Treacherous Roads","date":"2025-02-25","arxiv_id":"2502.17843","n_code_links":0,"syntology":null},{"paper":null,"title":"Autonomous Vision-Guided Resection of Central Airway Obstruction","date":"2025-02-25","arxiv_id":"2502.18586","n_code_links":0,"syntology":null},{"paper":null,"title":"GHOST 2.0: generative high-fidelity one shot transfer of heads","date":"2025-02-25","arxiv_id":"2502.18417","n_code_links":0,"syntology":null},{"paper":null,"title":"Analysis and Prediction of Coverage and Channel Rank for UAV Networks in Rural Scenarios with Foliage","date":"2025-02-14","arxiv_id":"2502.10324","n_code_links":0,"syntology":null},{"paper":null,"title":"Demystifying Catastrophic Forgetting in Two-Stage Incremental Object Detector","date":"2025-02-08","arxiv_id":"2502.05540","n_code_links":0,"syntology":null},{"paper":null,"title":"Adaptive Object Detection for Indoor Navigation Assistance: A Performance Evaluation of Real-Time Algorithms","date":"2025-01-30","arxiv_id":"2501.18444","n_code_links":0,"syntology":null},{"paper":null,"title":"Advanced technology in railway track monitoring using the GPR Technique: A Review","date":"2025-01-19","arxiv_id":"2501.11132","n_code_links":0,"syntology":null},{"paper":null,"title":"RDG-GS: Relative Depth Guidance with Gaussian Splatting for Real-time Sparse-View 3D Rendering","date":"2025-01-19","arxiv_id":"2501.11102","n_code_links":0,"syntology":null},{"paper":null,"title":"A method for estimating roadway billboard salience","date":"2025-01-13","arxiv_id":"2501.07342","n_code_links":0,"syntology":null},{"paper":"/paper/opengert-open-source-automated-geometry","title":"OpenGERT: Open Source Automated Geometry Extraction with Geometric and Electromagnetic Sensitivity Analyses for Ray-Tracing Propagation Models","date":"2025-01-12","arxiv_id":"2501.06945","n_code_links":1,"syntology":null},{"paper":null,"title":"DropoutGS: Dropping Out Gaussians for Better Sparse-view Rendering","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/object-detection","name":"Object Detection","papers":342},{"task":"/task/object-detection-1","name":"object-detection","papers":319},{"task":"/task/object","name":"Object","papers":190},{"task":"/task/region-proposal","name":"Region Proposal","papers":39},{"task":"/task/semantic-segmentation","name":"Semantic Segmentation","papers":37},{"task":"/task/image-classification","name":"Image Classification","papers":35},{"task":"/task/instance-segmentation","name":"Instance Segmentation","papers":29},{"task":"/task/classification","name":"General Classification","papers":26},{"task":"/task/image-classification","name":"image-classification","papers":25},{"task":"/task/deep-learning","name":"Deep Learning","papers":23},{"task":"/task/nerf","name":"NeRF","papers":20},{"task":"/task/autonomous-driving","name":"Autonomous Driving","papers":19},{"task":"/task/data-augmentation","name":"Data Augmentation","papers":19},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":19},{"task":"/task/pedestrian-detection","name":"Pedestrian Detection","papers":18},{"task":"/task/classification-1","name":"Classification","papers":16},{"task":null,"name":"GPU","papers":16},{"task":"/task/novel-view-synthesis","name":"Novel View Synthesis","papers":16},{"task":"/task/segmentation","name":"Segmentation","papers":16},{"task":"/task/domain-adaptation","name":"Domain Adaptation","papers":15}],"tasks_shown":20,"n_tasks":360,"usage_by_year":[{"year":"2015","papers":7},{"year":"2016","papers":18},{"year":"2017","papers":33},{"year":"2018","papers":68},{"year":"2019","papers":74},{"year":"2020","papers":79},{"year":"2021","papers":92},{"year":"2022","papers":78},{"year":"2023","papers":66},{"year":"2024","papers":74},{"year":"2025","papers":31}],"row_source":"embedded","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/roipool"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}