{"url":"/method/roi-align","slug":"roi-align","name":"RoIAlign","full_name":"RoIAlign","full_name_withheld":false,"description_markdown":"**Region of Interest Align**, or **RoIAlign**, is an operation for extracting a small feature map from each RoI in detection and segmentation based tasks. It removes the harsh quantization of [RoI Pool](https://paperswithcode.com/method/roi-pooling), properly *aligning* the extracted features with the input. To avoid any quantization of the RoI boundaries or bins (using $x/16$ instead of $[x/16]$), RoIAlign uses bilinear interpolation to compute the exact values of the input features at four regularly sampled locations in each RoI bin, and the result is then aggregated (using max or average).","description_state":"present","introduced_year":null,"introduced_by":{"title":"Mask R-CNN","paper":"/paper/mask-r-cnn","first_author":"Kaiming He","n_authors":4,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/mask-r-cnn"},"source":{"url":"http://arxiv.org/abs/1703.06870v3","title":"Mask R-CNN","url_on_a_paper_host":true},"code_snippet_url":"https://github.com/facebookresearch/detectron2/blob/bb9f5d8e613358519c9865609ab3fe7b6571f2ba/detectron2/layers/roi_align.py#L51","code_snippet_url_on_a_code_host":true,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"RoI Feature Extractors","url":"/methods/category/roi-feature-extractors","pwc_aliases":[]}],"n_papers_tagged":611,"archive_num_papers":611,"papers_newest_first":[{"paper":null,"title":"BlenderFusion: 3D-Grounded Visual Editing and Generative Compositing","date":"2025-06-20","arxiv_id":"2506.17450","n_code_links":0,"syntology":null},{"paper":null,"title":"A novel visual data-based diagnostic approach for estimation of regime transition in pool boiling","date":"2025-06-12","arxiv_id":"2506.10832","n_code_links":0,"syntology":null},{"paper":null,"title":"Bringing SAM to new heights: Leveraging elevation data for tree crown segmentation from drone imagery","date":"2025-06-05","arxiv_id":"2506.04970","n_code_links":0,"syntology":null},{"paper":null,"title":"Hierarchical Text Classification Using Contrastive Learning Informed Path Guided Hierarchy","date":"2025-06-04","arxiv_id":"2506.04381","n_code_links":0,"syntology":null},{"paper":"/paper/zero-to-hero-zero-shot-initialization","title":"Zero-to-Hero: Zero-Shot Initialization Empowering Reference-Based Video Appearance Editing","date":"2025-05-29","arxiv_id":"2505.23134","n_code_links":1,"syntology":null},{"paper":null,"title":"Force Prompting: Video Generation Models Can Learn and Generalize Physics-based Control Signals","date":"2025-05-26","arxiv_id":"2505.19386","n_code_links":0,"syntology":null},{"paper":"/paper/ob3d-a-new-dataset-for-benchmarking","title":"OB3D: A New Dataset for Benchmarking Omnidirectional 3D Reconstruction Using Blender","date":"2025-05-26","arxiv_id":"2505.20126","n_code_links":1,"syntology":null},{"paper":null,"title":"Detailed Evaluation of Modern Machine Learning Approaches for Optic Plastics Sorting","date":"2025-05-22","arxiv_id":"2505.16513","n_code_links":0,"syntology":null},{"paper":null,"title":"SurgPose: Generalisable Surgical Instrument Pose Estimation using Zero-Shot Learning and Stereo Vision","date":"2025-05-16","arxiv_id":"2505.11439","n_code_links":0,"syntology":null},{"paper":"/paper/kg-htc-integrating-knowledge-graphs-into-llms","title":"KG-HTC: Integrating Knowledge Graphs into LLMs for Effective Zero-shot Hierarchical Text Classification","date":"2025-05-08","arxiv_id":"2505.05583","n_code_links":1,"syntology":null},{"paper":null,"title":"DARTer: Dynamic Adaptive Representation Tracker for Nighttime UAV Tracking","date":"2025-05-01","arxiv_id":"2505.00752","n_code_links":0,"syntology":null},{"paper":null,"title":"A Robust Deep Networks based Multi-Object MultiCamera Tracking System for City Scale Traffic","date":"2025-05-01","arxiv_id":"2505.00534","n_code_links":0,"syntology":null},{"paper":null,"title":"Transcending Dimensions using Generative AI: Real-Time 3D Model Generation in Augmented Reality","date":"2025-04-27","arxiv_id":"2504.21033","n_code_links":0,"syntology":null},{"paper":null,"title":"Learning Underwater Active Perception in Simulation","date":"2025-04-23","arxiv_id":"2504.17817","n_code_links":0,"syntology":null},{"paper":null,"title":"Real-time Seafloor Segmentation and Mapping","date":"2025-04-14","arxiv_id":"2504.10750","n_code_links":0,"syntology":null},{"paper":null,"title":"From Specificity to Generality: Revisiting Generalizable Artifacts in Detecting Face Deepfakes","date":"2025-04-07","arxiv_id":"2504.04827","n_code_links":0,"syntology":null},{"paper":"/paper/blendergym-benchmarking-foundational-model","title":"BlenderGym: Benchmarking Foundational Model Systems for Graphics Editing","date":"2025-04-02","arxiv_id":"2504.01786","n_code_links":1,"syntology":null},{"paper":null,"title":"RipVIS: Rip Currents Video Instance Segmentation Benchmark for Beach Monitoring and Safety","date":"2025-04-01","arxiv_id":"2504.01128","n_code_links":0,"syntology":null},{"paper":"/paper/ai-assisted-colonoscopy-polyp-detection-and","title":"AI-Assisted Colonoscopy: Polyp Detection and Segmentation using Foundation Models","date":"2025-03-31","arxiv_id":"2503.24138","n_code_links":1,"syntology":null},{"paper":null,"title":"Assessing SAM for Tree Crown Instance Segmentation from Drone Imagery","date":"2025-03-26","arxiv_id":"2503.20199","n_code_links":0,"syntology":null},{"paper":"/paper/tgbformer-transformer-graphformer-blender","title":"TGBFormer: Transformer-GraphFormer Blender Network for Video Object Detection","date":"2025-03-18","arxiv_id":"2503.13903","n_code_links":0,"syntology":null},{"paper":null,"title":"YOLO-LLTS: Real-Time Low-Light Traffic Sign Detection via Prior-Guided Enhancement and Multi-Branch Feature Interaction","date":"2025-03-18","arxiv_id":"2503.13883","n_code_links":0,"syntology":null},{"paper":null,"title":"DivCon-NeRF: Generating Augmented Rays with Diversity and Consistency for Few-shot View Synthesis","date":"2025-03-17","arxiv_id":"2503.12947","n_code_links":0,"syntology":null},{"paper":null,"title":"Securing Virtual Reality Experiences: Unveiling and Tackling Cybersickness Attacks with Explainable AI","date":"2025-03-17","arxiv_id":"2503.13419","n_code_links":0,"syntology":null},{"paper":"/paper/motion-blender-gaussian-splatting-for-dynamic","title":"Motion Blender Gaussian Splatting for Dynamic Scene Reconstruction","date":"2025-03-12","arxiv_id":"2503.09040","n_code_links":1,"syntology":null},{"paper":"/paper/overlock-an-overview-first-look-closely-next","title":"OverLoCK: An Overview-first-Look-Closely-next ConvNet with Context-Mixing Dynamic Kernels","date":"2025-02-27","arxiv_id":"2502.20087","n_code_links":1,"syntology":{"ran":0,"of":3,"unverified":3,"pointer_only":0}},{"paper":null,"title":"GHOST 2.0: generative high-fidelity one shot transfer of heads","date":"2025-02-25","arxiv_id":"2502.18417","n_code_links":0,"syntology":null},{"paper":null,"title":"Analysis and Prediction of Coverage and Channel Rank for UAV Networks in Rural Scenarios with Foliage","date":"2025-02-14","arxiv_id":"2502.10324","n_code_links":0,"syntology":null},{"paper":null,"title":"Hybrid Answer Set Programming: Foundations and Applications","date":"2025-02-13","arxiv_id":"2502.09235","n_code_links":0,"syntology":null},{"paper":"/paper/sasvi-segment-any-surgical-video","title":"SASVi - Segment Any Surgical Video","date":"2025-02-12","arxiv_id":"2502.09653","n_code_links":1,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/semantic-segmentation","name":"Semantic Segmentation","papers":214},{"task":"/task/instance-segmentation","name":"Instance Segmentation","papers":191},{"task":"/task/object-detection","name":"Object Detection","papers":168},{"task":"/task/segmentation","name":"Segmentation","papers":150},{"task":"/task/object-detection-1","name":"object-detection","papers":144},{"task":"/task/object","name":"Object","papers":95},{"task":"/task/image-classification","name":"Image Classification","papers":29},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":25},{"task":"/task/classification-1","name":"Classification","papers":20},{"task":"/task/nerf","name":"NeRF","papers":20},{"task":"/task/pose-estimation","name":"Pose Estimation","papers":20},{"task":"/task/text-classification","name":"Text Classification","papers":20},{"task":"/task/text-classification-1","name":"text-classification","papers":19},{"task":"/task/data-augmentation","name":"Data Augmentation","papers":18},{"task":"/task/classification","name":"General Classification","papers":18},{"task":"/task/image-classification","name":"image-classification","papers":18},{"task":"/task/panoptic-segmentation","name":"Panoptic Segmentation","papers":17},{"task":"/task/regression-1","name":"regression","papers":17},{"task":"/task/image-segmentation","name":"Image Segmentation","papers":16},{"task":"/task/novel-view-synthesis","name":"Novel View Synthesis","papers":16}],"tasks_shown":20,"n_tasks":409,"usage_by_year":[{"year":"2017","papers":9},{"year":"2018","papers":33},{"year":"2019","papers":84},{"year":"2020","papers":99},{"year":"2021","papers":119},{"year":"2022","papers":65},{"year":"2023","papers":81},{"year":"2024","papers":79},{"year":"2025","papers":42}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/roi-align"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}