{"url":"/method/poolformer","slug":"poolformer","name":"PoolFormer","full_name":"PoolFormer","full_name_withheld":false,"description_markdown":"PoolFormer is instantiated from MetaFormer by specifying the token mixer as extremely simple operator, pooling. PoolFormer is utilized as a tool to verify MetaFormer hypothesis \"MetaFormer is actually what you need\" (vs \"Attention is all you need\").","description_state":"present","introduced_year":null,"introduced_by":{"title":"MetaFormer Is Actually What You Need for Vision","paper":"/paper/metaformer-is-actually-what-you-need-for","first_author":"Weihao Yu","n_authors":8,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/metaformer-is-actually-what-you-need-for"},"source":{"url":"https://arxiv.org/abs/2111.11418v3","title":"MetaFormer Is Actually What You Need for Vision","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Image Models","url":"/methods/category/image-models","pwc_aliases":[]}],"n_papers_tagged":9,"archive_num_papers":9,"papers_newest_first":[{"paper":null,"title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","date":"2025-02-07","arxiv_id":"2502.05091","n_code_links":0,"syntology":null},{"paper":"/paper/gestformer-multiscale-wavelet-pooling","title":"GestFormer: Multiscale Wavelet Pooling Transformer Network for Dynamic Hand Gesture Recognition","date":"2024-05-18","arxiv_id":"2405.11180","n_code_links":1,"syntology":null},{"paper":null,"title":"Spatiotemporal Pooling on Appropriate Topological Maps Represented as Two-Dimensional Images for EEG Classification","date":"2024-03-07","arxiv_id":"2403.04353","n_code_links":0,"syntology":null},{"paper":null,"title":"Enhancing Transformer-Based Segmentation for Breast Cancer Diagnosis using Auto-Augmentation and Search Optimisation Techniques","date":"2023-11-18","arxiv_id":"2311.11065","n_code_links":0,"syntology":null},{"paper":null,"title":"Advancing Ischemic Stroke Diagnosis: A Novel Two-Stage Approach for Blood Clot Origin Identification","date":"2023-04-26","arxiv_id":"2304.13775","n_code_links":0,"syntology":null},{"paper":"/paper/metaformer-baselines-for-vision","title":"MetaFormer Baselines for Vision","date":"2022-10-24","arxiv_id":"2210.13452","n_code_links":8,"syntology":{"ran":0,"of":4,"unverified":4,"pointer_only":0}},{"paper":"/paper/efficientformer-vision-transformers-at","title":"EfficientFormer: Vision Transformers at MobileNet Speed","date":"2022-06-02","arxiv_id":"2206.01191","n_code_links":13,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":1}},{"paper":"/paper/wavemix-lite-a-resource-efficient-neural","title":"WaveMix: A Resource-efficient Neural Network for Image Analysis","date":"2022-05-28","arxiv_id":"2205.14375","n_code_links":1,"syntology":null},{"paper":"/paper/metaformer-is-actually-what-you-need-for","title":"MetaFormer Is Actually What You Need for Vision","date":"2021-11-22","arxiv_id":"2111.11418","n_code_links":18,"syntology":{"ran":2,"of":3,"unverified":1,"pointer_only":0}}],"papers_shown":9,"tasks":[{"task":"/task/image-classification","name":"Image Classification","papers":4},{"task":"/task/semantic-segmentation","name":"Semantic Segmentation","papers":2},{"task":"/task/anomaly-detection","name":"Anomaly Detection","papers":1},{"task":"/task/brain-computer-interface","name":"Brain Computer Interface","papers":1},{"task":"/task/classification-1","name":"Classification","papers":1},{"task":"/task/computed-tomography-ct","name":"Computed Tomography (CT)","papers":1},{"task":"/task/domain-generalization","name":"Domain Generalization","papers":1},{"task":"/task/eeg-1","name":"EEG","papers":1},{"task":"/task/efficient-neural-network","name":"Efficient Neural Network","papers":1},{"task":null,"name":"GPU","papers":1},{"task":"/task/gesture-recognition","name":"Gesture Recognition","papers":1},{"task":"/task/hand-gesture-recognition","name":"Hand Gesture Recognition","papers":1},{"task":"/task/hand-gesture-recognition-1","name":"Hand-Gesture Recognition","papers":1},{"task":"/task/image-augmentation","name":"Image Augmentation","papers":1},{"task":"/task/image-text-retrieval","name":"Image-text Retrieval","papers":1},{"task":"/task/language-modeling","name":"Language Modeling","papers":1},{"task":"/task/language-modelling","name":"Language Modelling","papers":1},{"task":"/task/management","name":"Management","papers":1},{"task":"/task/motor-imagery","name":"Motor Imagery","papers":1},{"task":"/task/object-detection","name":"Object Detection","papers":1}],"tasks_shown":20,"n_tasks":29,"usage_by_year":[{"year":"2021","papers":1},{"year":"2022","papers":3},{"year":"2023","papers":2},{"year":"2024","papers":2},{"year":"2025","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/poolformer"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}