{"url":"/method/position-wise-feed-forward-layer","slug":"position-wise-feed-forward-layer","name":"Position-Wise Feed-Forward Layer","full_name":"Position-Wise Feed-Forward Layer","full_name_withheld":false,"description_markdown":"**Position-Wise Feed-Forward Layer** is a type of [feedforward layer](https://www.paperswithcode.com/method/category/feedforwad-networks) consisting of two [dense layers](https://www.paperswithcode.com/method/dense-connections) that applies to the last dimension, which means the same dense layers are used for each position item in the sequence, so called position-wise.","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":"https://arxiv.org/abs/1706.03762v7","title":"Attention Is All You Need","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Feedforward Networks","url":"/methods/category/feedforward-networks","pwc_aliases":[]}],"n_papers_tagged":13895,"archive_num_papers":null,"papers_newest_first":[{"paper":"/paper/towards-robust-multimodal-emotion-recognition","title":"Towards Robust Multimodal Emotion Recognition under Missing Modalities and Distribution Shifts","date":"2025-06-12","arxiv_id":"2506.10452","n_code_links":1,"syntology":null},{"paper":null,"title":"SparseSSM: Efficient Selective Structured State Space Models Can Be Pruned in One-Shot","date":"2025-06-11","arxiv_id":"2506.09613","n_code_links":0,"syntology":null},{"paper":null,"title":"Hierarchical Neural Collapse Detection Transformer for Class Incremental Object Detection","date":"2025-06-10","arxiv_id":"2506.08562","n_code_links":0,"syntology":null},{"paper":"/paper/hyperspectral-image-classification-via","title":"Hyperspectral Image Classification via Transformer-based Spectral-Spatial Attention Decoupling and Adaptive Gating","date":"2025-06-10","arxiv_id":"2506.08324","n_code_links":1,"syntology":null},{"paper":null,"title":"MedMoE: Modality-Specialized Mixture of Experts for Medical Vision-Language Understanding","date":"2025-06-10","arxiv_id":"2506.08356","n_code_links":0,"syntology":null},{"paper":"/paper/patchguard-adversarially-robust-anomaly-1","title":"PatchGuard: Adversarially Robust Anomaly Detection and Localization through Vision Transformers and Pseudo Anomalies","date":"2025-06-10","arxiv_id":"2506.09237","n_code_links":2,"syntology":null},{"paper":null,"title":"Robust Visual Localization via Semantic-Guided Multi-Scale Transformer","date":"2025-06-10","arxiv_id":"2506.08526","n_code_links":0,"syntology":null},{"paper":"/paper/tactic-translation-agents-with-cognitive","title":"TACTIC: Translation Agents with Cognitive-Theoretic Interactive Collaboration","date":"2025-06-10","arxiv_id":"2506.08403","n_code_links":1,"syntology":null},{"paper":null,"title":"4DGT: Learning a 4D Gaussian Transformer Using Real-World Monocular Videos","date":"2025-06-09","arxiv_id":"2506.08015","n_code_links":0,"syntology":null},{"paper":null,"title":"Lightweight Sequential Transformers for Blood Glucose Level Prediction in Type-1 Diabetes","date":"2025-06-09","arxiv_id":"2506.07864","n_code_links":0,"syntology":null},{"paper":null,"title":"MADFormer: Mixed Autoregressive and Diffusion Transformers for Continuous Image Generation","date":"2025-06-09","arxiv_id":"2506.07999","n_code_links":0,"syntology":null},{"paper":null,"title":"Quantum Graph Transformer for NLP Sentiment Classification","date":"2025-06-09","arxiv_id":"2506.07937","n_code_links":0,"syntology":null},{"paper":null,"title":"Breaking Data Silos: Towards Open and Scalable Mobility Foundation Models via Generative Continual Learning","date":"2025-06-07","arxiv_id":"2506.06694","n_code_links":0,"syntology":null},{"paper":null,"title":"Can In-Context Reinforcement Learning Recover From Reward Poisoning Attacks?","date":"2025-06-07","arxiv_id":"2506.06891","n_code_links":0,"syntology":null},{"paper":null,"title":"BEAST: Efficient Tokenization of B-Splines Encoded Action Sequences for Imitation Learning","date":"2025-06-06","arxiv_id":"2506.06072","n_code_links":0,"syntology":null},{"paper":null,"title":"Intentionally Unintentional: GenAI Exceptionalism and the First Amendment","date":"2025-06-05","arxiv_id":"2506.05211","n_code_links":0,"syntology":null},{"paper":null,"title":"Enhancing Automatic PT Tagging for MEDLINE Citations Using Transformer-Based Models","date":"2025-06-03","arxiv_id":"2506.03321","n_code_links":0,"syntology":null},{"paper":null,"title":"Rhythm Controllable and Efficient Zero-Shot Voice Conversion via Shortcut Flow Matching","date":"2025-06-01","arxiv_id":"2506.01014","n_code_links":0,"syntology":null},{"paper":null,"title":"Channel-Imposed Fusion: A Simple yet Effective Method for Medical Time Series Classification","date":"2025-05-31","arxiv_id":"2506.00337","n_code_links":0,"syntology":null},{"paper":null,"title":"Evaluating Robot Policies in a World Model","date":"2025-05-31","arxiv_id":"2506.00613","n_code_links":0,"syntology":null},{"paper":null,"title":"Machine vs Machine: Using AI to Tackle Generative AI Threats in Assessment","date":"2025-05-31","arxiv_id":"2506.02046","n_code_links":0,"syntology":null},{"paper":"/paper/translate-with-care-addressing-gender-bias","title":"Translate With Care: Addressing Gender Bias, Neutrality, and Reasoning in Large Language Model Translations","date":"2025-05-31","arxiv_id":"2506.00748","n_code_links":1,"syntology":null},{"paper":null,"title":"Cross-Attention Speculative Decoding","date":"2025-05-30","arxiv_id":"2505.24544","n_code_links":0,"syntology":null},{"paper":null,"title":"D2AF: A Dual-Driven Annotation and Filtering Framework for Visual Grounding","date":"2025-05-30","arxiv_id":"2505.24372","n_code_links":0,"syntology":null},{"paper":null,"title":"Leveraging Intermediate Features of Vision Transformer for Face Anti-Spoofing","date":"2025-05-30","arxiv_id":"2505.24402","n_code_links":0,"syntology":null},{"paper":"/paper/mamba-knockout-for-unraveling-factual","title":"Mamba Knockout for Unraveling Factual Information Flow","date":"2025-05-30","arxiv_id":"2505.24244","n_code_links":1,"syntology":{"ran":2,"of":6,"unverified":4,"pointer_only":0}},{"paper":"/paper/mastering-massive-multi-task-reinforcement","title":"Mastering Massive Multi-Task Reinforcement Learning via Mixture-of-Expert Decision Transformer","date":"2025-05-30","arxiv_id":"2505.24378","n_code_links":1,"syntology":null},{"paper":null,"title":"PCIE_Pose Solution for EgoExo4D Pose and Proficiency Estimation Challenge","date":"2025-05-30","arxiv_id":"2505.24411","n_code_links":0,"syntology":null},{"paper":null,"title":"PersianMedQA: Language-Centric Evaluation of LLMs in the Persian Medical Domain","date":"2025-05-30","arxiv_id":"2506.00250","n_code_links":0,"syntology":null},{"paper":null,"title":"SPPSFormer: High-quality Superpoint-based Transformer for Roof Plane Instance Segmentation from Point Clouds","date":"2025-05-30","arxiv_id":"2505.24475","n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":1235},{"task":"/task/decoder","name":"Decoder","papers":1067},{"task":"/task/language-modeling","name":"Language Modeling","papers":946},{"task":"/task/translation","name":"Translation","papers":757},{"task":"/task/machine-translation","name":"Machine Translation","papers":717},{"task":"/task/semantic-segmentation","name":"Semantic Segmentation","papers":685},{"task":"/task/object-detection","name":"Object Detection","papers":548},{"task":"/task/image-classification","name":"Image Classification","papers":516},{"task":"/task/question-answering","name":"Question Answering","papers":510},{"task":"/task/object-detection-1","name":"object-detection","papers":494},{"task":"/task/segmentation","name":"Segmentation","papers":464},{"task":"/task/retrieval","name":"Retrieval","papers":453},{"task":"/task/sentence","name":"Sentence","papers":452},{"task":"/task/representation-learning","name":"Representation Learning","papers":426},{"task":"/task/image-classification","name":"image-classification","papers":410},{"task":"/task/large-language-model","name":"Large Language Model","papers":406},{"task":"/task/time-series-1","name":"Time Series","papers":368},{"task":"/task/object","name":"Object","papers":348},{"task":"/task/text-generation","name":"Text Generation","papers":315},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":297}],"tasks_shown":20,"n_tasks":2147,"usage_by_year":[{"year":"2017","papers":21},{"year":"2018","papers":115},{"year":"2019","papers":510},{"year":"2020","papers":832},{"year":"2021","papers":1496},{"year":"2022","papers":1886},{"year":"2023","papers":3242},{"year":"2024","papers":4393},{"year":"2025","papers":1400}],"row_source":"embedded","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/position-wise-feed-forward-layer"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}