{"url":"/method/location-based-attention","slug":"location-based-attention","name":"Location-based Attention","full_name":"Location-based Attention","full_name_withheld":false,"description_markdown":"**Location-based Attention** is an attention mechanism in which the alignment scores are computed from solely the target hidden state $\\mathbf{h}\\_{t}$ as follows:\r\n\r\n$$ \\mathbf{a}\\_{t} = \\text{softmax}(\\mathbf{W}\\_{a}\\mathbf{h}_{t}) $$","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":"http://arxiv.org/abs/1508.04025v5","title":"Effective Approaches to Attention-based Neural Machine Translation","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Attention Mechanisms","url":"/methods/category/attention-mechanisms","pwc_aliases":["attention-mechanisms-1"]}],"n_papers_tagged":37,"archive_num_papers":null,"papers_newest_first":[{"paper":null,"title":"Intelligent DoS and DDoS Detection: A Hybrid GRU-NTM Approach to Network Security","date":"2025-04-10","arxiv_id":"2504.07478","n_code_links":0,"syntology":null},{"paper":"/paper/cove-context-and-veracity-prediction-for-out","title":"COVE: COntext and VEracity prediction for out-of-context images","date":"2025-02-03","arxiv_id":"2502.01194","n_code_links":2,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}},{"paper":"/paper/on-the-role-of-surrogates-in-conformal","title":"On the Role of Surrogates in Conformal Inference of Individual Causal Effects","date":"2024-12-16","arxiv_id":"2412.12365","n_code_links":1,"syntology":null},{"paper":"/paper/sgseg-enabling-text-free-inference-in","title":"SGSeg: Enabling Text-free Inference in Language-guided Segmentation of Chest X-rays via Self-guidance","date":"2024-09-07","arxiv_id":"2409.04758","n_code_links":1,"syntology":null},{"paper":null,"title":"Crossfusor: A Cross-Attention Transformer Enhanced Conditional Diffusion Model for Car-Following Trajectory Prediction","date":"2024-06-17","arxiv_id":"2406.11941","n_code_links":0,"syntology":null},{"paper":"/paper/cove-unleashing-the-diffusion-feature","title":"COVE: Unleashing the Diffusion Feature Correspondence for Consistent Video Editing","date":"2024-06-13","arxiv_id":"2406.08850","n_code_links":1,"syntology":null},{"paper":"/paper/a-comprehensive-survey-of-hallucination","title":"A Comprehensive Survey of Hallucination Mitigation Techniques in Large Language Models","date":"2024-01-02","arxiv_id":"2401.01313","n_code_links":1,"syntology":{"ran":0,"of":1,"unverified":1,"pointer_only":0}},{"paper":"/paper/memory-augmented-conformer-for-improved-end","title":"Memory-augmented conformer for improved end-to-end long-form ASR","date":"2023-09-22","arxiv_id":"2309.13029","n_code_links":1,"syntology":null},{"paper":"/paper/chain-of-verification-reduces-hallucination","title":"Chain-of-Verification Reduces Hallucination in Large Language Models","date":"2023-09-20","arxiv_id":"2309.11495","n_code_links":1,"syntology":null},{"paper":"/paper/co-evolving-vector-quantization-for-id-based","title":"Learning Category Trees for ID-Based Recommendation: Exploring the Power of Differentiable Vector Quantization","date":"2023-08-31","arxiv_id":"2308.16761","n_code_links":2,"syntology":null},{"paper":null,"title":"FashionNTM: Multi-turn Fashion Image Retrieval via Cascaded Memory","date":"2023-08-20","arxiv_id":"2308.10170","n_code_links":0,"syntology":null},{"paper":"/paper/token-turing-machines","title":"Token Turing Machines","date":"2022-11-16","arxiv_id":"2211.09119","n_code_links":1,"syntology":null},{"paper":null,"title":"High Quality Streaming Speech Synthesis with Low, Sentence-Length-Independent Latency","date":"2021-11-17","arxiv_id":"2111.09052","n_code_links":0,"syntology":null},{"paper":null,"title":"Unsupervised Speaker Adaptation using Attention-based Speaker Memory for End-to-End ASR","date":"2020-02-14","arxiv_id":"2002.06165","n_code_links":0,"syntology":null},{"paper":null,"title":"Location Attention for Extrapolation to Longer Sequences","date":"2019-11-10","arxiv_id":"1911.03872","n_code_links":0,"syntology":null},{"paper":"/paper/attention-enriched-deep-learning-model-for","title":"Attention Enriched Deep Learning Model for Breast Tumor Segmentation in Ultrasound Images","date":"2019-10-20","arxiv_id":"1910.08978","n_code_links":1,"syntology":null},{"paper":null,"title":"Memory-Augmented Recurrent Networks for Dialogue Coherence","date":"2019-10-16","arxiv_id":"1910.10487","n_code_links":0,"syntology":null},{"paper":null,"title":"A Neural Turing~Machine for Conditional Transition Graph Modeling","date":"2019-07-15","arxiv_id":"1907.06432","n_code_links":0,"syntology":null},{"paper":null,"title":"Understanding Memory Modules on Learning Simple Algorithms","date":"2019-07-01","arxiv_id":"1907.00820","n_code_links":0,"syntology":null},{"paper":null,"title":"A review on Neural Turing Machine","date":"2019-04-10","arxiv_id":"1904.05061","n_code_links":0,"syntology":null},{"paper":"/paper/fastfusionnet-new-state-of-the-art-for","title":"FastFusionNet: New State-of-the-Art for DAWNBench SQuAD","date":"2019-02-28","arxiv_id":"1902.11291","n_code_links":2,"syntology":null},{"paper":"/paper/few-shot-generalization-across-dialogue-tasks","title":"Few-Shot Generalization Across Dialogue Tasks","date":"2018-11-28","arxiv_id":"1811.11707","n_code_links":2,"syntology":null},{"paper":null,"title":"Language Modeling Teaches You More than Translation Does: Lessons Learned Through Auxiliary Syntactic Task Analysis","date":"2018-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Language Modeling Teaches You More Syntax than Translation Does: Lessons Learned Through Auxiliary Task Analysis","date":"2018-09-26","arxiv_id":"1809.10040","n_code_links":0,"syntology":null},{"paper":null,"title":"Improving Matching Models with Hierarchical Contextualized Representations for Multi-turn Response Selection","date":"2018-08-22","arxiv_id":"1808.07244","n_code_links":0,"syntology":null},{"paper":null,"title":"Context-Aware Neural Model for Temporal Information Extraction","date":"2018-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Jiangnan at SemEval-2018 Task 11: Deep Neural Network with Attention Method for Machine Comprehension Task","date":"2018-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"A Taxonomy for Neural Memory Networks","date":"2018-05-01","arxiv_id":"1805.00327","n_code_links":0,"syntology":null},{"paper":null,"title":"Meta-Learning via Feature-Label Memory Network","date":"2017-10-19","arxiv_id":"1710.07110","n_code_links":0,"syntology":null},{"paper":"/paper/learned-in-translation-contextualized-word","title":"Learned in Translation: Contextualized Word Vectors","date":"2017-08-01","arxiv_id":"1708.00107","n_code_links":5,"syntology":{"ran":3,"of":3,"unverified":0,"pointer_only":2}}],"papers_shown":30,"tasks":[{"task":"/task/question-answering","name":"Question Answering","papers":6},{"task":"/task/retrieval","name":"Retrieval","papers":5},{"task":"/task/translation","name":"Translation","papers":5},{"task":"/task/decoder","name":"Decoder","papers":4},{"task":"/task/machine-translation","name":"Machine Translation","papers":4},{"task":"/task/language-modeling","name":"Language Modeling","papers":3},{"task":"/task/language-modelling","name":"Language Modelling","papers":3},{"task":"/task/sentence","name":"Sentence","papers":3},{"task":"/task/automatic-speech-recognition-2","name":"Automatic Speech Recognition","papers":2},{"task":"/task/automatic-speech-recognition","name":"Automatic Speech Recognition (ASR)","papers":2},{"task":"/task/machine-learning","name":"BIG-bench Machine Learning","papers":2},{"task":"/task/denoising","name":"Denoising","papers":2},{"task":"/task/hallucination","name":"Hallucination","papers":2},{"task":"/task/information-retrieval","name":"Information Retrieval","papers":2},{"task":"/task/reading-comprehension","name":"Reading Comprehension","papers":2},{"task":"/task/segmentation","name":"Segmentation","papers":2},{"task":"/task/speech-recognition","name":"Speech Recognition","papers":2},{"task":"/task/text-generation","name":"Text Generation","papers":2},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":2},{"task":"/task/speech-recognition-1","name":"speech-recognition","papers":2}],"tasks_shown":20,"n_tasks":74,"usage_by_year":[{"year":"2015","papers":2},{"year":"2016","papers":4},{"year":"2017","papers":3},{"year":"2018","papers":7},{"year":"2019","papers":7},{"year":"2020","papers":1},{"year":"2021","papers":1},{"year":"2022","papers":1},{"year":"2023","papers":4},{"year":"2024","papers":5},{"year":"2025","papers":2}],"row_source":"embedded","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/location-based-attention"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}